{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# !pip install pyarrow\n# !pip install fastparquet","metadata":{"execution":{"iopub.status.busy":"2022-05-04T21:09:59.738077Z","iopub.execute_input":"2022-05-04T21:09:59.73869Z","iopub.status.idle":"2022-05-04T21:09:59.760897Z","shell.execute_reply.started":"2022-05-04T21:09:59.738575Z","shell.execute_reply":"2022-05-04T21:09:59.76023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np, pandas as pd, datetime as dt\nimport matplotlib.pyplot as plt; plt.style.use('ggplot')\nimport seaborn as sns\nfrom collections import defaultdict\n\ndef iter_to_str(iterable):\n    return \" \".join(map(lambda x: str(0) + str(x), iterable))\n\ndef apk(actual, predicted, k=12):\n    if len(predicted) > k:\n        predicted = predicted[:k]\n    score, nhits = 0.0, 0.0\n    for i, p in enumerate(predicted):\n        if p in actual and p not in predicted[:i]:\n            nhits += 1.0\n            score += nhits / (i + 1.0)\n    if not actual:\n        return 0.0\n    return score / min(len(actual), k)\n\ndef mapk(actual, predicted, k=12, return_apks=False):\n    assert len(actual) == len(predicted)\n    apks = [apk(ac, pr, k) for ac, pr in zip(actual, predicted) if 0 < len(ac)]\n    if return_apks:\n        return apks\n    return np.mean(apks)\n\ndef blend(dt, w=[], k=12):\n    if len(w) == 0:\n        w = [1] * (len(dt))\n    preds = []\n    for i in range(len(w)):\n        preds.append(dt[i].split())\n    res = {}\n    for i in range(len(preds)):\n        if w[i] < 0:\n            continue\n        for n, v in enumerate(preds[i]):\n            if v in res:\n                res[v] += (w[i] / (n + 1))\n            else:\n                res[v] = (w[i] / (n + 1))    \n    res = list(dict(sorted(res.items(), key=lambda item: -item[1])).keys())\n    return ' '.join(res[:k])\n\ndef prune(pred, ok_set, k=12):\n    pred = pred.split()\n    post = []\n    for item in pred:\n        if int(item) in ok_set and not item in post:\n            post.append(item)\n    return \" \".join(post[:k])\n\ndef validation(actual, predicted, grouping, score=0, index=-1, ignore=False, figsize=(12, 6)):\n    # actual, predicted : list of lists\n    # group : pandas Series\n    # score : pandas DataFrame\n    \n    vc = pd.Series(predicted).apply(len).value_counts()\n    print(\"Fill Rate = \", round(1 - sum(vc[k] * (12 - k) / 12 for k in (set(range(12)) & set(vc.index))) / len(actual), 3) * 100)\n    \n    \n    if ignore: return\n    ap12 = mapk(actual, predicted, return_apks=True)\n    map12 = round(np.mean(ap12), 6)\n    if isinstance(score, int): score = pd.DataFrame({g:[] for g in sorted(grouping.unique().tolist())})\n    if index == -1 : index = score.shape[0]\n    score.loc[index, \"All\"] = map12\n    plt.figure(figsize=figsize)\n    plt.subplot(1, 2, 1); sns.histplot(data=ap12, log_scale=(0, 10), bins=20); plt.title(f\"MAP@12 : {map12}\")\n    for g in grouping.unique():\n        map12 = round(mapk(actual[grouping == g], predicted[grouping == g]), 6)\n        score.loc[index, g] = map12\n    plt.subplot(1, 2, 2); score[[g for g in grouping.unique()[::-1]] + ['All']].loc[index].plot.barh(); plt.title(f\"MAP@12 of Groups\")\n    vc = pd.Series(predicted).apply(len).value_counts()\n    score.loc[index, \"Fill\"] = round(1 - sum(vc[k] * (12 - k) / 12 for k in (set(range(12)) & set(vc.index))) / len(actual), 3) * 100\n    display(score)\n    return score","metadata":{"execution":{"iopub.status.busy":"2022-05-09T13:13:34.306819Z","iopub.execute_input":"2022-05-09T13:13:34.307438Z","iopub.status.idle":"2022-05-09T13:13:35.485688Z","shell.execute_reply.started":"2022-05-09T13:13:34.30732Z","shell.execute_reply":"2022-05-09T13:13:35.484756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:38:11.51053Z","iopub.execute_input":"2022-05-06T13:38:11.510825Z","iopub.status.idle":"2022-05-06T13:38:17.234391Z","shell.execute_reply.started":"2022-05-06T13:38:11.510791Z","shell.execute_reply":"2022-05-06T13:38:17.233501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub0 = pd.read_csv(\"../input/hm-for-ensemble/submission_uucf0252.csv\")\nsub['sub0'] = sub0['prediction'].fillna(\"\")\ndel sub0","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:38:32.40108Z","iopub.execute_input":"2022-05-06T13:38:32.401726Z","iopub.status.idle":"2022-05-06T13:38:39.82451Z","shell.execute_reply.started":"2022-05-06T13:38:32.401683Z","shell.execute_reply":"2022-05-06T13:38:39.823653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub1 = pd.read_csv(\"../input/hm-for-ensemble/submission-blend-255.csv\")\nsub['sub1'] = sub1['prediction'].fillna(\"\")\ndel sub1","metadata":{"execution":{"iopub.status.busy":"2022-05-09T13:13:56.276089Z","iopub.execute_input":"2022-05-09T13:13:56.276645Z","iopub.status.idle":"2022-05-09T13:14:02.853845Z","shell.execute_reply.started":"2022-05-09T13:13:56.276604Z","shell.execute_reply":"2022-05-09T13:14:02.852718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub2 = pd.read_csv(\"../input/hm-for-ensemble/submission-magic-multi-brend-0240.csv\")\nsub['sub2'] = sub2['prediction'].fillna(\"\")\ndel sub2","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:38:47.18721Z","iopub.execute_input":"2022-05-06T13:38:47.187565Z","iopub.status.idle":"2022-05-06T13:38:53.782128Z","shell.execute_reply.started":"2022-05-06T13:38:47.187518Z","shell.execute_reply":"2022-05-06T13:38:53.781331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub3 = pd.read_csv(\"../input/handmsubmitfiles/LGBM_Ranker_submission_229.csv\")\nsub['sub3'] = sub3['prediction'].fillna(\"\")\ndel sub3","metadata":{"execution":{"iopub.status.busy":"2022-05-04T21:10:20.14463Z","iopub.execute_input":"2022-05-04T21:10:20.145187Z","iopub.status.idle":"2022-05-04T21:10:20.155176Z","shell.execute_reply.started":"2022-05-04T21:10:20.145155Z","shell.execute_reply":"2022-05-04T21:10:20.154618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#lstm\nsub4 = pd.read_csv(\"../input/lstm-model-with-item-infor-fix-missing-last-item/submission.csv\")\nsub['sub4'] = sub4['prediction'].fillna(\"\")\ndel sub4","metadata":{"execution":{"iopub.status.busy":"2022-05-04T21:10:20.156249Z","iopub.execute_input":"2022-05-04T21:10:20.156721Z","iopub.status.idle":"2022-05-04T21:10:20.166339Z","shell.execute_reply.started":"2022-05-04T21:10:20.156688Z","shell.execute_reply":"2022-05-04T21:10:20.165816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#collabo\nsub5 = pd.read_csv(\"../input/handmsubmitfiles/colabo_uucf_only_ver6.csv\")\nsub['sub5'] = sub5['prediction'].fillna(\"\")\ndel sub5","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#agegen\nsub6 = pd.read_csv(\"../input/h-m-easy-grouping-by-sex-attribute-age-en-jp/submission.csv\")\nsub['sub6'] = sub6['prediction'].fillna(\"\")\ndel sub6","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-05-04T21:10:20.168149Z","iopub.execute_input":"2022-05-04T21:10:20.168621Z","iopub.status.idle":"2022-05-04T21:10:20.190452Z","shell.execute_reply.started":"2022-05-04T21:10:20.168586Z","shell.execute_reply":"2022-05-04T21:10:20.18961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = ['sub0', 'sub1', 'sub2', 'sub3','sub4','sub5', 'sub6']\n#targets = ['sub0', 'sub1', 'sub2', 'sub3']\nweights = [1.6, 1, 0.20, 0.20, 0.10,0,0.10]# 0.0252 0.0255 0.0240 0.0229\n\n\n# 0.0255↓\n# targets = ['sub0', 'sub1', 'sub2', 'sub3', 'sub4' ]\n# weights = [1,1.1,0.8,0.7,0.7]\nsub['prediction'] = sub[targets].apply(blend, w=weights, axis=1, k=12)","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:39:00.41778Z","iopub.execute_input":"2022-05-06T13:39:00.418114Z","iopub.status.idle":"2022-05-06T13:39:59.404433Z","shell.execute_reply.started":"2022-05-06T13:39:00.418075Z","shell.execute_reply":"2022-05-06T13:39:59.403449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:39:59.405503Z","iopub.status.idle":"2022-05-06T13:39:59.405931Z","shell.execute_reply.started":"2022-05-06T13:39:59.405756Z","shell.execute_reply":"2022-05-06T13:39:59.405775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub[['customer_id', 'prediction']].to_csv('submission_ensamble.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-05-04T21:11:06.634527Z","iopub.execute_input":"2022-05-04T21:11:06.634746Z","iopub.status.idle":"2022-05-04T21:11:19.441407Z","shell.execute_reply.started":"2022-05-04T21:11:06.6347Z","shell.execute_reply":"2022-05-04T21:11:19.44063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}