{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-26T04:43:31.130411Z","iopub.execute_input":"2022-04-26T04:43:31.130782Z","iopub.status.idle":"2022-04-26T04:43:31.154290Z","shell.execute_reply.started":"2022-04-26T04:43:31.130664Z","shell.execute_reply":"2022-04-26T04:43:31.153589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Models from the following notebooks are used to create an ensemble\n* LB: 0.0217 - https://www.kaggle.com/tarique7/hnm-exponential-decay-with-alternate-items/notebook\n* LB: 0.0220 - https://www.kaggle.com/code/hengzheng/time-is-our-best-friend-v2/notebook\n* LB: 0.0221 - https://www.kaggle.com/astrung/lstm-sequential-modelwith-item-features-tutorial\n* LB: 0.0224 - https://www.kaggle.com/code/hirotakanogami/h-m-eda-customer-clustering-by-kmeans\n* LB: 0.0225 - https://www.kaggle.com/lunapandachan/h-m-trending-products-weekly-add-test/notebook\n* LB: 0.0227 - https://www.kaggle.com/code/hechtjp/h-m-eda-rule-base-by-customer-age\n* LB: 0.0231 - https://www.kaggle.com/code/ebn7amdi/trending/notebook?scriptVersionId=90980162","metadata":{}},{"cell_type":"code","source":"sub2 = pd.read_csv('../input/handmbestperforming/hnm-exponential-decay-with-alternate-items.csv').sort_values('customer_id').reset_index(drop=True)\nsub5 = pd.read_csv('../input/handmbestperforming/time-is-our-best-friend-v2.csv').sort_values('customer_id').reset_index(drop=True)\nsub3 = pd.read_csv('../input/handmbestperforming/lstm-sequential-modelwith-item-features-tutorial.csv').sort_values('customer_id').reset_index(drop=True)\nsub4 = pd.read_csv('../input/hm-00224-solution/submission.csv').sort_values('customer_id').reset_index(drop=True)\n\nsub0 = pd.read_csv('../input/hm-00231-solution/submission.csv').sort_values('customer_id').reset_index(drop=True)\nsub1 = pd.read_csv('../input/handmbestperforming/h-m-trending-products-weekly-add-test.csv').sort_values('customer_id').reset_index(drop=True)\nsub6 = pd.read_csv('../input/handmbestperforming/rule-based-by-customer-age.csv').sort_values('customer_id').reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-04-26T04:43:52.053501Z","iopub.execute_input":"2022-04-26T04:43:52.053799Z","iopub.status.idle":"2022-04-26T04:44:38.869574Z","shell.execute_reply.started":"2022-04-26T04:43:52.053769Z","shell.execute_reply":"2022-04-26T04:44:38.868749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub0.columns = ['customer_id', 'prediction0']\nsub0['prediction1'] = sub1['prediction']\nsub0['prediction2'] = sub2['prediction']\nsub0['prediction3'] = sub3['prediction']\nsub0['prediction4'] = sub4['prediction']\nsub0['prediction5'] = sub5['prediction']\nsub0['prediction6'] = sub6['prediction']\n\ndel sub1, sub2, sub3, sub4, sub5, sub6\ngc.collect()\nsub0.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-26T04:45:10.578109Z","iopub.execute_input":"2022-04-26T04:45:10.578376Z","iopub.status.idle":"2022-04-26T04:45:11.188364Z","shell.execute_reply.started":"2022-04-26T04:45:10.578340Z","shell.execute_reply":"2022-04-26T04:45:11.187555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cust_blend2(dt, weights):\n    REC = []\n    REC.append(dt['prediction0'].str.split())\n    REC.append(dt['prediction1'].str.split())\n    REC.append(dt['prediction2'].str.split())\n    REC.append(dt['prediction3'].str.split())\n    REC.append(dt['prediction4'].str.split())\n    REC.append(dt['prediction5'].str.split())\n    REC.append(dt['prediction6'].str.split())\n    \n    res = {}\n    W = weights\n    \n    for M in range(len(REC)):\n        for custom_cnt, rec_list in enumerate(REC[M]):\n            if custom_cnt not in res:\n                res[custom_cnt]={}\n                for idx, item in enumerate(rec_list):\n                    if item in res[custom_cnt]:\n                        res[custom_cnt][item] += W[M]/(idx+1)\n                    else:\n                        res[custom_cnt][item] = W[M]/(idx+1)\n    \n    for custom_cnt in range(len(res)):\n        res[custom_cnt] = ' '.join(list(dict(sorted(res[custom_cnt].items(), key=lambda item: -item[1])).keys())[:12])\n    \n    return res\n#     return pd.DataFrame.from_dict(res)","metadata":{"execution":{"iopub.status.busy":"2022-04-26T04:45:15.421156Z","iopub.execute_input":"2022-04-26T04:45:15.422175Z","iopub.status.idle":"2022-04-26T04:45:15.436493Z","shell.execute_reply.started":"2022-04-26T04:45:15.422123Z","shell.execute_reply":"2022-04-26T04:45:15.435364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"w = np.array([1.05,1.00,0.95,0.85,0.75,0.95,0.55]) \nresult = cust_blend2(sub0, w)","metadata":{"execution":{"iopub.status.busy":"2022-04-26T04:45:20.878074Z","iopub.execute_input":"2022-04-26T04:45:20.878872Z","iopub.status.idle":"2022-04-26T04:46:10.758332Z","shell.execute_reply.started":"2022-04-26T04:45:20.878829Z","shell.execute_reply":"2022-04-26T04:46:10.757441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import operator\nfor k,v in sorted(result.items(), key=operator.itemgetter(1))[:50]:\n    print (k,v)","metadata":{"execution":{"iopub.status.busy":"2022-04-26T04:49:58.123079Z","iopub.execute_input":"2022-04-26T04:49:58.123601Z","iopub.status.idle":"2022-04-26T04:49:59.689450Z","shell.execute_reply.started":"2022-04-26T04:49:58.123569Z","shell.execute_reply":"2022-04-26T04:49:59.688552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub0['prediction'] = np.array(list(result.values()))\n\nsub0.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-26T04:49:17.268671Z","iopub.execute_input":"2022-04-26T04:49:17.268997Z","iopub.status.idle":"2022-04-26T04:49:18.996553Z","shell.execute_reply.started":"2022-04-26T04:49:17.268965Z","shell.execute_reply":"2022-04-26T04:49:18.995604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make a submission","metadata":{}},{"cell_type":"code","source":"del sub0['prediction0']\ndel sub0['prediction1']\ndel sub0['prediction2']\ndel sub0['prediction3']\ndel sub0['prediction4']\ndel sub0['prediction5']\ndel sub0['prediction6']\ngc.collect()\n\n\nsub0.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-04-24T03:53:36.398281Z","iopub.execute_input":"2022-04-24T03:53:36.398597Z","iopub.status.idle":"2022-04-24T03:53:49.089221Z","shell.execute_reply.started":"2022-04-24T03:53:36.398563Z","shell.execute_reply":"2022-04-24T03:53:49.088463Z"},"trusted":true},"execution_count":null,"outputs":[]}]}