{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-20T10:36:58.830142Z","iopub.execute_input":"2022-03-20T10:36:58.830444Z","iopub.status.idle":"2022-03-20T10:36:58.835462Z","shell.execute_reply.started":"2022-03-20T10:36:58.830413Z","shell.execute_reply":"2022-03-20T10:36:58.834562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!ls ../input/hm-public-submissions","metadata":{"execution":{"iopub.status.busy":"2022-03-20T10:36:58.837433Z","iopub.execute_input":"2022-03-20T10:36:58.837919Z","iopub.status.idle":"2022-03-20T10:36:58.848898Z","shell.execute_reply.started":"2022-03-20T10:36:58.837870Z","shell.execute_reply":"2022-03-20T10:36:58.847885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Adding LSTM models to the ensemble**","metadata":{}},{"cell_type":"markdown","source":"# Note\n* I am not sure whether I should get any credit for this notebook as this is based(mostly) on the work of others.\n* I added two more submissions to the ensemble and tried a few options with them.","metadata":{}},{"cell_type":"markdown","source":"# New Notebooks\n* LB: 0.0210 - https://www.kaggle.com/astrung/recbole-lstm-sequential-for-recomendation-tutorial\n* LB: 0.0221 - https://www.kaggle.com/astrung/lstm-sequential-modelwith-item-features-tutorial\n* both notebooks by @astrung","metadata":{}},{"cell_type":"markdown","source":"# Predictions in this competition are a list of 12 itens ordered by most relevant first.\n# In this notebook I will show how to ensemble lists of different models\n# To ensemble I used submissions from 3 public notebooks:\n- LB: 0.0225 - https://www.kaggle.com/lichtlab/0-0226-byfone-chris-combination-approach/data?scriptVersionId=89289696\n- LB: 0.0225 - https://www.kaggle.com/lunapandachan/h-m-trending-products-weekly-add-test/notebook\n- LB: 0.0217 - https://www.kaggle.com/tarique7/hnm-exponential-decay-with-alternate-items/notebook","metadata":{}},{"cell_type":"code","source":"sub0 = pd.read_csv('../input/hm-public-submissions/0-0226-byfone-chris-combination-approach.csv').sort_values('customer_id').reset_index(drop=True)\nsub1 = pd.read_csv('../input/hm-public-submissions/h-m-trending-products-weekly-add-test.csv').sort_values('customer_id').reset_index(drop=True)\nsub2 = pd.read_csv('../input/hm-public-submissions/hnm-exponential-decay-with-alternate-items.csv').sort_values('customer_id').reset_index(drop=True)\n\nsub0.shape, sub1.shape, sub2.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-20T10:36:58.850653Z","iopub.execute_input":"2022-03-20T10:36:58.851266Z","iopub.status.idle":"2022-03-20T10:37:15.329804Z","shell.execute_reply.started":"2022-03-20T10:36:58.851229Z","shell.execute_reply":"2022-03-20T10:37:15.329006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub3 = pd.read_csv('../input/submission-recbole-lstm/submission.csv').sort_values('customer_id').reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-20T10:37:15.331072Z","iopub.execute_input":"2022-03-20T10:37:15.331672Z","iopub.status.idle":"2022-03-20T10:37:21.141795Z","shell.execute_reply.started":"2022-03-20T10:37:15.331639Z","shell.execute_reply":"2022-03-20T10:37:21.140815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub4 = pd.read_csv('../input/submission-lstm-sequential/submission (1).csv').sort_values('customer_id').reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-20T10:37:21.143891Z","iopub.execute_input":"2022-03-20T10:37:21.144135Z","iopub.status.idle":"2022-03-20T10:37:26.943367Z","shell.execute_reply.started":"2022-03-20T10:37:21.144104Z","shell.execute_reply":"2022-03-20T10:37:26.942402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/code/astrung/sequential-model-fixed-missing-last-item/notebook\nsub5 = pd.read_csv('../input/lstm-sequential-modelwith-item-features-tutorial/submission.csv').sort_values('customer_id').reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-20T10:37:26.944654Z","iopub.execute_input":"2022-03-20T10:37:26.944895Z","iopub.status.idle":"2022-03-20T10:37:32.701099Z","shell.execute_reply.started":"2022-03-20T10:37:26.944869Z","shell.execute_reply":"2022-03-20T10:37:32.700011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/code/astrung/lstm-model-with-item-infor-fix-missing-last-item/notebook\nsub6 = pd.read_csv('../input/lstm-model-with-item-infor-fix-missing-last-item/submission.csv').sort_values('customer_id').reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-20T10:37:32.702307Z","iopub.execute_input":"2022-03-20T10:37:32.702539Z","iopub.status.idle":"2022-03-20T10:37:38.486182Z","shell.execute_reply.started":"2022-03-20T10:37:32.702513Z","shell.execute_reply":"2022-03-20T10:37:38.485528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How many predictions are in common between models\n\nprint((sub0['prediction']==sub1['prediction']).mean())\nprint((sub0['prediction']==sub2['prediction']).mean())\nprint((sub1['prediction']==sub2['prediction']).mean())","metadata":{"execution":{"iopub.status.busy":"2022-03-20T10:37:38.487143Z","iopub.execute_input":"2022-03-20T10:37:38.487744Z","iopub.status.idle":"2022-03-20T10:37:39.492127Z","shell.execute_reply.started":"2022-03-20T10:37:38.487712Z","shell.execute_reply":"2022-03-20T10:37:39.491199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How many predictions are in common between old models and new \n\nprint((sub3['prediction']==sub0['prediction']).mean())\nprint((sub3['prediction']==sub1['prediction']).mean())\nprint((sub3['prediction']==sub2['prediction']).mean())","metadata":{"execution":{"iopub.status.busy":"2022-03-20T10:37:39.493322Z","iopub.execute_input":"2022-03-20T10:37:39.493545Z","iopub.status.idle":"2022-03-20T10:37:40.496212Z","shell.execute_reply.started":"2022-03-20T10:37:39.493517Z","shell.execute_reply":"2022-03-20T10:37:40.495333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How many predictions are in common between old models and new\n\nprint((sub4['prediction']==sub0['prediction']).mean())\nprint((sub4['prediction']==sub1['prediction']).mean())\nprint((sub4['prediction']==sub2['prediction']).mean())\nprint((sub4['prediction']==sub3['prediction']).mean())","metadata":{"execution":{"iopub.status.busy":"2022-03-20T10:37:40.497647Z","iopub.execute_input":"2022-03-20T10:37:40.498050Z","iopub.status.idle":"2022-03-20T10:37:41.848651Z","shell.execute_reply.started":"2022-03-20T10:37:40.498003Z","shell.execute_reply":"2022-03-20T10:37:41.847282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How many predictions are in common between old models and new\n\nprint((sub5['prediction']==sub0['prediction']).mean())\nprint((sub5['prediction']==sub1['prediction']).mean())\nprint((sub5['prediction']==sub2['prediction']).mean())\nprint((sub5['prediction']==sub3['prediction']).mean())\nprint((sub5['prediction']==sub4['prediction']).mean())","metadata":{"execution":{"iopub.status.busy":"2022-03-20T10:37:41.851108Z","iopub.execute_input":"2022-03-20T10:37:41.851406Z","iopub.status.idle":"2022-03-20T10:37:43.552216Z","shell.execute_reply.started":"2022-03-20T10:37:41.851370Z","shell.execute_reply":"2022-03-20T10:37:43.551315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How many predictions are in common between old models and new\n\nprint((sub6['prediction']==sub0['prediction']).mean())\nprint((sub6['prediction']==sub1['prediction']).mean())\nprint((sub6['prediction']==sub2['prediction']).mean())\nprint((sub6['prediction']==sub3['prediction']).mean())\nprint((sub6['prediction']==sub4['prediction']).mean())\nprint((sub6['prediction']==sub5['prediction']).mean())","metadata":{"execution":{"iopub.status.busy":"2022-03-20T10:37:43.553417Z","iopub.execute_input":"2022-03-20T10:37:43.553629Z","iopub.status.idle":"2022-03-20T10:37:45.693694Z","shell.execute_reply.started":"2022-03-20T10:37:43.553604Z","shell.execute_reply":"2022-03-20T10:37:45.692851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub0.columns = ['customer_id', 'prediction0']\nsub0['prediction1'] = sub1['prediction']\nsub0['prediction2'] = sub2['prediction']\nsub0['prediction3'] = sub3['prediction']\nsub0['prediction4'] = sub4['prediction']\nsub0['prediction5'] = sub5['prediction']\nsub0['prediction6'] = sub6['prediction']\ndel sub1, sub2, sub3, sub4, sub5, sub6\ngc.collect()\nsub0.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-20T10:37:45.695160Z","iopub.execute_input":"2022-03-20T10:37:45.695423Z","iopub.status.idle":"2022-03-20T10:37:46.732933Z","shell.execute_reply.started":"2022-03-20T10:37:45.695382Z","shell.execute_reply":"2022-03-20T10:37:46.731965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Ensembling configuration used for version 1\ndef cust_blend_old(dt, W = [1,1,1,1]):\n    #Global ensemble weights\n    #W = [1.15,0.95,0.85]\n    \n    #Create a list of all model predictions\n    REC = []\n    \n    # Second Try\n    REC.append(dt['prediction0'].split())\n    REC.append(dt['prediction1'].split())\n    REC.append(dt['prediction2'].split())\n    REC.append(dt['prediction3'].split())\n    \n    #Create a dictionary of items recommended. \n    #Assign a weight according the order of appearance and multiply by global weights\n    res = {}\n    for M in range(len(REC)):\n        for n, v in enumerate(REC[M]):\n            if v in res:\n                res[v] += (W[M]/(n+1))\n            else:\n                res[v] = (W[M]/(n+1))\n    \n    # Sort dictionary by item weights\n    res = list(dict(sorted(res.items(), key=lambda item: -item[1])).keys())\n    \n    # Return the top 12 itens only\n    return ' '.join(res[:12])","metadata":{"execution":{"iopub.status.busy":"2022-03-20T10:37:46.733866Z","iopub.execute_input":"2022-03-20T10:37:46.734074Z","iopub.status.idle":"2022-03-20T10:37:46.746583Z","shell.execute_reply.started":"2022-03-20T10:37:46.734049Z","shell.execute_reply":"2022-03-20T10:37:46.745597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sub0['prediction'] = sub0.apply(cust_blend, W = [1.05,1.00,0.95,0.85], axis=1)\n#sub0.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-20T10:37:46.747815Z","iopub.execute_input":"2022-03-20T10:37:46.748353Z","iopub.status.idle":"2022-03-20T10:37:46.757784Z","shell.execute_reply.started":"2022-03-20T10:37:46.748311Z","shell.execute_reply":"2022-03-20T10:37:46.755499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How many predictions are in common with ensemble\n\n#print((sub0['prediction']==sub0['prediction0']).mean())\n#print((sub0['prediction']==sub0['prediction1']).mean())\n#print((sub0['prediction']==sub0['prediction2']).mean())\n#print((sub0['prediction']==sub0['prediction3']).mean())\n#print((sub0['prediction']==sub0['prediction4']).mean())","metadata":{"execution":{"iopub.status.busy":"2022-03-20T10:37:46.759149Z","iopub.execute_input":"2022-03-20T10:37:46.759955Z","iopub.status.idle":"2022-03-20T10:37:46.768458Z","shell.execute_reply.started":"2022-03-20T10:37:46.759912Z","shell.execute_reply":"2022-03-20T10:37:46.767599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The results of comparison seem interesting**","metadata":{}},{"cell_type":"code","source":"# Ensembling configuration used for current version\ndef cust_blend(dt, W = [1,1,1,1]):\n    #Global ensemble weights\n    #W = [1.15,0.95,0.85]\n    \n    #Create a list of all model predictions\n    REC = []\n    \n    # Second Try\n    REC.append(dt['prediction0'].split())\n    REC.append(dt['prediction1'].split())\n    REC.append(dt['prediction2'].split())\n    #REC.append(dt['prediction6'].split()) # change from version 1 to version 2 \n    REC.append(dt['prediction5'].split()) # change from version 1 to version 3\n    \n    #Create a dictionary of items recommended. \n    #Assign a weight according the order of appearance and multiply by global weights\n    res = {}\n    for M in range(len(REC)):\n        for n, v in enumerate(REC[M]):\n            if v in res:\n                res[v] += (W[M]/(n+1))\n            else:\n                res[v] = (W[M]/(n+1))\n    \n    # Sort dictionary by item weights\n    res = list(dict(sorted(res.items(), key=lambda item: -item[1])).keys())\n    \n    # Return the top 12 itens only\n    return ' '.join(res[:12])","metadata":{"execution":{"iopub.status.busy":"2022-03-20T10:37:46.769706Z","iopub.execute_input":"2022-03-20T10:37:46.770138Z","iopub.status.idle":"2022-03-20T10:37:46.783084Z","shell.execute_reply.started":"2022-03-20T10:37:46.770096Z","shell.execute_reply":"2022-03-20T10:37:46.782107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub0['prediction'] = sub0.apply(cust_blend, W = [1.05,1.00,0.95,0.85], axis=1)\nsub0.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-20T10:37:46.784530Z","iopub.execute_input":"2022-03-20T10:37:46.784950Z","iopub.status.idle":"2022-03-20T10:39:12.717148Z","shell.execute_reply.started":"2022-03-20T10:37:46.784909Z","shell.execute_reply":"2022-03-20T10:39:12.716207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How many predictions are in common with ensemble\n\nprint((sub0['prediction']==sub0['prediction0']).mean())\nprint((sub0['prediction']==sub0['prediction1']).mean())\nprint((sub0['prediction']==sub0['prediction2']).mean())\nprint((sub0['prediction']==sub0['prediction3']).mean())\nprint((sub0['prediction']==sub0['prediction4']).mean())\nprint((sub0['prediction']==sub0['prediction5']).mean())\nprint((sub0['prediction']==sub0['prediction6']).mean())","metadata":{"execution":{"iopub.status.busy":"2022-03-20T10:39:12.719003Z","iopub.execute_input":"2022-03-20T10:39:12.719372Z","iopub.status.idle":"2022-03-20T10:39:15.035948Z","shell.execute_reply.started":"2022-03-20T10:39:12.719330Z","shell.execute_reply":"2022-03-20T10:39:15.034929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make a submission","metadata":{}},{"cell_type":"code","source":"del sub0['prediction0']\ndel sub0['prediction1']\ndel sub0['prediction2']\ndel sub0['prediction3']\ndel sub0['prediction4']\ndel sub0['prediction5']\ndel sub0['prediction6']\ngc.collect()\nsub0.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Version 2 score - 0.0231 drop from V1\n# Version 3 score - ","metadata":{},"execution_count":null,"outputs":[]}]}