{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-08T19:27:32.305074Z","iopub.execute_input":"2022-03-08T19:27:32.305672Z","iopub.status.idle":"2022-03-08T19:27:32.339921Z","shell.execute_reply.started":"2022-03-08T19:27:32.305576Z","shell.execute_reply":"2022-03-08T19:27:32.339177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls ../input/hm-public-submissions","metadata":{"execution":{"iopub.status.busy":"2022-03-08T19:27:34.345283Z","iopub.execute_input":"2022-03-08T19:27:34.345590Z","iopub.status.idle":"2022-03-08T19:27:35.120390Z","shell.execute_reply.started":"2022-03-08T19:27:34.345555Z","shell.execute_reply":"2022-03-08T19:27:35.119101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predictions in this competition are a list of 12 itens ordered by most relevant first.\n# In this notebook I will show how to ensemble lists of different models\n# To ensemble I used submissions from 3 public notebooks:\n- LB: 0.0225 - https://www.kaggle.com/lichtlab/0-0226-byfone-chris-combination-approach/data?scriptVersionId=89289696\n- LB: 0.0225 - https://www.kaggle.com/lunapandachan/h-m-trending-products-weekly-add-test/notebook\n- LB: 0.0217 - https://www.kaggle.com/tarique7/hnm-exponential-decay-with-alternate-items/notebook","metadata":{}},{"cell_type":"code","source":"sub0 = pd.read_csv('../input/hm-public-submissions/0-0226-byfone-chris-combination-approach.csv').sort_values('customer_id').reset_index(drop=True)\nsub1 = pd.read_csv('../input/hm-public-submissions/h-m-trending-products-weekly-add-test.csv').sort_values('customer_id').reset_index(drop=True)\nsub2 = pd.read_csv('../input/hm-public-submissions/hnm-exponential-decay-with-alternate-items.csv').sort_values('customer_id').reset_index(drop=True)\n\nsub0.shape, sub1.shape, sub2.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-08T19:35:53.310561Z","iopub.execute_input":"2022-03-08T19:35:53.311318Z","iopub.status.idle":"2022-03-08T19:36:10.914478Z","shell.execute_reply.started":"2022-03-08T19:35:53.311269Z","shell.execute_reply":"2022-03-08T19:36:10.913403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How many predictions are in common between models\n\nprint((sub0['prediction']==sub1['prediction']).mean())\nprint((sub0['prediction']==sub2['prediction']).mean())\nprint((sub1['prediction']==sub2['prediction']).mean())","metadata":{"execution":{"iopub.status.busy":"2022-03-08T19:36:10.916312Z","iopub.execute_input":"2022-03-08T19:36:10.916569Z","iopub.status.idle":"2022-03-08T19:36:11.957167Z","shell.execute_reply.started":"2022-03-08T19:36:10.916540Z","shell.execute_reply":"2022-03-08T19:36:11.956196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print( sub0.head() )\nprint()\nprint( sub1.head() )\nprint()\nprint( sub2.head() )","metadata":{"execution":{"iopub.status.busy":"2022-03-08T19:36:11.958393Z","iopub.execute_input":"2022-03-08T19:36:11.958629Z","iopub.status.idle":"2022-03-08T19:36:11.972245Z","shell.execute_reply.started":"2022-03-08T19:36:11.958600Z","shell.execute_reply":"2022-03-08T19:36:11.971133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub0.columns = ['customer_id', 'prediction0']\nsub0['prediction1'] = sub1['prediction']\nsub0['prediction2'] = sub2['prediction']\ndel sub1, sub2\nsub0.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-08T19:36:11.973800Z","iopub.execute_input":"2022-03-08T19:36:11.974081Z","iopub.status.idle":"2022-03-08T19:36:12.155252Z","shell.execute_reply.started":"2022-03-08T19:36:11.974043Z","shell.execute_reply":"2022-03-08T19:36:12.154208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cust_blend(dt, W = [1,1,1]):\n    #Global ensemble weights\n    #W = [1.15,0.95,0.85]\n    \n    #Create a list of all model predictions\n    REC = []\n    REC.append(dt['prediction0'].split())\n    REC.append(dt['prediction1'].split())\n    REC.append(dt['prediction2'].split())\n    \n    #Create a dictionary of items recommended. \n    #Assign a weight according the order of appearance and multiply by global weights\n    res = {}\n    for M in range(len(REC)):\n        for n, v in enumerate(REC[M]):\n            if v in res:\n                res[v] += (W[M]/(n+1))\n            else:\n                res[v] = (W[M]/(n+1))\n    \n    # Sort dictionary by item weights\n    res = list(dict(sorted(res.items(), key=lambda item: -item[1])).keys())\n    \n    # Return the top 12 itens only\n    return ' '.join(res[:12])\n\nsub0['prediction'] = sub0.apply(cust_blend, W = [1.05,1.00,0.95], axis=1)\nsub0.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-08T19:36:38.156615Z","iopub.execute_input":"2022-03-08T19:36:38.156926Z","iopub.status.idle":"2022-03-08T19:37:46.491477Z","shell.execute_reply.started":"2022-03-08T19:36:38.156892Z","shell.execute_reply":"2022-03-08T19:37:46.490581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How many predictions are in common with ensemble\n\nprint((sub0['prediction']==sub0['prediction0']).mean())\nprint((sub0['prediction']==sub0['prediction1']).mean())\nprint((sub0['prediction']==sub0['prediction2']).mean())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make a submission","metadata":{}},{"cell_type":"code","source":"del sub0['prediction0']\ndel sub0['prediction1']\ndel sub0['prediction2']\nsub0.to_csv('submission-blend-1.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}