{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport gc\n\ndef cust_blend(dt, W = [1]):\n    #Create a list of all model predictions\n    REC = []\n\n    # Second Try\n    for i in range(len(dt)-1):\n        REC.append(dt[f'prediction{i}'].split())\n    \n    #Create a dictionary of items recommended.\n    #Assign a weight according the order of appearance and multiply by global weights\n    res = {}\n    for M in range(len(REC)):\n        for n, v in enumerate(REC[M]):\n            if v == '': continue\n            if v in res:\n                res[v] += (W[M%len(W)]/(n+1))\n            else:\n                res[v] = (W[M%len(W)]/(n+1))\n\n    # Sort dictionary by item weights\n    res = list(dict(sorted(res.items(), key=lambda item: -item[1])).keys())\n\n    # Return the top 12 items only\n    return ' '.join(res[:12])\n\ndef prep_subs(submissions):\n    sub0 = submissions[0]\n    if len(sub0.columns) == 2:\n        sub0.columns = ['customer_id', 'prediction0']\n    for i in range(1, len(subs)):\n        sub0[f'prediction{i}'] = submissions[i]['prediction'].fillna('')\n        sub0[f'prediction{i}'] = sub0[f'prediction{i}'].astype(str)\n\n    gc.collect()\n    sub0.head(3)\n    return sub0\n\ndef clean_sub(submission, weights):\n    for i in range(len(weights)):\n        del submission[f'prediction{i}']\n    gc.collect()\n    return submission","metadata":{"papermill":{"duration":146.042668,"end_time":"2022-04-23T14:11:19.816751","exception":false,"start_time":"2022-04-23T14:08:53.774083","status":"completed"},"tags":[],"scrolled":true,"execution":{"iopub.status.busy":"2022-05-08T18:10:51.689543Z","iopub.execute_input":"2022-05-08T18:10:51.689915Z","iopub.status.idle":"2022-05-08T18:10:51.723764Z","shell.execute_reply.started":"2022-05-08T18:10:51.689815Z","shell.execute_reply":"2022-05-08T18:10:51.723015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# To ensemble I used submissions from 9 public notebooks:\n* LB: 0.0225 - https://www.kaggle.com/lunapandachan/h-m-trending-products-weekly-add-test/notebook\n* LB: 0.0217 - https://www.kaggle.com/tarique7/hnm-exponential-decay-with-alternate-items/notebook\n* LB: 0.0221 - https://www.kaggle.com/astrung/lstm-sequential-modelwith-item-features-tutorial\n* LB: 0.0224 - https://www.kaggle.com/code/hirotakanogami/h-m-eda-customer-clustering-by-kmeans\n* LB: 0.0220 - https://www.kaggle.com/code/hengzheng/time-is-our-best-friend-v2/notebook\n* LB: 0.0227 - https://www.kaggle.com/code/hechtjp/h-m-eda-rule-base-by-customer-age\n* LB: 0.0231 - https://www.kaggle.com/code/ebn7amdi/trending/notebook?scriptVersionId=90980162\n* LB: 0.0225 - https://www.kaggle.com/code/mayukh18/svd-model-reranking-implicit-to-explicit-feedback\n\n# our own models:\n* LB: 0.0226 - LightSAN recbole model\n* LB: 0.0211 - LGBM\n* LB: ? - items together ONLY","metadata":{"papermill":{"duration":0.007358,"end_time":"2022-04-23T14:07:52.13554","exception":false,"start_time":"2022-04-23T14:07:52.128182","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%time\nsubs = []\nsubs.append(pd.read_csv('../input/handmbestperforming/h-m-trending-products-weekly-add-test.csv').sort_values('customer_id').reset_index(drop=True))\nsubs.append(pd.read_csv('../input/handmbestperforming/hnm-exponential-decay-with-alternate-items.csv').sort_values('customer_id').reset_index(drop=True))\nsubs.append(pd.read_csv('../input/handmbestperforming/lstm-sequential-modelwith-item-features-tutorial.csv').sort_values('customer_id').reset_index(drop=True))\nsubs.append(pd.read_csv('../input/hm-00224-solution/submission.csv').sort_values('customer_id').reset_index(drop=True))\nsubs.append(pd.read_csv('../input/handmbestperforming/time-is-our-best-friend-v2.csv').sort_values('customer_id').reset_index(drop=True))\nsubs.append(pd.read_csv('../input/handmbestperforming/rule-based-by-customer-age.csv').sort_values('customer_id').reset_index(drop=True))\nsubs.append(pd.read_csv('../input/h-m-faster-trending-products-weekly/submission.csv').sort_values('customer_id').reset_index(drop=True))\n#subs.append(pd.read_csv('../input/hm-00231-solution/submission.csv').sort_values('customer_id').reset_index(drop=True))\n#subs.append(pd.read_csv('../input/h-m-framework-for-partitioned-validation/submission.csv').sort_values('customer_id').reset_index(drop=True)) \nsubs.append(pd.read_csv('../input/0237-ensemble-submission-handm/0226_lightsan.csv.gzip').sort_values('customer_id').reset_index(drop=True))     # 0.0226\nsubs.append(pd.read_csv('../input/my-best-submissions-to-ensemble/basic_model_submission.csv').sort_values('customer_id').reset_index(drop=True))                  # 0.0211\nsubs.append(pd.read_parquet('../input/0237-ensemble-submission-handm/model_subs/items_together_sub.parquet.gzip').sort_values('customer_id').reset_index(drop=True))","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:10:51.730147Z","iopub.execute_input":"2022-05-08T18:10:51.731793Z","iopub.status.idle":"2022-05-08T18:11:58.764762Z","shell.execute_reply.started":"2022-05-08T18:10:51.731763Z","shell.execute_reply":"2022-05-08T18:11:58.763241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub0 = prep_subs(subs)","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:11:58.766545Z","iopub.execute_input":"2022-05-08T18:11:58.766782Z","iopub.status.idle":"2022-05-08T18:12:01.683369Z","shell.execute_reply.started":"2022-05-08T18:11:58.766742Z","shell.execute_reply":"2022-05-08T18:12:01.682657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\nimport numpy as np\n#scores = [0.0231, 0.0225, 0.0217, 0.0221, 0.0224, 0.022, 0.0227, 0.0231, 0.0225, 0.0226, 0.0211]\nscores = [0.0225, 0.0217, 0.0221, 0.0224, 0.022, 0.0227, 0.0231, 0.0225, 0.0226, 0.0211]\nprint(f'{len(scores)} ensembles being weighted')\nprint(f'avg score: {np.mean(scores)}')\navg_score = np.mean(scores)\nweights = np.array([math.e**((x-avg_score)*500) for x in scores])\n\nprint(f'suggested weights: {weights}')","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:12:01.684624Z","iopub.execute_input":"2022-05-08T18:12:01.684979Z","iopub.status.idle":"2022-05-08T18:12:01.692924Z","shell.execute_reply.started":"2022-05-08T18:12:01.684944Z","shell.execute_reply":"2022-05-08T18:12:01.692237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### first ensemble","metadata":{}},{"cell_type":"code","source":"# weights = [1.05,0.78,0.86,0.85,0.68,0.64,0.7,0.24,1.01, 0.8, 0.9] # 0.0239\n# weights = [1.05,0.78,0.86,0.85,0.68,0.64,0.7,0.5,1.01, 0.8, 0.9]\n# weights = [1.05, 0.78, 0.86, 0.85, 0.68, 0.64, 0.70, 0.24, 1.0, 0.8, 0.7] # 0.0241\n# weights = [1.05, 0.78, 0.86, 0.85, 0.68, 0.64, 0.70, 0.24, 1.2, 0.8, 0.7] # 0.0241\n# weights = [1.05, 0.78, 0.86, 0.85, 0.68, 0.64, 0.70, 0.24, 1.2, 0.5, 0.6] # 0.0241\n# weights = [1.05, 0.78, 0.86, 0.85, 0.68, 0.64, 0.70, 0.24, 1.0, 1.5, 0.5, 0.6] # 0.0240\n#weights = [1.05, 0.78, 0.86, 0.85, 0.68, 0.64, 0.70, 0.24, 1.2, 0.5] # 0.0242\n#weights = [1.05, 0.78, 0.86, 0.85, 0.68, 0.64, 0.70, 1.2, 0.24, 0.5] # 0.0240\n\nweights = [1.05, 0.78, 0.86, 0.85, 0.68, 0.64, 0.70, 0.24, 1.2, 0.5] # 0.0242\nprint(f'using weights {weights} for custom blend')\nsub0['prediction'] = sub0.apply(cust_blend, W = weights, axis=1)\nsub0.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:12:01.694841Z","iopub.execute_input":"2022-05-08T18:12:01.695467Z","iopub.status.idle":"2022-05-08T18:15:04.232644Z","shell.execute_reply.started":"2022-05-08T18:12:01.695413Z","shell.execute_reply":"2022-05-08T18:15:04.231932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub0 = clean_sub(sub0, weights)","metadata":{"papermill":{"duration":11.582373,"end_time":"2022-04-23T14:11:31.425482","exception":false,"start_time":"2022-04-23T14:11:19.843109","status":"completed"},"tags":[],"scrolled":true,"execution":{"iopub.status.busy":"2022-05-08T18:15:04.234025Z","iopub.execute_input":"2022-05-08T18:15:04.234269Z","iopub.status.idle":"2022-05-08T18:15:04.328029Z","shell.execute_reply.started":"2022-05-08T18:15:04.234236Z","shell.execute_reply":"2022-05-08T18:15:04.326887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub0.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:15:04.329816Z","iopub.execute_input":"2022-05-08T18:15:04.330092Z","iopub.status.idle":"2022-05-08T18:15:04.344284Z","shell.execute_reply.started":"2022-05-08T18:15:04.330056Z","shell.execute_reply":"2022-05-08T18:15:04.342872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub0.to_parquet('init_blend.parquet.gzip', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:15:04.345697Z","iopub.execute_input":"2022-05-08T18:15:04.346073Z","iopub.status.idle":"2022-05-08T18:15:05.822975Z","shell.execute_reply.started":"2022-05-08T18:15:04.346016Z","shell.execute_reply":"2022-05-08T18:15:05.822225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### second ensemble","metadata":{}},{"cell_type":"code","source":"sub0 = pd.read_parquet('init_blend.parquet.gzip')\nsub0.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:46:56.395804Z","iopub.execute_input":"2022-05-08T18:46:56.396068Z","iopub.status.idle":"2022-05-08T18:46:58.264993Z","shell.execute_reply.started":"2022-05-08T18:46:56.396039Z","shell.execute_reply":"2022-05-08T18:46:58.264329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nsubs = []\n# init mix from first ensemble\nsubs.append(sub0)#pd.read_parquet('../input/0237-ensemble-submission-handm/init_lsan_lgbm_it_ensemble.parquet.gzip'))\n\n# stuff for 2nd ensemble\nsubs.append(pd.read_csv('../input/h-m-framework-for-partitioned-validation/submission.csv').sort_values('customer_id').reset_index(drop=True)) \nsubs.append(pd.read_csv('../input/hm-00231-solution/submission.csv').sort_values('customer_id').reset_index(drop=True))","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:46:58.811657Z","iopub.execute_input":"2022-05-08T18:46:58.812326Z","iopub.status.idle":"2022-05-08T18:47:08.151153Z","shell.execute_reply.started":"2022-05-08T18:46:58.812275Z","shell.execute_reply":"2022-05-08T18:47:08.150390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub0 = prep_subs(subs)\nsub0.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:47:08.152695Z","iopub.execute_input":"2022-05-08T18:47:08.153102Z","iopub.status.idle":"2022-05-08T18:47:08.800207Z","shell.execute_reply.started":"2022-05-08T18:47:08.153063Z","shell.execute_reply":"2022-05-08T18:47:08.799512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# weights = [1.20, 0.85, 0.8]\n# weights = [1.20, 0.85] # .0241\n# weights = [1.20, 0.90] # .0241\n# weights = [1.20, 1.00] # .0239\n# weights = [1.30, 0.85] # .0240\n# weights = [1.20, 0.85, 0.1] # .0241\n# weights = [1.20, 0.85, 0.3] # .0241\n# weights = [1.20, 1.1, 0.9] # 0.0241\n# weights = [1.20, 0.85, 0.9] # 0.0242\n# weights = [1.20, 0.9, 0.85] # 0.0242\n\nweights = [1.20, 0.85, 0.75] # \nsub0['prediction'] = sub0.apply(cust_blend, W = weights, axis=1)\nsub0.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:47:29.452303Z","iopub.execute_input":"2022-05-08T18:47:29.452908Z","iopub.status.idle":"2022-05-08T18:48:39.946998Z","shell.execute_reply.started":"2022-05-08T18:47:29.452848Z","shell.execute_reply":"2022-05-08T18:48:39.946312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub0 = clean_sub(sub0, weights)\nsub0.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:48:39.948641Z","iopub.execute_input":"2022-05-08T18:48:39.948903Z","iopub.status.idle":"2022-05-08T18:48:40.049279Z","shell.execute_reply.started":"2022-05-08T18:48:39.948867Z","shell.execute_reply":"2022-05-08T18:48:40.048420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'2nd ensemble weights: {weights}')","metadata":{"execution":{"iopub.status.busy":"2022-05-08T18:48:40.050694Z","iopub.execute_input":"2022-05-08T18:48:40.050992Z","iopub.status.idle":"2022-05-08T18:48:40.055603Z","shell.execute_reply.started":"2022-05-08T18:48:40.050954Z","shell.execute_reply":"2022-05-08T18:48:40.054763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make a submission","metadata":{"papermill":{"duration":0.008635,"end_time":"2022-04-23T14:11:19.834502","exception":false,"start_time":"2022-04-23T14:11:19.825867","status":"completed"},"tags":[]}},{"cell_type":"code","source":"sub0.to_csv('the69toRuleThemAllv3.csv.gz', index=False)","metadata":{"papermill":{"duration":0.008509,"end_time":"2022-04-23T14:11:31.442856","exception":false,"start_time":"2022-04-23T14:11:31.434347","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-05-08T18:48:40.057750Z","iopub.execute_input":"2022-05-08T18:48:40.058279Z","iopub.status.idle":"2022-05-08T18:49:24.943367Z","shell.execute_reply.started":"2022-05-08T18:48:40.058240Z","shell.execute_reply":"2022-05-08T18:49:24.942655Z"},"trusted":true},"execution_count":null,"outputs":[]}]}