{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nfrom collections import defaultdict\nfrom heapq import nlargest\nfrom tqdm import tqdm\nfrom pathlib import Path\n\ndata_path = Path('/kaggle/input/h-and-m-personalized-fashion-recommendations/')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-01-18T14:21:39.850594Z","iopub.execute_input":"2022-01-18T14:21:39.850976Z","iopub.status.idle":"2022-01-18T14:21:39.857065Z","shell.execute_reply.started":"2022-01-18T14:21:39.85092Z","shell.execute_reply":"2022-01-18T14:21:39.856028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions = pd.read_csv(\n    data_path / 'transactions_train.csv',\n    # set dtype or pandas will drop the leading '0' and convert to int\n    dtype={'article_id': str} \n)\n\nsubmission = pd.read_csv(data_path / 'sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-01-18T14:21:40.788564Z","iopub.execute_input":"2022-01-18T14:21:40.788872Z","iopub.status.idle":"2022-01-18T14:23:00.532626Z","shell.execute_reply.started":"2022-01-18T14:21:40.788837Z","shell.execute_reply":"2022-01-18T14:23:00.531574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-18T14:23:33.901611Z","iopub.execute_input":"2022-01-18T14:23:33.902092Z","iopub.status.idle":"2022-01-18T14:23:33.936972Z","shell.execute_reply.started":"2022-01-18T14:23:33.902054Z","shell.execute_reply":"2022-01-18T14:23:33.935944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-18T14:23:34.936601Z","iopub.execute_input":"2022-01-18T14:23:34.936928Z","iopub.status.idle":"2022-01-18T14:23:34.949066Z","shell.execute_reply.started":"2022-01-18T14:23:34.936896Z","shell.execute_reply":"2022-01-18T14:23:34.947889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For each customer_id, count each article_id they've previously purchased\n\ncounter = defaultdict(dict) # nested dict\n\nfor idx, row in tqdm(transactions.iterrows()):\n    customer_id = row['customer_id']\n    article_id = row['article_id']\n    counter[customer_id][article_id] = counter[customer_id].get(article_id, 0) + 1\n\nmost_common_benchmark = submission.set_index('customer_id', drop=True)\n\nfor customer_id, purchase_dict in tqdm(counter.items()):\n    top_purchases = ' '.join(nlargest(12, purchase_dict, key=purchase_dict.get)) # top 12 purchases\n    most_common_benchmark.loc[customer_id, 'prediction'] = top_purchases\n\nmost_common_benchmark.to_csv('most_common_benchmark.csv')","metadata":{},"execution_count":null,"outputs":[]}]}