{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nfrom math import sqrt\nfrom pathlib import Path\nfrom tqdm import tqdm\ntqdm.pandas()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-19T20:37:22.499645Z","iopub.execute_input":"2022-02-19T20:37:22.500208Z","iopub.status.idle":"2022-02-19T20:37:22.505281Z","shell.execute_reply.started":"2022-02-19T20:37:22.500164Z","shell.execute_reply":"2022-02-19T20:37:22.504743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_path = Path('../input/h-and-m-personalized-fashion-recommendations/')\nN = 12","metadata":{"execution":{"iopub.status.busy":"2022-02-19T20:37:22.506594Z","iopub.execute_input":"2022-02-19T20:37:22.507333Z","iopub.status.idle":"2022-02-19T20:37:22.516148Z","shell.execute_reply.started":"2022-02-19T20:37:22.507291Z","shell.execute_reply":"2022-02-19T20:37:22.515707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Read the transactions data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(data_path / 'transactions_train.csv',\n                 usecols = ['t_dat', 'customer_id', 'article_id'],\n                 dtype={'article_id': str})\n\ndf['t_dat'] = pd.to_datetime(df['t_dat'])\nlast_ts = df['t_dat'].max()","metadata":{"execution":{"iopub.status.busy":"2022-02-19T20:37:22.517033Z","iopub.execute_input":"2022-02-19T20:37:22.517742Z","iopub.status.idle":"2022-02-19T20:38:06.917426Z","shell.execute_reply.started":"2022-02-19T20:37:22.517708Z","shell.execute_reply":"2022-02-19T20:38:06.916546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Add the last day of billing week","metadata":{}},{"cell_type":"code","source":"df['ldbw'] = df['t_dat'].progress_apply(lambda d: last_ts - (last_ts - d).floor('7D'))","metadata":{"execution":{"iopub.status.busy":"2022-02-19T20:38:06.918583Z","iopub.execute_input":"2022-02-19T20:38:06.918816Z","iopub.status.idle":"2022-02-19T20:38:18.866973Z","shell.execute_reply.started":"2022-02-19T20:38:06.918789Z","shell.execute_reply":"2022-02-19T20:38:18.866237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Count the number of transactions per week ","metadata":{}},{"cell_type":"code","source":"weekly_sales = df.drop('customer_id', axis=1).groupby(['ldbw', 'article_id']).count()\nweekly_sales = weekly_sales.rename(columns={'t_dat': 'count'})","metadata":{"execution":{"iopub.status.busy":"2022-02-19T20:38:18.869039Z","iopub.execute_input":"2022-02-19T20:38:18.869255Z","iopub.status.idle":"2022-02-19T20:38:18.912405Z","shell.execute_reply.started":"2022-02-19T20:38:18.869228Z","shell.execute_reply":"2022-02-19T20:38:18.911669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.join(weekly_sales, on=['ldbw', 'article_id'])","metadata":{"execution":{"iopub.status.busy":"2022-02-19T20:38:18.913505Z","iopub.execute_input":"2022-02-19T20:38:18.913711Z","iopub.status.idle":"2022-02-19T20:38:18.942832Z","shell.execute_reply.started":"2022-02-19T20:38:18.913674Z","shell.execute_reply":"2022-02-19T20:38:18.942125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's assume that in the target week sales will be similar to the last week of the training data","metadata":{}},{"cell_type":"code","source":"weekly_sales = weekly_sales.reset_index().set_index('article_id')\nlast_day = last_ts.strftime('%Y-%m-%d')\n\ndf = df.join(\n    weekly_sales.loc[weekly_sales['ldbw']==last_day, ['count']],\n    on='article_id', rsuffix=\"_targ\")\n\ndf['count_targ'].fillna(0, inplace=True)\ndel weekly_sales","metadata":{"execution":{"iopub.status.busy":"2022-02-19T20:38:18.943958Z","iopub.execute_input":"2022-02-19T20:38:18.944156Z","iopub.status.idle":"2022-02-19T20:38:18.976953Z","shell.execute_reply.started":"2022-02-19T20:38:18.944131Z","shell.execute_reply":"2022-02-19T20:38:18.976223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Calculate sales rate adjusted for changes in product popularity ","metadata":{}},{"cell_type":"code","source":"df['quotient'] = df['count_targ'] / df['count']","metadata":{"execution":{"iopub.status.busy":"2022-02-19T20:38:18.978101Z","iopub.execute_input":"2022-02-19T20:38:18.978299Z","iopub.status.idle":"2022-02-19T20:38:18.982617Z","shell.execute_reply.started":"2022-02-19T20:38:18.978276Z","shell.execute_reply":"2022-02-19T20:38:18.981917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Take supposedly popular products","metadata":{}},{"cell_type":"code","source":"target_sales = df.drop('customer_id', axis=1).groupby('article_id')['quotient'].sum()\ngeneral_pred = target_sales.nlargest(N).index.tolist()\ndel target_sales","metadata":{"execution":{"iopub.status.busy":"2022-02-19T20:38:18.983677Z","iopub.execute_input":"2022-02-19T20:38:18.98398Z","iopub.status.idle":"2022-02-19T20:38:19.034406Z","shell.execute_reply.started":"2022-02-19T20:38:18.983956Z","shell.execute_reply":"2022-02-19T20:38:19.033751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Fill in purchase dictionary","metadata":{}},{"cell_type":"code","source":"purchase_dict = {}\n\nfor i in tqdm(df.index):\n    cust_id = df.at[i, 'customer_id']\n    art_id = df.at[i, 'article_id']\n    t_dat = df.at[i, 't_dat']\n\n    if cust_id not in purchase_dict:\n        purchase_dict[cust_id] = {}\n\n    if art_id not in purchase_dict[cust_id]:\n        purchase_dict[cust_id][art_id] = 0\n    \n    x = max(1, (last_ts - t_dat).days)\n\n    a, b, c, d = 2.5e4, 1.5e5, 2e-1, 1e3\n    y = a / np.sqrt(x) + b * np.exp(-c*x) - d\n\n    value = df.at[i, 'quotient'] * max(0, y)\n    purchase_dict[cust_id][art_id] += value","metadata":{"execution":{"iopub.status.busy":"2022-02-19T20:38:19.035605Z","iopub.execute_input":"2022-02-19T20:38:19.035842Z","iopub.status.idle":"2022-02-19T20:38:23.638062Z","shell.execute_reply.started":"2022-02-19T20:38:19.035815Z","shell.execute_reply":"2022-02-19T20:38:23.637245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Make a submission","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv(data_path / 'sample_submission.csv')\n\npred_list = []\nfor cust_id in tqdm(sub['customer_id']):\n    if cust_id in purchase_dict:\n        series = pd.Series(purchase_dict[cust_id])\n        series = series[series > 0]\n        l = series.nlargest(N).index.tolist()\n        if len(l) < N:\n            l = l + general_pred[:(N-len(l))]\n    else:\n        l = general_pred\n    pred_list.append(' '.join(l))\n\nsub['prediction'] = pred_list\nsub.to_csv('submission.csv', index=None)","metadata":{"execution":{"iopub.status.busy":"2022-02-19T20:38:23.639566Z","iopub.execute_input":"2022-02-19T20:38:23.640218Z","iopub.status.idle":"2022-02-19T20:39:03.290862Z","shell.execute_reply.started":"2022-02-19T20:38:23.640174Z","shell.execute_reply":"2022-02-19T20:39:03.289966Z"},"trusted":true},"execution_count":null,"outputs":[]}]}