{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom math import sqrt\nfrom pathlib import Path\nfrom tqdm import tqdm\ntqdm.pandas()\n\nN = 12\ndf_trans = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv',dtype={'article_id': str})\ndf_trans['t_dat'] = pd.to_datetime(df_trans['t_dat'])\n","metadata":{"execution":{"iopub.status.busy":"2022-03-20T07:33:23.283018Z","iopub.execute_input":"2022-03-20T07:33:23.283603Z","iopub.status.idle":"2022-03-20T07:34:41.817897Z","shell.execute_reply.started":"2022-03-20T07:33:23.283512Z","shell.execute_reply":"2022-03-20T07:34:41.81687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step1\ndf = df_trans[['t_dat', 'customer_id', 'article_id']].copy()\nlast_ts = df['t_dat'].max()\ndf['ldbw'] = df['t_dat'].apply(lambda d: last_ts - (last_ts - d).floor('7D'))\nweekly_sales = df.drop('customer_id', axis=1).groupby(['ldbw', 'article_id']).count()\nweekly_sales = weekly_sales.rename(columns={'t_dat': 'count'})\ndf = df.join(weekly_sales, on=['ldbw', 'article_id'])\nweekly_sales = weekly_sales.reset_index().set_index('article_id')\nlast_day = last_ts.strftime('%Y-%m-%d')\n\ndf = df.join(\n    weekly_sales.loc[weekly_sales['ldbw']==last_day, ['count']],\n    on='article_id', rsuffix=\"_targ\")\n\ndf['count_targ'].fillna(0, inplace=True)\ndel weekly_sales\ndf['quotient'] = df['count_targ'] / df['count']\n\npurchase_dict = {}\n\nfor i in tqdm(df.index):\n    cust_id = df.at[i, 'customer_id']\n    art_id = df.at[i, 'article_id']\n    t_dat = df.at[i, 't_dat']\n\n    if cust_id not in purchase_dict:\n        purchase_dict[cust_id] = {}\n\n    if art_id not in purchase_dict[cust_id]:\n        purchase_dict[cust_id][art_id] = 0\n    \n    x = max(1, (last_ts - t_dat).days)\n\n    a, b, c, d = 2.5e4, 1.5e5, 2e-2, 1e3\n    y = a / np.sqrt(x) + b * np.exp(-c*x) - d\n\n    value = df.at[i, 'quotient'] * max(0, y)\n    purchase_dict[cust_id][art_id] += value\n\ntarget_sales = df.drop('customer_id', axis=1).groupby('article_id')['quotient'].sum()\ngeneral_pred = target_sales.nlargest(N).index.tolist()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-20T07:34:41.819772Z","iopub.execute_input":"2022-03-20T07:34:41.820456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step2 & Step3\npairs = np.load('../input/hmitempairs/pairs_cudf.npy',allow_pickle=True).item()\nsub = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv')\n\npred_list = []\nfor cust_id in tqdm(sub['customer_id']):\n    if cust_id in purchase_dict:\n        series = pd.Series(purchase_dict[cust_id])\n        series = series[series > 160]\n        l = series.nlargest(N).index.tolist()\n        tmp_l = l.copy()\n        for elm in tmp_l:\n            if len(l) < N and int(elm) in pairs.keys():\n                itm = pairs[int(elm)]\n                l.append('0' + str(itm))\n        if len(l) < N:\n            l = l + general_pred[:(N-len(l))]\n    else:\n        l = general_pred\n    pred_list.append(' '.join(l))\n\nsub['prediction'] = pred_list\nsub.to_csv(f'submission.csv',index=False)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}