{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"* Borrowed some of the preprocessing and EDA code from Ahmet Erdem's [H&M Pure Pytorch Baseline](https://www.kaggle.com/code/aerdem4/h-m-pure-pytorch-baseline/notebook) notebook.\n* Employed some tricks from [this thread by Chris Deotte](https://www.kaggle.com/competitions/h-and-m-personalized-fashion-recommendations/discussion/308635) to reduce memory footprint.","metadata":{}},{"cell_type":"code","source":"import gc\nimport sys\nfrom itertools import chain\n\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import OrdinalEncoder","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-27T09:44:49.042884Z","iopub.execute_input":"2022-03-27T09:44:49.043733Z","iopub.status.idle":"2022-03-27T09:44:51.212141Z","shell.execute_reply.started":"2022-03-27T09:44:49.043627Z","shell.execute_reply":"2022-03-27T09:44:51.211283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reading and preprocessing data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\", dtype={\"article_id\": str}, parse_dates=[\"t_dat\"])\ndf['customer_id_int'] = df['customer_id'].apply(lambda x: int(x[-16:],16) ).astype('int64')\ndel df['customer_id']\ndf['article_id'] = df['article_id'].astype('int32')\nprint(df.shape)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:44:51.214289Z","iopub.execute_input":"2022-03-27T09:44:51.214623Z","iopub.status.idle":"2022-03-27T09:46:46.195014Z","shell.execute_reply.started":"2022-03-27T09:44:51.214578Z","shell.execute_reply":"2022-03-27T09:46:46.193971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv').drop(\"prediction\", axis=1)\ntest_df['customer_id_int'] = test_df['customer_id'].apply(lambda x: int(x[-16:],16) ).astype('int64')","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:46:46.196793Z","iopub.execute_input":"2022-03-27T09:46:46.197067Z","iopub.status.idle":"2022-03-27T09:46:52.478534Z","shell.execute_reply.started":"2022-03-27T09:46:46.197033Z","shell.execute_reply":"2022-03-27T09:46:52.477652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Max `t_dat`:\", df[\"t_dat\"].max())\nactive_articles = df.groupby(\"article_id\")[\"t_dat\"].max().reset_index()\nactive_articles = active_articles[active_articles[\"t_dat\"] >= \"2019-09-01\"].reset_index()\nn_classes = active_articles.shape[0] + 1\nactive_articles.shape, n_classes","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:46:52.480627Z","iopub.execute_input":"2022-03-27T09:46:52.480835Z","iopub.status.idle":"2022-03-27T09:46:53.970091Z","shell.execute_reply.started":"2022-03-27T09:46:52.480810Z","shell.execute_reply":"2022-03-27T09:46:53.969040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# restrict the entries to articles that has appearance after 2019-01-01\ndf = df[df[\"article_id\"].isin(active_articles[\"article_id\"])].reset_index(drop=True)\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:46:53.971443Z","iopub.execute_input":"2022-03-27T09:46:53.971771Z","iopub.status.idle":"2022-03-27T09:46:57.631835Z","shell.execute_reply.started":"2022-03-27T09:46:53.971730Z","shell.execute_reply":"2022-03-27T09:46:57.630919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"week\"] = (df[\"t_dat\"].max() - df[\"t_dat\"]).dt.days // 7\nprint(df[\"week\"].nunique())","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:46:57.633173Z","iopub.execute_input":"2022-03-27T09:46:57.633401Z","iopub.status.idle":"2022-03-27T09:46:59.221798Z","shell.execute_reply.started":"2022-03-27T09:46:57.633375Z","shell.execute_reply":"2022-03-27T09:46:59.220882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Most-Common-Items baseline","metadata":{}},{"cell_type":"code","source":"item_counts = df[df.week < 2].article_id.value_counts()\nitem_counts[:12]","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:46:59.223164Z","iopub.execute_input":"2022-03-27T09:46:59.223474Z","iopub.status.idle":"2022-03-27T09:46:59.889692Z","shell.execute_reply.started":"2022-03-27T09:46:59.223432Z","shell.execute_reply":"2022-03-27T09:46:59.888747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"most_frequent_items = item_counts[:12].index.to_numpy()\nprediction_str = \" \".join(map(\"{:010d}\".format, most_frequent_items))\nprediction_str","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:46:59.891117Z","iopub.execute_input":"2022-03-27T09:46:59.891424Z","iopub.status.idle":"2022-03-27T09:46:59.898358Z","shell.execute_reply.started":"2022-03-27T09:46:59.891385Z","shell.execute_reply":"2022-03-27T09:46:59.897596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[\"prediction\"] = prediction_str\ntest_df.to_csv(\"submission_most_common_items.csv.gz\", compression=\"gzip\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:46:59.899649Z","iopub.execute_input":"2022-03-27T09:46:59.899872Z","iopub.status.idle":"2022-03-27T09:47:24.596154Z","shell.execute_reply.started":"2022-03-27T09:46:59.899839Z","shell.execute_reply":"2022-03-27T09:47:24.595482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  Recently-Purchased-Items Baseline","metadata":{}},{"cell_type":"code","source":"df_tmp = df[df[\"week\"] <= 7].sort_values([\"customer_id_int\", \"t_dat\"], ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:48:50.551722Z","iopub.execute_input":"2022-03-27T09:48:50.552533Z","iopub.status.idle":"2022-03-27T09:48:51.509985Z","shell.execute_reply.started":"2022-03-27T09:48:50.552498Z","shell.execute_reply":"2022-03-27T09:48:51.509290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def keep_latest_k(articles, k=12):\n    result = []\n    # Use most commonly bought items to fill the empty spots\n    for item in chain(articles, most_frequent_items):\n        if item in result:\n            continue\n        result.append(item)\n        if len(result) == k:\n            break\n    return result","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:50:15.248070Z","iopub.execute_input":"2022-03-27T09:50:15.248704Z","iopub.status.idle":"2022-03-27T09:50:15.254109Z","shell.execute_reply.started":"2022-03-27T09:50:15.248673Z","shell.execute_reply":"2022-03-27T09:50:15.253249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_latest_items = df_tmp.groupby(\"customer_id_int\").agg({\"article_id\": keep_latest_k}).reset_index()\ndf_latest_items[\"prediction\"] = df_latest_items.article_id.apply(lambda x: \" \".join(map(\"{:010d}\".format, x)))\ndf_latest_items.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:50:15.491491Z","iopub.execute_input":"2022-03-27T09:50:15.492213Z","iopub.status.idle":"2022-03-27T09:50:27.476042Z","shell.execute_reply.started":"2022-03-27T09:50:15.492175Z","shell.execute_reply":"2022-03-27T09:50:27.475102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = test_df.drop('prediction', axis=1).merge(\n    df_latest_items[[\"customer_id_int\", \"prediction\"]], \n    how=\"left\", on=\"customer_id_int\")\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:52:42.126232Z","iopub.execute_input":"2022-03-27T09:52:42.126554Z","iopub.status.idle":"2022-03-27T09:52:42.916599Z","shell.execute_reply.started":"2022-03-27T09:52:42.126507Z","shell.execute_reply":"2022-03-27T09:52:42.915922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use most-bought-items for customers without recent purchase history\ntest_df[\"prediction\"] = test_df[\"prediction\"].fillna(prediction_str)\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:53:17.360940Z","iopub.execute_input":"2022-03-27T09:53:17.361286Z","iopub.status.idle":"2022-03-27T09:53:17.609556Z","shell.execute_reply.started":"2022-03-27T09:53:17.361247Z","shell.execute_reply":"2022-03-27T09:53:17.608588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[[\"customer_id\", \"prediction\"]].to_csv(\"submission_recently_purchased.csv.gz\", compression=\"gzip\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:54:02.158580Z","iopub.execute_input":"2022-03-27T09:54:02.159328Z","iopub.status.idle":"2022-03-27T09:54:31.542036Z","shell.execute_reply.started":"2022-03-27T09:54:02.159270Z","shell.execute_reply":"2022-03-27T09:54:31.540935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}