{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this notebook, I would like to make a submission whose score is probably around 0.007.   \n\nBasically, I will follow [this notebook](https://www.kaggle.com/julian3833/h-m-content-based-12-most-popular-items-0-003) and adopt the \"*recommend the most popular items*\" approach.  \nWhat I'm going to try here is some adjustments of the popularity using two kinds of information; age and time.  \n\nI'm glad if you could find something useful to you.  ","metadata":{}},{"cell_type":"code","source":"import numpy as np, pandas as pd, datetime as dt\nimport matplotlib.pyplot as plt\nimport seaborn as sns; sns.set()\nfrom collections import Counter, defaultdict\nfrom PIL import Image\nfrom pathlib import Path\npath = Path(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/\")\n\ndef show_images(article_ids, cols=1, texts=[], suptitle=''):\n    if isinstance(article_ids, int) or isinstance(article_ids, str):\n        article_ids = [article_ids]\n    rows = (len(article_ids) // cols) + 1\n    plt.figure(figsize=(3 + 3.5 * cols, 3 + 5 * rows))\n    for i, article_id in enumerate(article_ids):\n        article_id = (\"0\" + str(article_id))[-10:]\n        text = '' if len(texts) <= i else ('\\n' + texts[i])\n        plt.subplot(rows, cols, i + 1)\n        plt.axis('off')\n        plt.title(f\"{article_id}{text}\", fontsize=16)\n        try:\n            image = Image.open(f\"/kaggle/input/h-and-m-personalized-fashion-recommendations/images/{article_id[:3]}/{article_id}.jpg\")\n            plt.imshow(image)\n        except:\n            pass\n    if suptitle != '': plt.suptitle(suptitle, fontsize=36, fontweight='bold')\n    plt.tight_layout()\n\ndef iter_to_str(iterable):\n    return \" \".join(map(lambda x: str(0) + str(x), iterable))\n\ndef apk(actual, predicted, k=12):\n    if len(predicted) > k:\n        predicted = predicted[:k]\n    score, nhits = 0.0, 0.0\n    for i, p in enumerate(predicted):\n        if p in actual and p not in predicted[:i]:\n            nhits += 1.0\n            score += nhits / (i + 1.0)\n    if not actual:\n        return 0.0\n    return score / min(len(actual), k)\n\ndef mapk(actual, predicted, k=12, return_apks=False):\n    assert len(actual) == len(predicted)\n    apks = [apk(ac, pr, k) for ac, pr in zip(actual, predicted) if 0 < len(ac)]\n    if return_apks:\n        return apks\n    return np.mean(apks)\n\ndf = pd.read_parquet('../input/hm-parquets-of-datasets/transactions_train.parquet')\ncdf = pd.read_parquet('../input/hm-parquets-of-datasets/customers.parquet')\nsub = pd.read_csv(path / 'sample_submission.csv')\n\nvalid_week = 105 # number of week to be used in a validation\nvalid = df[df.week == valid_week].groupby('customer_id').article_id.apply(iter_to_str).reset_index()\\\n    .merge(cdf['customer_id'], on='customer_id', how='right')\nactual = valid.article_id.apply(lambda s: [] if pd.isna(s) else s.split())\nlast_date = df[df.week < valid_week]['t_dat'].max()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-14T08:32:51.181810Z","iopub.execute_input":"2022-05-14T08:32:51.182779Z","iopub.status.idle":"2022-05-14T08:32:52.282525Z","shell.execute_reply.started":"2022-05-14T08:32:51.182598Z","shell.execute_reply":"2022-05-14T08:32:52.281904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Age-adjusted and Time-discounted Popular Items","metadata":{}},{"cell_type":"markdown","source":"### Time-discounting\nSince H&M is a fast-fashion brand and time is one of the most important key of the competition, here I use only the data of last 21 days, adjusting the weights of each transaction according to how many days old the transaction is.  \nThe weight of transactions in the t-th day from the last day is calculated as follows.\n$$\nweight(t) = \\frac{1}{t ^ {1.4}}\n$$\n### Age-Adjustment\nThen, a recommendation for x-year-old customers can be created as 12 items that are popular in customers whose age are in the range from x-w to x+w.","metadata":{}},{"cell_type":"code","source":"init_date = last_date - dt.timedelta(days=21 - 1)\ntrain = df.loc[(df.t_dat >= init_date) & (df.t_dat <= last_date)].copy()\n\n# time discount\ntrain['factor'] = (1 / ((last_date - train['t_dat']).dt.days + 1)) ** 1.4\n\n# replace NA values of the customer's age with the mean\ncdf.loc[pd.isna(cdf['age']), 'age'] = cdf['age'].mean()\n\n# add age information to transactions\ntrain = train.merge(cdf[['customer_id', 'age']], on='customer_id', how='left')\n\ndef age_adjusted_popular_items(x, width, k=12):\n    temp = train[(x - width <= train['age']) & (train['age'] <= x + width)].reset_index()\n    recommend = iter_to_str(temp.groupby('article_id')['factor'].sum().nlargest(k).index.to_list())\n    return recommend\n\n# widths for each age\nwidth_dict = defaultdict(int)\nfor x in range(22): width_dict[x] = 2\nfor x in range(22, 30): width_dict[x] = 3\nfor x in range(30, 36): width_dict[x] = 4\nfor x in range(36, 45): width_dict[x] = 5\nfor x in range(45, 50): width_dict[x] = 5\nfor x in range(50, 60): width_dict[x] = 6\nfor x in range(60, 100): width_dict[x] = 10\n\nrecommend = defaultdict(str)    \nfor x in cdf['age'].unique():\n    recommend[x] = age_adjusted_popular_items(x, width_dict[x])\n\nsub['prediction'] = cdf['age'].map(recommend)","metadata":{"execution":{"iopub.status.busy":"2022-05-14T08:33:07.610322Z","iopub.execute_input":"2022-05-14T08:33:07.610544Z","iopub.status.idle":"2022-05-14T08:33:08.371061Z","shell.execute_reply.started":"2022-05-14T08:33:07.610517Z","shell.execute_reply":"2022-05-14T08:33:08.370062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for age in range(18, 70, 8):\n    show_images(recommend[age].split(), 12, suptitle=f'Recommendation for {age}-year-old Customers')","metadata":{"execution":{"iopub.status.busy":"2022-05-14T08:34:58.375741Z","iopub.execute_input":"2022-05-14T08:34:58.375966Z","iopub.status.idle":"2022-05-14T08:35:38.492917Z","shell.execute_reply.started":"2022-05-14T08:34:58.375940Z","shell.execute_reply":"2022-05-14T08:35:38.492020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{"execution":{"iopub.status.busy":"2022-02-09T14:23:39.571322Z","iopub.execute_input":"2022-02-09T14:23:39.572263Z","iopub.status.idle":"2022-02-09T14:23:39.576232Z","shell.execute_reply.started":"2022-02-09T14:23:39.572209Z","shell.execute_reply":"2022-02-09T14:23:39.575589Z"}}},{"cell_type":"code","source":"display(sub.head())\nsub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-05-14T08:36:04.144292Z","iopub.execute_input":"2022-05-14T08:36:04.144603Z","iopub.status.idle":"2022-05-14T08:36:19.844127Z","shell.execute_reply.started":"2022-05-14T08:36:04.144570Z","shell.execute_reply":"2022-05-14T08:36:19.842730Z"},"trusted":true},"execution_count":null,"outputs":[]}]}