{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Weighted Top Pop","metadata":{}},{"cell_type":"markdown","source":"As other participants showcased in their notebooks, there are two important factors for every item: \n* Seasonality of products \n* Age of products (Products sold in days closer to the week we must predict are more likely to be sold )\n\nAbout the seasonality: some products are sold only in a specific time of the year. \nAbout the age: products sold in days closer to the test_set week are more likely to be bought in the test week.\n\nTo address these two factors in a Top Popularity recommendation approach, I added a weight to each transaction.\nThis weight is multiplied by an exponentially decaying weight following the formula:\n    e^(-(days/temperature))\nWhere days indicates the distance in days between the start of the test week and temperature is a parameter which is used to tune how fast is the decay.\nWith a lower temperature the weight for older interactions becomes really low, so they get less likely to be ranked as top popular.\n\nTo address seasonality I multiplied the weight with an element of a vector which has an element for each month.\nConsidering that the predictions must be done for the last days of September 2020, I gave a weight of 1 to interactions for September of any year and lower weights to interactions for other months.\n\nChanging these parameters the Map@12 on local validation set consisting of the transaction in the week before the week we must make predictions on can change from 0.0026  of a basic TopPop to values like 0.0088 using a really low temperature value (which has results similar to a Top Popular Recommender considering only most recent transactions). ","metadata":{}},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nfrom datetime import datetime\nis_test=True","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-14T21:11:16.38584Z","iopub.execute_input":"2022-02-14T21:11:16.386704Z","iopub.status.idle":"2022-02-14T21:11:16.416132Z","shell.execute_reply.started":"2022-02-14T21:11:16.386566Z","shell.execute_reply":"2022-02-14T21:11:16.41499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare Data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\",dtype={\"article_id\":str})","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:11:16.417783Z","iopub.execute_input":"2022-02-14T21:11:16.418826Z","iopub.status.idle":"2022-02-14T21:12:30.345327Z","shell.execute_reply.started":"2022-02-14T21:11:16.418772Z","shell.execute_reply":"2022-02-14T21:12:30.344385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:12:30.346866Z","iopub.execute_input":"2022-02-14T21:12:30.347116Z","iopub.status.idle":"2022-02-14T21:12:30.375514Z","shell.execute_reply.started":"2022-02-14T21:12:30.347071Z","shell.execute_reply":"2022-02-14T21:12:30.37465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"date_time\"]=pd.to_datetime(df[\"t_dat\"])","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:12:30.377528Z","iopub.execute_input":"2022-02-14T21:12:30.377891Z","iopub.status.idle":"2022-02-14T21:12:35.930553Z","shell.execute_reply.started":"2022-02-14T21:12:30.377841Z","shell.execute_reply":"2022-02-14T21:12:35.929518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop([c for c in df.columns if c not in [\"date_time\",\"article_id\",\"customer_id\"]],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:12:35.93174Z","iopub.execute_input":"2022-02-14T21:12:35.932476Z","iopub.status.idle":"2022-02-14T21:12:37.100475Z","shell.execute_reply.started":"2022-02-14T21:12:35.932442Z","shell.execute_reply":"2022-02-14T21:12:37.099672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if is_test:\n    last_week_start = datetime.strptime(\"24/09/20 00:00:00\", '%d/%m/%y %H:%M:%S')\nelse:\n    last_week_start = datetime.strptime(\"16/09/20 00:00:00\", '%d/%m/%y %H:%M:%S')","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:12:37.101457Z","iopub.execute_input":"2022-02-14T21:12:37.102088Z","iopub.status.idle":"2022-02-14T21:12:37.108081Z","shell.execute_reply.started":"2022-02-14T21:12:37.102041Z","shell.execute_reply":"2022-02-14T21:12:37.107221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_valid = df.loc[df[\"date_time\"] >= last_week_start ].drop(\"date_time\",axis=1)\nuser_to_evaluate=df_valid[\"customer_id\"].unique()\ndf_train = df.loc[df[\"date_time\"] <  last_week_start ].drop(\"customer_id\",axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:12:37.109398Z","iopub.execute_input":"2022-02-14T21:12:37.110431Z","iopub.status.idle":"2022-02-14T21:12:39.348252Z","shell.execute_reply.started":"2022-02-14T21:12:37.110364Z","shell.execute_reply":"2022-02-14T21:12:39.347405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_valid[\"list\"] = df_valid.groupby(\"customer_id\")[\"article_id\"].transform(lambda x: \" \".join([str(i) for i in x]))\ndf_valid.drop_duplicates(\"customer_id\",inplace=True)  ","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:12:39.349651Z","iopub.execute_input":"2022-02-14T21:12:39.349951Z","iopub.status.idle":"2022-02-14T21:12:39.362565Z","shell.execute_reply.started":"2022-02-14T21:12:39.34991Z","shell.execute_reply":"2022-02-14T21:12:39.361724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid=df_valid[\"list\"].to_numpy()","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:12:39.364126Z","iopub.execute_input":"2022-02-14T21:12:39.364653Z","iopub.status.idle":"2022-02-14T21:12:39.376021Z","shell.execute_reply.started":"2022-02-14T21:12:39.364609Z","shell.execute_reply":"2022-02-14T21:12:39.375129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_list=[x.split(\" \") for x in valid]","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:12:39.378889Z","iopub.execute_input":"2022-02-14T21:12:39.379717Z","iopub.status.idle":"2022-02-14T21:12:39.387673Z","shell.execute_reply.started":"2022-02-14T21:12:39.37967Z","shell.execute_reply":"2022-02-14T21:12:39.386883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"month\"] = df_train[\"date_time\"].dt.month\ndf_train[\"days_distance\"] = (last_week_start - df_train[\"date_time\"]).dt.days","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:12:39.389023Z","iopub.execute_input":"2022-02-14T21:12:39.389482Z","iopub.status.idle":"2022-02-14T21:12:43.324266Z","shell.execute_reply.started":"2022-02-14T21:12:39.389439Z","shell.execute_reply":"2022-02-14T21:12:43.323598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Calculate Weight for each product","metadata":{}},{"cell_type":"code","source":"temperature = 3 # parameter of exponential decay \ndf_train[\"weight\"] = 1\ndf_train[\"weight\"] *= np.exp(-(df_train[\"days_distance\"]/temperature))","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:12:43.32532Z","iopub.execute_input":"2022-02-14T21:12:43.325592Z","iopub.status.idle":"2022-02-14T21:12:44.002378Z","shell.execute_reply.started":"2022-02-14T21:12:43.325562Z","shell.execute_reply":"2022-02-14T21:12:44.001488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"month_weights = [0,0,0,0,0,0,0.2,0.6,1,0.8,0.4,0.1] #weight of products bought in every month\n\ndf_train[\"weight\"]*=df_train[\"month\"].apply(lambda x: month_weights[x-1])\n","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:12:44.003802Z","iopub.execute_input":"2022-02-14T21:12:44.004127Z","iopub.status.idle":"2022-02-14T21:13:00.285114Z","shell.execute_reply.started":"2022-02-14T21:12:44.004086Z","shell.execute_reply":"2022-02-14T21:13:00.284332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_g = df_train.groupby(\"article_id\").sum().reset_index() #sum weight for every product bought\n","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:13:00.286242Z","iopub.execute_input":"2022-02-14T21:13:00.286478Z","iopub.status.idle":"2022-02-14T21:13:07.84418Z","shell.execute_reply.started":"2022-02-14T21:13:00.286451Z","shell.execute_reply":"2022-02-14T21:13:07.843242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_sorted=df_train_g.sort_values(by=\"weight\",ascending=False)\n\nproducts = df_train_sorted[\"article_id\"].to_numpy()[:12]","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:13:07.845883Z","iopub.execute_input":"2022-02-14T21:13:07.84638Z","iopub.status.idle":"2022-02-14T21:13:07.878348Z","shell.execute_reply.started":"2022-02-14T21:13:07.846335Z","shell.execute_reply":"2022-02-14T21:13:07.877737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_sorted[[\"article_id\",\"weight\"]].head(20)","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:13:07.88082Z","iopub.execute_input":"2022-02-14T21:13:07.881138Z","iopub.status.idle":"2022-02-14T21:13:07.901073Z","shell.execute_reply.started":"2022-02-14T21:13:07.881096Z","shell.execute_reply":"2022-02-14T21:13:07.900271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"products","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:13:07.902489Z","iopub.execute_input":"2022-02-14T21:13:07.902721Z","iopub.status.idle":"2022-02-14T21:13:07.908225Z","shell.execute_reply.started":"2022-02-14T21:13:07.902694Z","shell.execute_reply":"2022-02-14T21:13:07.907458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualize Weighted Top Popular Items","metadata":{}},{"cell_type":"markdown","source":"code from https://www.kaggle.com/negoto/h-m-sales-period-of-fashion-items-with-k-means#kln-69","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns; sns.set()\nfrom PIL import Image\nfrom pathlib import Path\npath = Path(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/\")\n\ndef show_images(article_ids, cols=1, rows=-1):\n    if isinstance(article_ids, int) or isinstance(article_ids, str):\n        article_ids = [article_ids]\n    article_count = len(article_ids)\n    if rows < 0: rows = (article_count // cols) + 1\n    plt.figure(figsize=(3 + 3.5 * cols, 3 + 5 * rows))\n    for i in range(article_count):\n        article_id = (\"0\" + str(article_ids[i]))[-10:]\n        plt.subplot(rows, cols, i + 1)\n        plt.axis('off')\n        plt.title(article_id)\n        try:\n            image = Image.open(f\"/kaggle/input/h-and-m-personalized-fashion-recommendations/images/{article_id[:3]}/{article_id}.jpg\")\n            plt.imshow(image)\n        except:\n            pass","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:13:07.910168Z","iopub.execute_input":"2022-02-14T21:13:07.910921Z","iopub.status.idle":"2022-02-14T21:13:08.987886Z","shell.execute_reply.started":"2022-02-14T21:13:07.910866Z","shell.execute_reply":"2022-02-14T21:13:08.986934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" show_images(products)","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:13:08.989144Z","iopub.execute_input":"2022-02-14T21:13:08.989386Z","iopub.status.idle":"2022-02-14T21:13:14.273787Z","shell.execute_reply.started":"2022-02-14T21:13:08.989359Z","shell.execute_reply":"2022-02-14T21:13:14.27265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compute Metric MaP@12","metadata":{}},{"cell_type":"markdown","source":"Code from https://github.com/benhamner/Metrics","metadata":{}},{"cell_type":"code","source":"def apk(actual, predicted, k=12):\n    \"\"\"\n    Computes the average precision at k.\n    This function computes the average prescision at k between two lists of\n    items.\n    Parameters\n    ----------\n    actual : list\n             A list of elements that are to be predicted (order doesn't matter)\n    predicted : list\n                A list of predicted elements (order does matter)\n    k : int, optional\n        The maximum number of predicted elements\n    Returns\n    -------\n    score : double\n            The average precision at k over the input lists\n    \"\"\"\n    if len(predicted)>k:\n        predicted = predicted[:k]\n\n    score = 0.0\n    num_hits = 0.0\n\n    for i,p in enumerate(predicted):\n        if p in actual and p not in predicted[:i]:\n            num_hits += 1.0\n            score += num_hits / (i+1.0)\n\n    if not actual:\n        return 0.0\n\n    return score / min(len(actual), k)\n\ndef mapk(actual, predicted, k=12):\n    \"\"\"\n    Computes the mean average precision at k.\n    This function computes the mean average prescision at k between two lists\n    of lists of items.\n    Parameters\n    ----------\n    actual : list\n             A list of lists of elements that are to be predicted \n             (order doesn't matter in the lists)\n    predicted : list\n                A list of lists of predicted elements\n                (order matters in the lists)\n    k : int, optional\n        The maximum number of predicted elements\n    Returns\n    -------\n    score : double\n            The mean average precision at k over the input lists\n    \"\"\"\n    return np.mean([apk(a,predicted,k) for a in actual])\n        ","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:13:14.276171Z","iopub.execute_input":"2022-02-14T21:13:14.276828Z","iopub.status.idle":"2022-02-14T21:13:14.286105Z","shell.execute_reply.started":"2022-02-14T21:13:14.27678Z","shell.execute_reply":"2022-02-14T21:13:14.285354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not is_test:\n    mapk(valid_list,[str(x) for x in products])","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:13:14.287307Z","iopub.execute_input":"2022-02-14T21:13:14.288074Z","iopub.status.idle":"2022-02-14T21:13:14.319459Z","shell.execute_reply.started":"2022-02-14T21:13:14.288036Z","shell.execute_reply":"2022-02-14T21:13:14.318768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"df_test = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:13:14.320937Z","iopub.execute_input":"2022-02-14T21:13:14.32134Z","iopub.status.idle":"2022-02-14T21:13:19.339981Z","shell.execute_reply.started":"2022-02-14T21:13:14.321309Z","shell.execute_reply":"2022-02-14T21:13:19.338999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test[\"prediction\"]=\" \".join([str(x) for x in products])","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:13:19.341087Z","iopub.execute_input":"2022-02-14T21:13:19.341295Z","iopub.status.idle":"2022-02-14T21:13:19.35881Z","shell.execute_reply.started":"2022-02-14T21:13:19.34127Z","shell.execute_reply":"2022-02-14T21:13:19.357824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.to_csv(\"/kaggle/working/submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-14T21:13:19.360052Z","iopub.execute_input":"2022-02-14T21:13:19.360274Z","iopub.status.idle":"2022-02-14T21:13:32.015091Z","shell.execute_reply.started":"2022-02-14T21:13:19.360248Z","shell.execute_reply":"2022-02-14T21:13:32.013955Z"},"trusted":true},"execution_count":null,"outputs":[]}]}