{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Task 1 Recommenders: CF + Non Personalized Recommendation for H&M Challenge\n### Ana de Garay, Núria Camí\n### March 4, 2022\n\nPublic notebook via Kaggle: https://www.kaggle.com/nuriacami/hym-recommenders-notebook ","metadata":{}},{"cell_type":"markdown","source":"## RAPIDS cuDF ","metadata":{}},{"cell_type":"code","source":"'''\nImport RAPIDS cuDF in order to accelerate dataframe operations\n'''\nimport cudf\nprint('RAPIDS version',cudf.__version__)\n\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2022-03-07T14:45:45.367052Z","iopub.execute_input":"2022-03-07T14:45:45.367369Z","iopub.status.idle":"2022-03-07T14:45:48.730094Z","shell.execute_reply.started":"2022-03-07T14:45:45.367277Z","shell.execute_reply":"2022-03-07T14:45:48.727172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Read data ('transactions_train.csv') ","metadata":{}},{"cell_type":"code","source":"# load data\ntrain = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\n\n# set features to a specific format\ntrain['customer_id'] = train['customer_id'].str[-16:].str.hex_to_int().astype('int64')\ntrain['article_id'] = train.article_id.astype('int32')\ntrain.t_dat = cudf.to_datetime(train.t_dat)\n\n# keep only the desired columns\ntrain = train[['t_dat','customer_id','article_id']]\ntrain.to_parquet('train.pqt',index=False)\n\nprint(train.shape)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-07T14:46:05.563473Z","iopub.execute_input":"2022-03-07T14:46:05.564049Z","iopub.status.idle":"2022-03-07T14:46:45.235586Z","shell.execute_reply.started":"2022-03-07T14:46:05.564009Z","shell.execute_reply":"2022-03-07T14:46:45.234945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Some exploration of the dataset","metadata":{}},{"cell_type":"code","source":"print(\"Nº of distinct customers:\", train.customer_id.nunique())\nprint(\"Nº of distinct items:\", train.article_id.nunique())","metadata":{"execution":{"iopub.status.busy":"2022-03-04T17:52:04.686476Z","iopub.execute_input":"2022-03-04T17:52:04.687006Z","iopub.status.idle":"2022-03-04T17:52:04.770317Z","shell.execute_reply.started":"2022-03-04T17:52:04.686968Z","shell.execute_reply":"2022-03-04T17:52:04.769559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.customer_id.value_counts().describe()","metadata":{"execution":{"iopub.status.busy":"2022-03-04T17:52:21.131819Z","iopub.execute_input":"2022-03-04T17:52:21.132122Z","iopub.status.idle":"2022-03-04T17:52:21.221011Z","shell.execute_reply.started":"2022-03-04T17:52:21.132077Z","shell.execute_reply":"2022-03-04T17:52:21.22021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.article_id.value_counts().describe()","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:00:49.066555Z","iopub.execute_input":"2022-03-04T18:00:49.067025Z","iopub.status.idle":"2022-03-04T18:00:49.125884Z","shell.execute_reply.started":"2022-03-04T18:00:49.066985Z","shell.execute_reply":"2022-03-04T18:00:49.12518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# nº of times that a item has been purchased (without duplicates)\ntrain.groupby('article_id')['customer_id'].nunique().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:01:23.737973Z","iopub.execute_input":"2022-03-04T18:01:23.738654Z","iopub.status.idle":"2022-03-04T18:01:24.223609Z","shell.execute_reply.started":"2022-03-04T18:01:23.738619Z","shell.execute_reply":"2022-03-04T18:01:24.222906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# nº of distinct clients buying a product\ntrain.groupby('article_id')['customer_id'].nunique().describe()","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:02:09.793141Z","iopub.execute_input":"2022-03-04T18:02:09.794029Z","iopub.status.idle":"2022-03-04T18:02:10.274598Z","shell.execute_reply.started":"2022-03-04T18:02:09.793984Z","shell.execute_reply":"2022-03-04T18:02:10.273798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reduction of the dataset","metadata":{}},{"cell_type":"code","source":"# select only the items purchased 15 times or more\na = train.article_id.value_counts() >= 15\nf_items = a[a].index \n\n# select only the customers that have done at least 5 purchases\na = train.customer_id.value_counts() >= 5\nf_cust = a[a].index \n\n# filter by popular items\ndf_f1 = train[train.article_id.isin(f_items)].reset_index(drop=True)\n\n# filter by popular customers\ndf_f2 = df_f1[df_f1.customer_id.isin(f_cust)].reset_index(drop=True)\n\n# filter by last month\ndf_f3 = df_f2[df_f2.t_dat >= cudf.to_datetime('2020-08-25')].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:03:11.473593Z","iopub.execute_input":"2022-03-04T18:03:11.473853Z","iopub.status.idle":"2022-03-04T18:03:12.149728Z","shell.execute_reply.started":"2022-03-04T18:03:11.473825Z","shell.execute_reply":"2022-03-04T18:03:12.149005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_f3.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:03:18.940531Z","iopub.execute_input":"2022-03-04T18:03:18.940779Z","iopub.status.idle":"2022-03-04T18:03:18.962219Z","shell.execute_reply.started":"2022-03-04T18:03:18.940751Z","shell.execute_reply":"2022-03-04T18:03:18.961532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_f3.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:03:30.210026Z","iopub.execute_input":"2022-03-04T18:03:30.210649Z","iopub.status.idle":"2022-03-04T18:03:30.215881Z","shell.execute_reply.started":"2022-03-04T18:03:30.21061Z","shell.execute_reply":"2022-03-04T18:03:30.215227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Split Train and Validation sets","metadata":{}},{"cell_type":"code","source":"# recap that our reduced dataset has transactions from 25/08/2020 onwards\n\n# TRAIN: the first three weeks \ndf_train = df_f3[df_f3.t_dat <= cudf.to_datetime('2020-09-15')].reset_index(drop=True)\n\n# VAL: the last week\ndf_val = df_f3[df_f3.t_dat > cudf.to_datetime('2020-09-15')].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:04:40.928989Z","iopub.execute_input":"2022-03-04T18:04:40.929977Z","iopub.status.idle":"2022-03-04T18:04:40.95439Z","shell.execute_reply.started":"2022-03-04T18:04:40.929929Z","shell.execute_reply":"2022-03-04T18:04:40.953659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Nº of items to train:\", df_train.article_id.nunique())\nprint(\"Nº of customers to train:\", df_train.customer_id.nunique())","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:04:42.882541Z","iopub.execute_input":"2022-03-04T18:04:42.883096Z","iopub.status.idle":"2022-03-04T18:04:42.893183Z","shell.execute_reply.started":"2022-03-04T18:04:42.88304Z","shell.execute_reply":"2022-03-04T18:04:42.892256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Nº of items to validate:\", df_val.article_id.nunique())\nprint(\"Nº of customers to validate:\", df_val.customer_id.nunique())","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:04:45.533807Z","iopub.execute_input":"2022-03-04T18:04:45.534084Z","iopub.status.idle":"2022-03-04T18:04:45.541879Z","shell.execute_reply.started":"2022-03-04T18:04:45.534033Z","shell.execute_reply":"2022-03-04T18:04:45.541016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating Items x Customers matrix (n x m)","metadata":{}},{"cell_type":"code","source":"'''\nAvoid the duplicated transactions (items buyed by the same customer more than once)\n''' \ndf_train_g = df_train.groupby(['customer_id','article_id']).size().reset_index().rename(columns={0:'counts'})\ndf_train_g['counts_simple'] = 1","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:07:19.431186Z","iopub.execute_input":"2022-03-04T18:07:19.431516Z","iopub.status.idle":"2022-03-04T18:07:20.505707Z","shell.execute_reply.started":"2022-03-04T18:07:19.431482Z","shell.execute_reply":"2022-03-04T18:07:20.504979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_g.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:07:29.42469Z","iopub.execute_input":"2022-03-04T18:07:29.424949Z","iopub.status.idle":"2022-03-04T18:07:29.445601Z","shell.execute_reply.started":"2022-03-04T18:07:29.424921Z","shell.execute_reply":"2022-03-04T18:07:29.444891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nIn order to efficiently create our desired matrix, we map customer and article's ids into plain indexes\n''' \n\ncust_indexes = dict(df_train_g.customer_id.unique())\nitems_indexes = dict(df_train_g.article_id.unique())\n\ninv_map_cust = {v: k for k, v in cust_indexes.items()}\ninv_map_items = {v: k for k, v in items_indexes.items()}\n\ndf_train_g['customer_id'] = df_train_g['customer_id'].map(inv_map_cust)\ndf_train_g['article_id'] = df_train_g['article_id'].map(inv_map_items)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:07:34.330093Z","iopub.execute_input":"2022-03-04T18:07:34.330777Z","iopub.status.idle":"2022-03-04T18:21:20.246653Z","shell.execute_reply.started":"2022-03-04T18:07:34.330739Z","shell.execute_reply":"2022-03-04T18:21:20.245909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_g.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:22:47.500026Z","iopub.execute_input":"2022-03-04T18:22:47.500764Z","iopub.status.idle":"2022-03-04T18:22:47.523518Z","shell.execute_reply.started":"2022-03-04T18:22:47.500725Z","shell.execute_reply":"2022-03-04T18:22:47.522859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nMatrix of 1's and 0's indicating whether a customer has purchased an item or not, respectively\n''' \n\nimport numpy as np\n\nm = len(df_train_g.customer_id.unique())\nn = len(df_train_g.article_id.unique())\n\nic_matrix = np.zeros((n,m))\n\nfor i,row in df_train_g.to_pandas().iterrows():\n    ic_matrix[row['article_id'], row['customer_id']] = 1","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:22:57.914604Z","iopub.execute_input":"2022-03-04T18:22:57.914875Z","iopub.status.idle":"2022-03-04T18:23:31.558734Z","shell.execute_reply.started":"2022-03-04T18:22:57.914845Z","shell.execute_reply":"2022-03-04T18:23:31.557991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Shape of the final matrix: n x m\")\nprint(\"n (number of items) =\", n)\nprint(\"m (number of customers) =\", m)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:30:51.952938Z","iopub.execute_input":"2022-03-04T18:30:51.95364Z","iopub.status.idle":"2022-03-04T18:30:51.958875Z","shell.execute_reply.started":"2022-03-04T18:30:51.9536Z","shell.execute_reply":"2022-03-04T18:30:51.957901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(m*100)/train.customer_id.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:31:11.871835Z","iopub.execute_input":"2022-03-04T18:31:11.872279Z","iopub.status.idle":"2022-03-04T18:31:11.906887Z","shell.execute_reply.started":"2022-03-04T18:31:11.872243Z","shell.execute_reply":"2022-03-04T18:31:11.906203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Note that we have reduced the number of customers to a ~13% of the original dataset.","metadata":{}},{"cell_type":"markdown","source":"## Matrix reduction","metadata":{}},{"cell_type":"code","source":"'''\nRelation between nº of features and variance of the reducted matrix \n'''\n\nimport matplotlib.pyplot as plt\nfrom scipy.sparse import csr_matrix\nfrom sklearn.decomposition import TruncatedSVD\n\nK_list = [10,50,100,200,500,800]\nvar_list = []\n\nX_sparse = csr_matrix(ic_matrix)\n\nfor i in K_list:\n    svd = TruncatedSVD(n_components = i)\n    X_items = svd.fit_transform(X_sparse)    \n    var_list.append(sum(svd.fit(X_sparse).explained_variance_ratio_))","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:32:18.637367Z","iopub.execute_input":"2022-03-04T18:32:18.637628Z","iopub.status.idle":"2022-03-04T18:38:58.399502Z","shell.execute_reply.started":"2022-03-04T18:32:18.637599Z","shell.execute_reply":"2022-03-04T18:38:58.39875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nAs we can see, the variance increases along with the number of features\n'''\n\n# plot k vs. variance\nimport matplotlib.pyplot as plt\nfig1 = plt.figure(figsize=(8, 6))\nplot1 = fig1.add_subplot(111)\nplot1.plot(K_list,var_list,'green')\nplot1.set_title('Nº of features k vs. Variance')\nplot1.set_ylabel('Variance')\nplot1.set_xlabel('k')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:38:58.40117Z","iopub.execute_input":"2022-03-04T18:38:58.401418Z","iopub.status.idle":"2022-03-04T18:38:58.595514Z","shell.execute_reply.started":"2022-03-04T18:38:58.401381Z","shell.execute_reply":"2022-03-04T18:38:58.594847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"After some testings, we realized that 200 was the value of k that gives lower variance having the maximum performance, so we keep that value for our approach.","metadata":{}},{"cell_type":"code","source":"# optimum number of features\nK = 200\n\n# SVD decomposition\nX_sparse = csr_matrix(ic_matrix)\nsvd = TruncatedSVD(n_components = K)\nX_items = svd.fit_transform(X_sparse)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:39:50.524188Z","iopub.execute_input":"2022-03-04T18:39:50.524863Z","iopub.status.idle":"2022-03-04T18:40:36.311444Z","shell.execute_reply.started":"2022-03-04T18:39:50.524826Z","shell.execute_reply":"2022-03-04T18:40:36.310673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Shape of the reduced matrix: n x k\")\nprint(\"n (number of items) =\", n)\nprint(\"k(number of features) =\", K)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:41:19.661504Z","iopub.execute_input":"2022-03-04T18:41:19.661784Z","iopub.status.idle":"2022-03-04T18:41:19.667931Z","shell.execute_reply.started":"2022-03-04T18:41:19.661752Z","shell.execute_reply":"2022-03-04T18:41:19.667131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# saving our items matrix\nnp.save(\"matrix_items.npy\", X_items, allow_pickle=True, fix_imports=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:41:24.583684Z","iopub.execute_input":"2022-03-04T18:41:24.584424Z","iopub.status.idle":"2022-03-04T18:41:24.635066Z","shell.execute_reply.started":"2022-03-04T18:41:24.58438Z","shell.execute_reply":"2022-03-04T18:41:24.633912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Non-personalised top Recommendations","metadata":{}},{"cell_type":"markdown","source":"Our first approach was to simply recommend to the all \"non-seen\" customers the same 12 items, corresponding to the 12 items more times purchased. As this was a very naive approach, we tried to improve it taking the idea from the \"Time Decaying Popularity Benchmark\" notebook. The idea of this second approach was to create a popularity factor for each item that gave more \"weight\" to that items more recently purchased. Then, all the score given for each transaction is added for each item and that leads to a ranking of the most popular items.","metadata":{}},{"cell_type":"code","source":"import datetime\n\n# popularity factor\ndf_train['pop_factor'] = df_train.to_pandas()['t_dat'].apply(lambda x: 1/(datetime.datetime(2020,9,16) - x).days)\n\n# sum all the factors for every article\npopular_items_group = df_train.groupby(['article_id'])['pop_factor'].sum()\n\n# keep only the first 12\ntop_12_items_2 = popular_items_group.sort_values(ascending=False).index.values[:12]","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:43:17.011594Z","iopub.execute_input":"2022-03-04T18:43:17.012415Z","iopub.status.idle":"2022-03-04T18:43:30.435616Z","shell.execute_reply.started":"2022-03-04T18:43:17.012366Z","shell.execute_reply":"2022-03-04T18:43:30.434882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nRanking of the 12 most popular items according to popularity factor\n'''\ntop_12_items_2","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:43:39.714522Z","iopub.execute_input":"2022-03-04T18:43:39.714809Z","iopub.status.idle":"2022-03-04T18:43:39.722639Z","shell.execute_reply.started":"2022-03-04T18:43:39.714775Z","shell.execute_reply":"2022-03-04T18:43:39.721857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We decided to do this implementation because at the end was the one that gave us better performance. Another approach that was discarted was similary to this one. The other one consisted in keeping the top 25 of popular items according to this ranking and recommend for every \"non-seen\" user, firstly, the top 5, and secondly, from among the products from the 6th to the 25th position, choose randomly the 7 left. However, as we said, we discarted it as it was not giving a better score. ","metadata":{}},{"cell_type":"markdown","source":"## Collaborative filtering","metadata":{}},{"cell_type":"code","source":"'''\nAssignation of the predicted articles for each user \n'''\n\nimport random\n\n# for each user\ndef get_top_preds(customer, top_popular, top_k=12):\n    \n    # we look if it made purchases\n    items_idx = df_train[df_train.customer_id == customer].article_id.map(inv_map_items).dropna().unique().values.get()\n    \n    # if not, is a non-personalized case (already explained)\n    if len(items_idx) == 0:\n        top_items = top_popular\n    \n    # if it made purchases\n    else:\n        c_vector = np.mean(X_items[items_idx, :], axis=0 )\n\n        # build prediction vector per item from customer vector\n        p_vector = np.dot(X_items, np.array([c_vector]).T).flatten()\n\n        # get item affinity vector for the customer\n        \n        # keep only those items that are affine enough for the customer\n        items_flt_aff = np.array([(i,aff) for i,aff in enumerate(p_vector) if aff > 0.2])\n        \n        # for an affinity < 0.2, we consider that user as a \"non-seen\" one\n        if len(items_flt_aff) == 0:\n            top_items = top_popular\n            \n        # if the affinity is >= 0.2 \n        else:\n        \n            # we get indexes of top k affine items to the user\n            items_idx_0 = items_flt_aff[:,1].argsort()[-top_k:][::-1]\n            items_idx_1 = items_flt_aff[:,0][items_idx_0]\n\n            # we translate item index to item id\n            top_items = np.array([items_indexes[i] for i in items_idx_1])\n            # if our top items are less than k, we fill the last positions with firsts positions of top popular products \n            l = len(top_items)\n\n            if l<top_k:\n                r = top_k - l\n                top_items = list(top_items) + list(top_popular[:r])\n\n    return top_items","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:47:47.820459Z","iopub.execute_input":"2022-03-04T18:47:47.821394Z","iopub.status.idle":"2022-03-04T18:47:47.832696Z","shell.execute_reply.started":"2022-03-04T18:47:47.821329Z","shell.execute_reply":"2022-03-04T18:47:47.831814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Validation","metadata":{}},{"cell_type":"code","source":"'''\nFunctions to score the validation\n'''\n\ndef apk(actual, predicted, k=12):\n    if len(predicted)>k:\n        predicted = predicted[:k]\n\n    score = 0.0\n    num_hits = 0.0\n\n    for i,p in enumerate(predicted):\n        if p in actual and p not in predicted[:i]:\n            num_hits += 1.0\n            score += num_hits / (i+1.0)\n\n    if not actual:\n        return 0.0\n\n    return score / min(len(actual), k)\n\ndef mapk(actual, predicted, k=12):\n    return np.mean([apk(a,p,k) for a,p in zip(actual, predicted)])","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:47:59.165708Z","iopub.execute_input":"2022-03-04T18:47:59.166261Z","iopub.status.idle":"2022-03-04T18:47:59.173222Z","shell.execute_reply.started":"2022-03-04T18:47:59.166221Z","shell.execute_reply":"2022-03-04T18:47:59.172178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get validation customers and items from validation dataset\nval_customers = df_val.to_pandas().groupby(['customer_id'])['article_id'].apply(list).index.values\nval_items = df_val.to_pandas().groupby(['customer_id'])['article_id'].apply(list).values\n\noutputs = []\n\n# do the prediction\nfor customer in val_customers:\n    outputs.append(get_top_preds(customer, top_12_items_2, top_k=12))\n\n#save output\nnp.save(\"validation_pred.npy\", np.array(outputs), allow_pickle=True, fix_imports=True)\n\n# score\nprint(\"mAP Score on Validation set:\", mapk(val_items, outputs))","metadata":{"execution":{"iopub.status.busy":"2022-03-04T10:20:28.372028Z","iopub.status.idle":"2022-03-04T10:20:28.373037Z","shell.execute_reply.started":"2022-03-04T10:20:28.372729Z","shell.execute_reply":"2022-03-04T10:20:28.372763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"At this point we got a score of 0.014 for the validation set. Now we can make the submission.","metadata":{}},{"cell_type":"markdown","source":"## Making a submission","metadata":{}},{"cell_type":"code","source":"'''\nRead the submission file and convert columns into an specific format\n'''\nsubmission = cudf.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv\")\n\nsubmission['customer_id'] = submission['customer_id'].str[-16:].str.hex_to_int().astype('int64')\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:49:11.368892Z","iopub.execute_input":"2022-03-04T18:49:11.369714Z","iopub.status.idle":"2022-03-04T18:49:14.381467Z","shell.execute_reply.started":"2022-03-04T18:49:11.36967Z","shell.execute_reply":"2022-03-04T18:49:14.380744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nDo the prediction for each user of the submissions file \n'''\n\nfrom tqdm import tqdm\n\nsub_customers = submission.to_pandas().customer_id.values\noutputs_sub = []\n\nfor customer in tqdm(sub_customers):\n    outputs_sub.append(get_top_preds(customer, top_12_items_2, top_k=12))\n    \nstr_outputs = []\nfor output in outputs_sub:\n    str_outputs.append(\" \".join([str(x) for x in output]))","metadata":{"execution":{"iopub.status.busy":"2022-03-04T10:20:28.37641Z","iopub.status.idle":"2022-03-04T10:20:28.376837Z","shell.execute_reply.started":"2022-03-04T10:20:28.376636Z","shell.execute_reply":"2022-03-04T10:20:28.376666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nSave the results into a csv file\n'''\n\n# duplicate the submission dataframe and store there the predictions \npreds = submission#[:N]\npreds['prediction'] = str_outputs\n\n# from the submission file, keep only the customer_id values and replace the articles column with the predictions\nsubmission = submission[['customer_id']]\n# then merge both dataframes and delete the customer_id column leftover\nsubmission = submission.merge(preds.rename({'customer_id':'customer_id_2'},axis=1),\\\n    on='customer_id_2', how='left').fillna('')\ndel submission['customer_id_2']","metadata":{"execution":{"iopub.status.busy":"2022-03-04T10:20:28.378522Z","iopub.status.idle":"2022-03-04T10:20:28.379134Z","shell.execute_reply.started":"2022-03-04T10:20:28.378884Z","shell.execute_reply":"2022-03-04T10:20:28.378918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nIteration to put a 0 at the beggining of each article_id \n'''\n# problem: this took 27h and it was unaffordable to run. \n\n# for i in tqdm(range(0,submission.shape[0])):\n#     aux = submission['prediction'][i][:0] + '0' + submission['prediction'][i][0:]\n#     aux = aux[:11] + '0' + aux[11:]\n#     aux = aux[:22] + '0' + aux[22:]\n#     aux = aux[:33] + '0' + aux[33:]\n#     aux = aux[:44] + '0' + aux[44:]\n#     aux = aux[:55] + '0' + aux[55:]\n#     aux = aux[:66] + '0' + aux[66:]\n#     aux = aux[:77] + '0' + aux[77:]\n#     aux = aux[:88] + '0' + aux[88:]\n#     aux = aux[:99] + '0' + aux[99:]\n#     aux = aux[:110] + '0' + aux[110:]\n#     submission['prediction'][i] = aux[:121] + '0' + aux[121:]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nSave the submission to csv file\n'''\nsubmission.to_csv('submission_04032022.csv',index=False)\nsubmission.head()","metadata":{},"execution_count":null,"outputs":[]}]}