{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook is copied from \nhttps://www.kaggle.com/cdeotte/customers-who-bought-this-frequently-buy-this \n\nThen change the code to generate item-pair dictionary in purpose to calculate MAP@12 of validation data in https://www.kaggle.com/hervind/h-m-calculate-map-12-from-public-notebooks using algorithm in https://www.kaggle.com/cdeotte/recommend-items-purchased-together-0-021\n\nSince the data is to large and takes ~7 hours to compute in Kaggle Kernels, I reduce the data period only last 3 months before validation date. \n\nSo it will only get item-pairs of the transaction between '2020-06-16' and '2020-09-15'","metadata":{}},{"cell_type":"markdown","source":"# Customers Who Bought This Frequently Buy This!\nIn this notebook we will explore which items were frequently purchased together. Using this information, we can predict which items a customer will buy after we observe what they have already bought!","metadata":{}},{"cell_type":"code","source":"import cudf, gc\nimport cv2, matplotlib.pyplot as plt\nfrom os.path import exists\nprint('RAPIDS version',cudf.__version__)\n\nfrom tqdm import tqdm\nimport multiprocessing as mp\nfrom multiprocessing import Pool\nfrom functools import partial\nimport numpy as np ","metadata":{"execution":{"iopub.status.busy":"2022-02-27T05:47:23.294669Z","iopub.execute_input":"2022-02-27T05:47:23.294987Z","iopub.status.idle":"2022-02-27T05:47:27.104780Z","shell.execute_reply.started":"2022-02-27T05:47:23.294906Z","shell.execute_reply":"2022-02-27T05:47:27.103905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Transactions","metadata":{}},{"cell_type":"code","source":"# LOAD TRANSACTIONS DATAFRAME\ndf = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\nprint('Transactions shape',df.shape)\n\n# REDUCE MEMORY OF DATAFRAME\ndf = df.loc[((df['t_dat'] >= '2020-06-16') & (df['t_dat'] < '2020-09-16'))] \ndf = df[['customer_id','article_id']]\ndf.customer_id = df.customer_id.str[-16:].str.hex_to_int().astype('int64')\ndf.article_id = df.article_id.astype('int32')\nprint('Reduced Transactions shape',df.shape)\ndisplay( df.head() )\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-02-27T05:47:27.108871Z","iopub.execute_input":"2022-02-27T05:47:27.109126Z","iopub.status.idle":"2022-02-27T05:48:04.346630Z","shell.execute_reply.started":"2022-02-27T05:47:27.109093Z","shell.execute_reply":"2022-02-27T05:48:04.345863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Created by my notebook here. Just change the line pairs[i.item()] = [vc2.index[0], vc2.index[1], vc2.index[2]] into pairs[i.item()] = vc2.index[0] and run the for-loop over the entire list of items with for j,i in enumerate(vc.index.values):","metadata":{}},{"cell_type":"markdown","source":"# Find Items Purchased Together\nWe will use RAPID cuDF to speed up the dataframe search commands below","metadata":{}},{"cell_type":"code","source":"# FIND ITEMS PURCHASED TOGETHER\nvc = df.article_id.value_counts()\nlist_articles = vc.index.values\n\npairs = {}\nfor i in tqdm(list_articles):\n    df_u_temp = df.loc[df.article_id==i.item(),'customer_id']\n    if len(df_u_temp) > 1: \n        USERS = df_u_temp.unique()\n        df_filter = df.loc[(df.customer_id.isin(USERS))&(df.article_id!=i.item()),'article_id']\n        if len(df_filter) == 1: \n            pairs[i.item()] = df_filter.values[0][0].item()\n        elif len(df_filter) == 0: \n            continue\n        else:     \n            vc2 = df_filter.value_counts()    \n            pairs[i.item()] = vc2.index[0]","metadata":{"execution":{"iopub.status.busy":"2022-02-27T05:48:15.004123Z","iopub.execute_input":"2022-02-27T05:48:15.004397Z","iopub.status.idle":"2022-02-27T06:21:38.207100Z","shell.execute_reply.started":"2022-02-27T05:48:15.004347Z","shell.execute_reply":"2022-02-27T06:21:38.206414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.save(\"item_pair_3_months.npy\", pairs)","metadata":{"execution":{"iopub.status.busy":"2022-02-27T06:30:41.518468Z","iopub.execute_input":"2022-02-27T06:30:41.518760Z","iopub.status.idle":"2022-02-27T06:30:41.645953Z","shell.execute_reply.started":"2022-02-27T06:30:41.518728Z","shell.execute_reply":"2022-02-27T06:30:41.645232Z"},"trusted":true},"execution_count":null,"outputs":[]}]}