{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Customers Who Bought This Frequently Buy This!\nIn this notebook we will explore which items were frequently purchased together. Using this information, we can predict which items a customer will buy after we observe what they have already bought!","metadata":{}},{"cell_type":"code","source":"import cudf, gc\nimport pandas as pd\nimport numpy as np\nimport cv2, matplotlib.pyplot as plt\nfrom os.path import exists\nprint('RAPIDS version',cudf.__version__)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T02:42:59.263672Z","iopub.execute_input":"2022-04-12T02:42:59.263967Z","iopub.status.idle":"2022-04-12T02:43:03.451392Z","shell.execute_reply.started":"2022-04-12T02:42:59.263934Z","shell.execute_reply":"2022-04-12T02:43:03.450669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def apk(actual, predicted, k=10):\n    \"\"\"\n    Computes the average precision at k.\n    This function computes the average prescision at k between two lists of\n    items.\n    Parameters\n    ----------\n    actual : list\n             A list of elements that are to be predicted (order doesn't matter)\n    predicted : list\n                A list of predicted elements (order does matter)\n    k : int, optional\n        The maximum number of predicted elements\n    Returns\n    -------\n    score : double\n            The average precision at k over the input lists\n    \"\"\"\n    if len(predicted)>k:\n        predicted = predicted[:k]\n\n    score = 0.0\n    num_hits = 0.0\n\n    for i,p in enumerate(predicted):\n        if p in actual and p not in predicted[:i]:\n            num_hits += 1.0\n            score += num_hits / (i+1.0)\n\n    if not actual:\n        return 0.0\n\n    return score / min(len(actual), k)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T02:43:03.453095Z","iopub.execute_input":"2022-04-12T02:43:03.453346Z","iopub.status.idle":"2022-04-12T02:43:03.460066Z","shell.execute_reply.started":"2022-04-12T02:43:03.453312Z","shell.execute_reply":"2022-04-12T02:43:03.45937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mapk(actual, predicted, k=10):\n    \"\"\"\n    Computes the mean average precision at k.\n    This function computes the mean average prescision at k between two lists\n    of lists of items.\n    Parameters\n    ----------\n    actual : list\n             A list of lists of elements that are to be predicted \n             (order doesn't matter in the lists)\n    predicted : list\n                A list of lists of predicted elements\n                (order matters in the lists)\n    k : int, optional\n        The maximum number of predicted elements\n    Returns\n    -------\n    score : double\n            The mean average precision at k over the input lists\n    \"\"\"\n    return np.mean([apk(a,p,k) for a,p in zip(actual, predicted)])","metadata":{"execution":{"iopub.status.busy":"2022-04-12T02:43:03.461525Z","iopub.execute_input":"2022-04-12T02:43:03.462015Z","iopub.status.idle":"2022-04-12T02:43:03.474052Z","shell.execute_reply.started":"2022-04-12T02:43:03.461978Z","shell.execute_reply":"2022-04-12T02:43:03.473239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-04-12T02:43:03.476498Z","iopub.execute_input":"2022-04-12T02:43:03.476838Z","iopub.status.idle":"2022-04-12T02:43:03.618347Z","shell.execute_reply.started":"2022-04-12T02:43:03.476798Z","shell.execute_reply":"2022-04-12T02:43:03.617558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Tranactions","metadata":{}},{"cell_type":"code","source":"# LOAD TRANSACTIONS DATAFRAME\n\ndf = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\nprint('Transactions shape',df.shape)\ndisplay( df.head() )\n\n# REDUCE MEMORY OF DATAFRAME\ndf = df[['customer_id','article_id']]\n# df.customer_id = df.customer_id.str[-16:].str.hex_to_int().astype('int64')\n# df.article_id = df.article_id.astype('int32')\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-04-12T02:43:03.619529Z","iopub.execute_input":"2022-04-12T02:43:03.619789Z","iopub.status.idle":"2022-04-12T02:44:01.397074Z","shell.execute_reply.started":"2022-04-12T02:43:03.619755Z","shell.execute_reply":"2022-04-12T02:44:01.396332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nvc = df.article_id.value_counts()\npairs = {}\nfor article in vc.index.values[:100]:\n    users = df.loc[df.article_id==article.item(), 'customer_id'].unique()\n    vc2 = df.loc[(df.customer_id.isin(users))&(df.article_id!=article.item()),'article_id'].value_counts()\n    pairs[article.item()] = [vc2.index[0], vc2.index[1], vc2.index[2]]\nprint(len(pairs))","metadata":{"execution":{"iopub.status.busy":"2022-04-12T02:44:01.398433Z","iopub.execute_input":"2022-04-12T02:44:01.398704Z","iopub.status.idle":"2022-04-12T02:48:43.066335Z","shell.execute_reply.started":"2022-04-12T02:44:01.398671Z","shell.execute_reply":"2022-04-12T02:48:43.065593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv',dtype={'article_id':str})\ntrain.t_dat = pd.to_datetime( train.t_dat )","metadata":{"execution":{"iopub.status.busy":"2022-04-12T02:48:43.067696Z","iopub.execute_input":"2022-04-12T02:48:43.06794Z","iopub.status.idle":"2022-04-12T02:49:24.454563Z","shell.execute_reply.started":"2022-04-12T02:48:43.067907Z","shell.execute_reply":"2022-04-12T02:49:24.453705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\nvalid.t_dat = pd.to_datetime( valid.t_dat )\nvalid = valid.loc[ valid.t_dat >= pd.to_datetime('2020-09-16') ]\nvalid = valid.groupby('customer_id').article_id.apply(list).reset_index()\nvalid = valid.rename({'article_id':'prediction'},axis=1)\nvalid['prediction'] =\\\n    valid.prediction.apply(lambda x: ' '.join(['0'+str(k) for k in x]))","metadata":{"execution":{"iopub.status.busy":"2022-04-12T02:49:24.458246Z","iopub.execute_input":"2022-04-12T02:49:24.458974Z","iopub.status.idle":"2022-04-12T02:50:06.101558Z","shell.execute_reply.started":"2022-04-12T02:49:24.458932Z","shell.execute_reply":"2022-04-12T02:50:06.100699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transaction_3w = train[train['t_dat'] >= pd.to_datetime('2020-08-24')].copy()\ntransaction_2w = train[train['t_dat'] >= pd.to_datetime('2020-08-31')].copy()\ntransaction_1w = train[train['t_dat'] >= pd.to_datetime('2020-09-07')].copy()","metadata":{"execution":{"iopub.status.busy":"2022-04-12T02:50:06.103056Z","iopub.execute_input":"2022-04-12T02:50:06.103335Z","iopub.status.idle":"2022-04-12T02:50:06.579091Z","shell.execute_reply.started":"2022-04-12T02:50:06.1033Z","shell.execute_reply":"2022-04-12T02:50:06.578353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#submission\ntransaction_3w = train[train['t_dat'] >= pd.to_datetime('2020-08-31')].copy()\ntransaction_2w = train[train['t_dat'] >= pd.to_datetime('2020-09-07')].copy()\ntransaction_1w = train[train['t_dat'] >= pd.to_datetime('2020-09-14')].copy()","metadata":{"execution":{"iopub.status.busy":"2022-04-12T02:50:06.581749Z","iopub.execute_input":"2022-04-12T02:50:06.582071Z","iopub.status.idle":"2022-04-12T02:50:06.956316Z","shell.execute_reply.started":"2022-04-12T02:50:06.582034Z","shell.execute_reply":"2022-04-12T02:50:06.95559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"purchase_dict_3w = {}\n\nfor i,x in enumerate(zip(transaction_3w['customer_id'], transaction_3w['article_id'])):\n    cust_id, art_id = x\n    if cust_id not in purchase_dict_3w:\n        purchase_dict_3w[cust_id] = {}\n        \n    if art_id not in purchase_dict_3w:\n        purchase_dict_3w[cust_id][art_id] = 0\n    \n    purchase_dict_3w[cust_id][art_id] += 1\n\nprint(len(purchase_dict_3w))\n\ndummy_list_3w = list((transaction_3w['article_id'].value_counts()).index)[:12]\n\nprint(dummy_list_3w)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T02:50:06.957492Z","iopub.execute_input":"2022-04-12T02:50:06.958105Z","iopub.status.idle":"2022-04-12T02:50:08.004969Z","shell.execute_reply.started":"2022-04-12T02:50:06.958061Z","shell.execute_reply":"2022-04-12T02:50:08.004223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"purchase_dict_2w = {}\n\nfor i,x in enumerate(zip(transaction_2w['customer_id'], transaction_2w['article_id'])):\n    cust_id, art_id = x\n    if cust_id not in purchase_dict_2w:\n        purchase_dict_2w[cust_id] = {}\n        \n    if art_id not in purchase_dict_2w:\n        purchase_dict_2w[cust_id][art_id] = 0\n    \n    purchase_dict_2w[cust_id][art_id] += 1\n\nprint(len(purchase_dict_2w))\n\ndummy_list_2w = list((transaction_2w['article_id'].value_counts()).index)[:12]\n\nprint(dummy_list_2w)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T02:50:08.006245Z","iopub.execute_input":"2022-04-12T02:50:08.006532Z","iopub.status.idle":"2022-04-12T02:50:08.964839Z","shell.execute_reply.started":"2022-04-12T02:50:08.006487Z","shell.execute_reply":"2022-04-12T02:50:08.963265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"purchase_dict_1w = {}\n\nfor i,x in enumerate(zip(transaction_1w['customer_id'], transaction_1w['article_id'])):\n    cust_id, art_id = x\n    if cust_id not in purchase_dict_1w:\n        purchase_dict_1w[cust_id] = {}\n        \n    if art_id not in purchase_dict_1w:\n        purchase_dict_1w[cust_id][art_id] = 0\n    \n    purchase_dict_1w[cust_id][art_id] += 1\n\nprint(len(purchase_dict_1w))\n\ndummy_list_1w = list((transaction_1w['article_id'].value_counts()).index)[:12]\n\nprint(dummy_list_1w)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T02:50:08.966257Z","iopub.execute_input":"2022-04-12T02:50:08.966512Z","iopub.status.idle":"2022-04-12T02:50:09.340043Z","shell.execute_reply.started":"2022-04-12T02:50:08.966478Z","shell.execute_reply":"2022-04-12T02:50:09.339231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv')\nbenchmark = submission[['customer_id']]\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-12T02:50:09.341291Z","iopub.execute_input":"2022-04-12T02:50:09.341555Z","iopub.status.idle":"2022-04-12T02:50:13.740174Z","shell.execute_reply.started":"2022-04-12T02:50:09.341509Z","shell.execute_reply":"2022-04-12T02:50:13.73947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(list(pairs.keys())[0])","metadata":{"execution":{"iopub.status.busy":"2022-04-12T02:50:13.741423Z","iopub.execute_input":"2022-04-12T02:50:13.741679Z","iopub.status.idle":"2022-04-12T02:50:13.747025Z","shell.execute_reply.started":"2022-04-12T02:50:13.741651Z","shell.execute_reply":"2022-04-12T02:50:13.746363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rec_bought_together(l,hit):\n    res = l\n    for y in res:\n        y = int(y)\n        if y in pairs:\n            hit += 1\n            res = [str(pairs[y][0])] + res\n    return res,hit","metadata":{"execution":{"iopub.status.busy":"2022-04-12T02:50:13.74835Z","iopub.execute_input":"2022-04-12T02:50:13.748795Z","iopub.status.idle":"2022-04-12T02:50:13.757499Z","shell.execute_reply.started":"2022-04-12T02:50:13.748754Z","shell.execute_reply":"2022-04-12T02:50:13.756751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_list = []\n\ndummy_pred = ' '.join(dummy_list_1w)\n\ncnt = 0\nhit = 0\nfor i, cust_id in enumerate(submission['customer_id'].values.reshape((-1,))):\n    \n    if cust_id in purchase_dict_1w:\n        l = sorted(purchase_dict_1w[cust_id].items(), key = lambda x: x[1], reverse = True)\n        l = [y[0] for y in l]\n        #l,hit = rec_bought_together(l,hit)\n\n        if len(l) > 12:\n            s = ' '.join(l[:12])\n        else:\n            s = ' '.join(l+dummy_list_1w[:(12 - len(l))])\n\n    elif cust_id in purchase_dict_2w:\n        l = sorted(purchase_dict_2w[cust_id].items(), key = lambda x: x[1], reverse = True)\n        l = [y[0] for y in l]\n        #l,hit = rec_bought_together(l,hit)\n        if len(l) > 12:\n            s = ' '.join(l[:12])\n        else:\n            s = ' '.join(l+dummy_list_2w[:(12 - len(l))])\n        \n    elif cust_id in purchase_dict_3w:\n        l = sorted(purchase_dict_3w[cust_id].items(), key = lambda x: x[1], reverse = True)\n        l = [y[0] for y in l]\n        #l,hit = rec_bought_together(l,hit)\n        if len(l) > 12:\n            s = ' '.join(l[:12])\n        else:\n            s = ' '.join(l+dummy_list_3w[:(12 - len(l))])\n    \n    else:\n        cnt += 1\n        s = dummy_pred\n    \n    #try using dummy\n    #s = dummy_pred\n    prediction_list.append(s)\n\n\nbenchmark['prediction'] = prediction_list\nprint(benchmark.shape)\nprint(cnt,hit)\nbenchmark.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-12T02:51:47.355208Z","iopub.execute_input":"2022-04-12T02:51:47.355675Z","iopub.status.idle":"2022-04-12T02:51:49.231507Z","shell.execute_reply.started":"2022-04-12T02:51:47.355629Z","shell.execute_reply":"2022-04-12T02:51:49.230802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"benchmark.to_csv('benchmark040604.csv', index = False)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-04-12T02:51:52.115886Z","iopub.execute_input":"2022-04-12T02:51:52.116738Z","iopub.status.idle":"2022-04-12T02:52:03.13549Z","shell.execute_reply.started":"2022-04-12T02:51:52.116697Z","shell.execute_reply":"2022-04-12T02:52:03.134629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Prediction Score\nsub = pd.read_csv('./benchmark040604.csv')\nsub = sub.set_index('customer_id').loc[valid.customer_id].reset_index()\nmapk( valid.prediction.str.split(), sub.prediction.str.split(), k=12)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T02:52:03.137228Z","iopub.execute_input":"2022-04-12T02:52:03.137662Z","iopub.status.idle":"2022-04-12T02:52:06.825142Z","shell.execute_reply.started":"2022-04-12T02:52:03.137619Z","shell.execute_reply":"2022-04-12T02:52:06.824402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"0.84\n0.9148 before l\n0.8806","metadata":{"execution":{"iopub.status.busy":"2022-04-12T02:50:30.732482Z","iopub.execute_input":"2022-04-12T02:50:30.73275Z","iopub.status.idle":"2022-04-12T02:50:30.738726Z","shell.execute_reply.started":"2022-04-12T02:50:30.732716Z","shell.execute_reply":"2022-04-12T02:50:30.737721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"using 400 most frequent:\nwith all 3 frequent item: 0.82277\n\nwith first one: 0.82277","metadata":{}}]}