{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Customers Who Bought This Frequently Buy This!\nIn this notebook we will explore which items were frequently purchased together. Using this information, we can predict which items a customer will buy after we observe what they have already bought!","metadata":{}},{"cell_type":"code","source":"import cudf, gc\nimport cv2, matplotlib.pyplot as plt\nfrom os.path import exists\nprint('RAPIDS version',cudf.__version__)","metadata":{"execution":{"iopub.status.busy":"2022-06-05T11:23:31.24041Z","iopub.execute_input":"2022-06-05T11:23:31.241096Z","iopub.status.idle":"2022-06-05T11:23:32.817058Z","shell.execute_reply.started":"2022-06-05T11:23:31.240996Z","shell.execute_reply":"2022-06-05T11:23:32.814993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Tranactions","metadata":{}},{"cell_type":"code","source":"# LOAD TRANSACTIONS DATAFRAME\ndf = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\nprint('Transactions shape',df.shape)\ndisplay( df.head() )\ndate = df[['t_dat']]\ndisplay(date.head())\n# REDUCE MEMORY OF DATAFRAME\ndf = df[['customer_id','article_id']]\ndf.customer_id = df.customer_id.str[-16:].str.hex_to_int().astype('int64')\ndf.article_id = df.article_id.astype('int32')\ndisplay(df.head())\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-05T11:23:32.818498Z","iopub.execute_input":"2022-06-05T11:23:32.819866Z","iopub.status.idle":"2022-06-05T11:23:36.920167Z","shell.execute_reply.started":"2022-06-05T11:23:32.819823Z","shell.execute_reply":"2022-06-05T11:23:36.919427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Find Items Purchased Together\nWe will use RAPID cuDF to speed up the dataframe search commands below","metadata":{}},{"cell_type":"code","source":"# FIND ITEMS PURCHASED TOGETHER\nvc = df.article_id.value_counts()\npairs = {}\nfor j,i in enumerate(vc.index.values[1000:1050]):\n    #if j%10==0: print(j,', ',end='')\n    USERS = df.loc[df.article_id==i.item(),'customer_id'].unique()\n    vc2 = df.loc[(df.customer_id.isin(USERS))&(df.article_id!=i.item()),'article_id'].value_counts()\n    pairs[i.item()] = [vc2.index[0], vc2.index[1], vc2.index[2]]\n    ","metadata":{"execution":{"iopub.status.busy":"2022-06-05T11:23:36.921437Z","iopub.execute_input":"2022-06-05T11:23:36.921877Z","iopub.status.idle":"2022-06-05T11:23:49.399297Z","shell.execute_reply.started":"2022-06-05T11:23:36.921829Z","shell.execute_reply":"2022-06-05T11:23:49.398381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"items = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv')\nBASE = '../input/h-and-m-personalized-fashion-recommendations/images/'\n\nfor i,(k,v) in enumerate( pairs.items() ):\n    name1 = BASE+'0'+str(k)[:2]+'/0'+str(k)+'.jpg'\n    name2 = BASE+'0'+str(v[0])[:2]+'/0'+str(v[0])+'.jpg'\n    name3 = BASE+'0'+str(v[1])[:2]+'/0'+str(v[1])+'.jpg'\n    name4 = BASE+'0'+str(v[2])[:2]+'/0'+str(v[2])+'.jpg'\n    if exists(name1) & exists(name2) & exists(name3) & exists(name4):\n        plt.figure(figsize=(20,5))\n        img1 = cv2.imread(name1)[:,:,::-1]\n        img2 = cv2.imread(name2)[:,:,::-1]\n        img3 = cv2.imread(name3)[:,:,::-1]\n        img4 = cv2.imread(name4)[:,:,::-1]\n        plt.subplot(1,4,1)\n        plt.title('When customers buy this',size=18)\n        plt.imshow(img1)\n        plt.subplot(1,4,2)\n        plt.title('They buy this',size=18)\n        plt.imshow(img2)\n        plt.subplot(1,4,3)\n        plt.title('They buy this',size=18)\n        plt.imshow(img3)\n        plt.subplot(1,4,4)\n        plt.title('They buy this',size=18)\n        plt.imshow(img4)\n        \n        plt.show()\n    #if i==63: break","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-06-05T11:23:49.405605Z","iopub.execute_input":"2022-06-05T11:23:49.408385Z","iopub.status.idle":"2022-06-05T11:24:47.324477Z","shell.execute_reply.started":"2022-06-05T11:23:49.406023Z","shell.execute_reply":"2022-06-05T11:24:47.323782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Matrix = [[] for i in range(100)]\n_ = gc.collect()\ndt = []\nfor m, i in enumerate(vc.index.values[0:100]):\n    dt.append(df.loc[(df.article_id == i.item()),'customer_id'].unique())\nfor m, i in enumerate(vc.index.values[0:100]):\n    for n, j in enumerate(vc.index.values[0:100]):        \n        d = cudf.concat([dt[m], dt[n]]).value_counts()\n        Matrix[m].append(len(d.loc[d.values == 2]))\n","metadata":{"execution":{"iopub.status.busy":"2022-06-05T11:24:47.325755Z","iopub.execute_input":"2022-06-05T11:24:47.326446Z","iopub.status.idle":"2022-06-05T11:26:16.729367Z","shell.execute_reply.started":"2022-06-05T11:24:47.326405Z","shell.execute_reply":"2022-06-05T11:26:16.728645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport matplotlib.pyplot as plt\nM = [[[Matrix[i][j],0,0] for i in range(10)] for j in range(10)]# change 10 to any number < 100\n\nfrom PIL import Image\nimport numpy as np\nM = np.array(M)\nM = M / 20000.\nplt.imshow(M)","metadata":{"execution":{"iopub.status.busy":"2022-06-05T11:26:16.730518Z","iopub.execute_input":"2022-06-05T11:26:16.731884Z","iopub.status.idle":"2022-06-05T11:26:16.904239Z","shell.execute_reply.started":"2022-06-05T11:26:16.731841Z","shell.execute_reply":"2022-06-05T11:26:16.903563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Get the article_freq\nlabel_list = range(len(vc.values[:50000:1000]))\nlabel_list = np.array(label_list)\ncnt_list = list(int(i) for i in vc.values[:50000:1000])\n#print(cnt_list)\ncnt_list = np.array(cnt_list)\nrects1 = plt.bar(x=label_list, height=cnt_list, width=0.3, alpha=1., color='blue')\nplt.ylim(0, 60000)     # y轴取值范围\nplt.ylabel(\"\")\n\"\"\"\n设置x轴刻度显示值\n参数一：中点坐标\n参数二：显示值\n\"\"\"\nplt.xticks([index for index in label_list][::20], label_list[::20]*1000)\n\n#plt.xlabel(\"年份\")\nplt.title(\"frequency_distribution\")\n#plt.legend()     # 设置题注\n# 编辑文本\n#for rect in rects1:\n#    height = rect.get_height()\n#    plt.text(rect.get_x() + rect.get_width() / 2, height+1, str(height), ha=\"center\", va=\"bottom\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-05T11:26:16.905669Z","iopub.execute_input":"2022-06-05T11:26:16.906168Z","iopub.status.idle":"2022-06-05T11:26:17.12681Z","shell.execute_reply.started":"2022-06-05T11:26:16.906129Z","shell.execute_reply":"2022-06-05T11:26:17.126166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"date_collect = date['t_dat'].value_counts()\ndates = date['t_dat'].unique()\n#print(date_collect,dates)\nx = range(734)\ny = list(int(i) for i in date_collect[dates[x]].values[:])\n#print(x, y)\nplt.plot(x,y)\nplt.xticks(x[::50],[dates[i] for i in x[::50]],rotation = 'vertical')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-05T11:26:17.12816Z","iopub.execute_input":"2022-06-05T11:26:17.128399Z","iopub.status.idle":"2022-06-05T11:26:18.454078Z","shell.execute_reply.started":"2022-06-05T11:26:17.128366Z","shell.execute_reply":"2022-06-05T11:26:18.453364Z"},"trusted":true},"execution_count":null,"outputs":[]}]}