{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Problem Statement\nTo develop product recommendations based on data from previous transactions, as well as from customer and product meta data. The available meta data spans from simple data, such as garment type and customer age, to text data from product descriptions, to image data from garment images.\n\nAlthough this problem can employ NLP or image processing to improve recommendations, in this notebook, I'm going to take a simpler approach. ","metadata":{}},{"cell_type":"markdown","source":"# Approach \nFor recommending products to a user, the approach employed here is: \n* Recommending products based on previously purchased items \n* Recommending products that are usually bought together with the previously bought products \n* Recommending popular products","metadata":{}},{"cell_type":"markdown","source":"# Importing Necessary Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport cudf\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-18T08:43:39.114163Z","iopub.execute_input":"2022-11-18T08:43:39.114534Z","iopub.status.idle":"2022-11-18T08:43:39.120228Z","shell.execute_reply.started":"2022-11-18T08:43:39.114504Z","shell.execute_reply":"2022-11-18T08:43:39.118844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-18T08:43:39.122692Z","iopub.execute_input":"2022-11-18T08:43:39.123691Z","iopub.status.idle":"2022-11-18T08:43:41.753861Z","shell.execute_reply.started":"2022-11-18T08:43:39.123652Z","shell.execute_reply":"2022-11-18T08:43:41.752863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['customer_id'][0]","metadata":{"execution":{"iopub.status.busy":"2022-11-18T08:43:41.755207Z","iopub.execute_input":"2022-11-18T08:43:41.755805Z","iopub.status.idle":"2022-11-18T08:43:41.788116Z","shell.execute_reply.started":"2022-11-18T08:43:41.755758Z","shell.execute_reply":"2022-11-18T08:43:41.786941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Processing the Customer ID column since the entire ID is not required for preserving the uniqueness of a customer. ","metadata":{}},{"cell_type":"code","source":"train['customer_id'] = train['customer_id'].str[-16:].str.hex_to_int().astype('int64')\ntrain['article_id'] = train.article_id.astype('int32')\ntrain.t_dat = cudf.to_datetime(train.t_dat)\ntrain = train[['t_dat','customer_id','article_id']]\ntrain.to_parquet('train.pqt',index=False)\nprint( train.shape )","metadata":{"execution":{"iopub.status.busy":"2022-11-18T08:43:41.789629Z","iopub.execute_input":"2022-11-18T08:43:41.790168Z","iopub.status.idle":"2022-11-18T08:43:42.677023Z","shell.execute_reply.started":"2022-11-18T08:43:41.790132Z","shell.execute_reply":"2022-11-18T08:43:42.675936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets find out the previous week's purchases for each customer. ","metadata":{}},{"cell_type":"code","source":"tmp = train.groupby('customer_id').t_dat.max().reset_index()\ntmp.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-18T08:43:42.680460Z","iopub.execute_input":"2022-11-18T08:43:42.681254Z","iopub.status.idle":"2022-11-18T08:43:42.750757Z","shell.execute_reply.started":"2022-11-18T08:43:42.681200Z","shell.execute_reply":"2022-11-18T08:43:42.749726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp.columns = ['customer_id','max_dat']\ntrain = train.merge(tmp,on=['customer_id'],how='left')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-18T08:43:42.752601Z","iopub.execute_input":"2022-11-18T08:43:42.753006Z","iopub.status.idle":"2022-11-18T08:43:42.857961Z","shell.execute_reply.started":"2022-11-18T08:43:42.752965Z","shell.execute_reply":"2022-11-18T08:43:42.856743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['diff_dat'] = (train.max_dat - train.t_dat).dt.days\ntrain = train.loc[train['diff_dat']<=6]\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-18T08:43:42.859601Z","iopub.execute_input":"2022-11-18T08:43:42.860016Z","iopub.status.idle":"2022-11-18T08:43:42.932828Z","shell.execute_reply.started":"2022-11-18T08:43:42.859977Z","shell.execute_reply":"2022-11-18T08:43:42.931678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PART 1: Items previously purchased(most often)","metadata":{}},{"cell_type":"code","source":"tmp = train.groupby(['customer_id','article_id'])['t_dat'].agg('count').reset_index()\ntmp.columns = ['customer_id','article_id','ct']\ntmp.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-11-18T08:44:41.929639Z","iopub.execute_input":"2022-11-18T08:44:41.930006Z","iopub.status.idle":"2022-11-18T08:44:41.973320Z","shell.execute_reply.started":"2022-11-18T08:44:41.929969Z","shell.execute_reply":"2022-11-18T08:44:41.972268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.merge(tmp,on=['customer_id','article_id'],how='left')\ntrain = train.sort_values(['ct','t_dat'],ascending=False)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-18T08:44:58.107423Z","iopub.execute_input":"2022-11-18T08:44:58.107952Z","iopub.status.idle":"2022-11-18T08:44:58.236883Z","shell.execute_reply.started":"2022-11-18T08:44:58.107912Z","shell.execute_reply":"2022-11-18T08:44:58.235621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.drop_duplicates(['customer_id','article_id'])\ntrain = train.sort_values(['ct','t_dat'],ascending=False)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-18T08:45:17.874216Z","iopub.execute_input":"2022-11-18T08:45:17.874593Z","iopub.status.idle":"2022-11-18T08:45:18.027089Z","shell.execute_reply.started":"2022-11-18T08:45:17.874561Z","shell.execute_reply":"2022-11-18T08:45:18.026010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}