{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport tensorflow as tf\n\nimport pickle","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:41:34.298514Z","iopub.execute_input":"2022-04-19T18:41:34.29898Z","iopub.status.idle":"2022-04-19T18:41:34.304453Z","shell.execute_reply.started":"2022-04-19T18:41:34.298911Z","shell.execute_reply":"2022-04-19T18:41:34.303593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\nfile_dir = '/kaggle/input/h-and-m-personalized-fashion-recommendations/'\narticles_df = pd.read_csv(file_dir + 'articles.csv')\ncustomers_df = pd.read_csv(file_dir + 'customers.csv')\ntransactions_df = pd.read_csv(file_dir + 'transactions_train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:41:34.305985Z","iopub.execute_input":"2022-04-19T18:41:34.306244Z","iopub.status.idle":"2022-04-19T18:42:56.667701Z","shell.execute_reply.started":"2022-04-19T18:41:34.306212Z","shell.execute_reply":"2022-04-19T18:42:56.666066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"articles_df length: \", articles_df.shape)\nprint(\"customers_df length: \", customers_df.shape)\nprint(\"transactions_df length: \", transactions_df.shape)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:42:56.670344Z","iopub.execute_input":"2022-04-19T18:42:56.670759Z","iopub.status.idle":"2022-04-19T18:42:56.678763Z","shell.execute_reply.started":"2022-04-19T18:42:56.670705Z","shell.execute_reply":"2022-04-19T18:42:56.677638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**TRANSACTIONS**","metadata":{}},{"cell_type":"code","source":"# calculate number of days since purchase\ntransactions_df['t_dat'] = pd.to_datetime(transactions_df['t_dat'], format=\"%Y-%m-%d\")\nlast_date = transactions_df.sort_values('t_dat', ascending = False).iloc[0,0]\ntransactions_df['nb_days'] = (-(transactions_df.loc[:,'t_dat'] - last_date)).dt.days\n\ntransactions_df['nb_days_token'] = pd.cut(transactions_df['nb_days'], 100, labels= range(0,100))\n\n# get unique article_ids from 0 to max articles (same as customer id)\ntransactions_df['article_id'] = pd.Categorical(transactions_df['article_id'])\ntransactions_df['article_token'] = transactions_df['article_id'].cat.codes\n\ntransactions_df['customer_id'] = pd.Categorical(transactions_df['customer_id'])\ntransactions_df['customer_token'] = transactions_df['customer_id'].cat.codes\n\n#create price bins\ntransactions_df['price_token'] = pd.qcut(transactions_df['price'], 10, labels=range(0,10))\ntransactions_df.sort_values(['customer_id','t_dat','article_id'], ascending=True)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:42:56.680858Z","iopub.execute_input":"2022-04-19T18:42:56.681262Z","iopub.status.idle":"2022-04-19T18:44:03.917046Z","shell.execute_reply.started":"2022-04-19T18:42:56.681214Z","shell.execute_reply":"2022-04-19T18:44:03.916029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#filter columns\ntransaction_cols = ['customer_id', 'article_id', 'customer_token', 'article_token', 'nb_days_token', 'price_token']\ntransactions_tokens = transactions_df[transaction_cols]\ntransactions_tokens","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:44:03.919654Z","iopub.execute_input":"2022-04-19T18:44:03.920539Z","iopub.status.idle":"2022-04-19T18:44:05.149548Z","shell.execute_reply.started":"2022-04-19T18:44:03.920482Z","shell.execute_reply":"2022-04-19T18:44:05.14857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**ARTICLES**","metadata":{}},{"cell_type":"code","source":"articles_df.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:44:05.150936Z","iopub.execute_input":"2022-04-19T18:44:05.151168Z","iopub.status.idle":"2022-04-19T18:44:05.33646Z","shell.execute_reply.started":"2022-04-19T18:44:05.15114Z","shell.execute_reply":"2022-04-19T18:44:05.335423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_cols = ['product_code', 'product_type_no', 'graphical_appearance_no', 'colour_group_code', 'perceived_colour_value_id', 'perceived_colour_master_id', 'department_no' , 'index_code' , 'index_group_no', 'section_no', 'garment_group_no']\n\ndef tokenize(df):\n    df2 = pd.DataFrame(df['article_id'])\n    \n    for col in articles_cols:\n        x = col + '_token'\n        df2[x]= pd.Categorical(df[col])\n        df2[x] = df2[x].cat.codes\n    \n    return df2\n\narticles_tokens = tokenize(articles_df)\narticles_tokens.sort_values('article_id', ascending=True)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:44:05.338301Z","iopub.execute_input":"2022-04-19T18:44:05.338618Z","iopub.status.idle":"2022-04-19T18:44:05.412319Z","shell.execute_reply.started":"2022-04-19T18:44:05.338576Z","shell.execute_reply":"2022-04-19T18:44:05.411281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**CUSTOMERS**","metadata":{}},{"cell_type":"markdown","source":"Here, we concatenate the entire purchase history, as well as the purchase_weeks (to account for the history in rNN).","metadata":{}},{"cell_type":"code","source":"customers_df.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:44:05.413982Z","iopub.execute_input":"2022-04-19T18:44:05.41494Z","iopub.status.idle":"2022-04-19T18:44:07.389161Z","shell.execute_reply.started":"2022-04-19T18:44:05.414892Z","shell.execute_reply":"2022-04-19T18:44:07.38816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bins =[0., 20., 30., 40., 50., 60.]\ncustomers_tokens = pd.DataFrame(customers_df['customer_id'])\ncustomers_tokens['age'] = pd.cut(customers_df['age'], bins, labels= range(0,5))\ncustomers_tokens.sort_values('customer_id', ascending=True)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:44:07.390706Z","iopub.execute_input":"2022-04-19T18:44:07.390958Z","iopub.status.idle":"2022-04-19T18:44:09.224103Z","shell.execute_reply.started":"2022-04-19T18:44:07.390927Z","shell.execute_reply":"2022-04-19T18:44:09.22309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**MERGE TRANSACTION, ARTICLES, CUSTOMERS**","metadata":{}},{"cell_type":"code","source":"a = transactions_tokens\nb = customers_tokens\nc = articles_tokens\n\n\ndf_ = pd.merge(a, b, how='left', on='customer_id')\ndf = pd.merge(df_, c, how='left', on='article_id')\n\ndf.to_pickle('/kaggle/working/df.pkl') ","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:44:09.225978Z","iopub.execute_input":"2022-04-19T18:44:09.226308Z","iopub.status.idle":"2022-04-19T18:45:14.264136Z","shell.execute_reply.started":"2022-04-19T18:44:09.226262Z","shell.execute_reply":"2022-04-19T18:45:14.263327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df = pd.read_pickle('/kaggle/working/df.pkl')\ndf","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:45:14.269238Z","iopub.execute_input":"2022-04-19T18:45:14.269588Z","iopub.status.idle":"2022-04-19T18:45:14.306936Z","shell.execute_reply.started":"2022-04-19T18:45:14.269546Z","shell.execute_reply":"2022-04-19T18:45:14.305892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_list = pd.DataFrame(\n    df.groupby('customer_id').agg(\n        item_id = ('article_token',lambda x: list(x)),\n        nb_days = ('nb_days_token', lambda x: list(x)), \n        price = ('price_token', lambda x: list(x)), \n        age = ('age', lambda x: list(x)),\n        product_code = ('product_code_token',lambda x: list(x)),\n        product_type_no = ('product_type_no_token',lambda x: list(x)),\n        graphical_appearance = ('graphical_appearance_no_token',lambda x: list(x)),\n        perceived_colour_value_id = ('perceived_colour_value_id_token',lambda x: list(x)),\n        perceived_colour_master_id = ('perceived_colour_master_id_token',lambda x: list(x)),\n        department_no = ('department_no_token',lambda x: list(x)),\n        index_code = ('index_code_token',lambda x: list(x)),\n        index_group_no = ('index_group_no_token',lambda x: list(x)),\n        section_no = ('section_no_token',lambda x: list(x)),\n        garment_group_no = ('garment_group_no_token',lambda x: list(x))    \n    ))","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:45:14.308519Z","iopub.execute_input":"2022-04-19T18:45:14.309252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_list.to_pickle('/kaggle/working/df_list.pkl') \n# df_list = pd.read_pickle('/kaggle/working/df_list.pkl')\n# df_list","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here, we analyse the number of purchases by customer. This is done to determine the cut-off of input vector","metadata":{}},{"cell_type":"code","source":"df_list['number_of_purchases'] = df_list['item_id'].str.len()\nprint('50% of customers have less than ', df_list.number_of_purchases.quantile(0.5), ' purchases in the full year')\nprint('90% of customers have less than ', df_list.number_of_purchases.quantile(0.9), ' purchases in the full year')\nprint('95% of customers have less than ', df_list.number_of_purchases.quantile(0.95), ' purchases in the full year')\ndf_list['number_of_purchases'].hist(bins=100)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# history_length = 50\n# test['train_purchases'] = test['purchased_items'].str[:-1]\n# test['purchased_weeks'] = test['purchased_weeks'].str[:-1]\n# test['target_purchases'] = test['purchased_items'].str[1:]\n\n# # Here, we add trailing zeros to the list, and then take the last \"n = history_length\" items as this comprises our training data set\n# test['train_purchases'] = test['train_purchases'].apply(lambda x : [0] * history_length + x).str[-history_length:]\n# test['purchased_weeks'] = test['purchased_weeks'].apply(lambda x : [0] * history_length + x).str[-history_length:]\n# test['target_purchases'] = test['target_purchases'].apply(lambda x : [0] * history_length + x).str[-history_length:]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test.to_pickle('/kaggle/working/df_train.pkl') ","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}