{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-19T18:01:08.474254Z","iopub.execute_input":"2022-04-19T18:01:08.474683Z","iopub.status.idle":"2022-04-19T18:01:08.495781Z","shell.execute_reply.started":"2022-04-19T18:01:08.474590Z","shell.execute_reply":"2022-04-19T18:01:08.495006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport math","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:01:08.510333Z","iopub.execute_input":"2022-04-19T18:01:08.510740Z","iopub.status.idle":"2022-04-19T18:01:08.519669Z","shell.execute_reply.started":"2022-04-19T18:01:08.510709Z","shell.execute_reply":"2022-04-19T18:01:08.518820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we start by putting in place the structure of our code","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:01:08.575560Z","iopub.execute_input":"2022-04-19T18:01:08.575995Z","iopub.status.idle":"2022-04-19T18:01:08.579508Z","shell.execute_reply.started":"2022-04-19T18:01:08.575952Z","shell.execute_reply":"2022-04-19T18:01:08.578766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create_folds.py","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:01:08.620171Z","iopub.execute_input":"2022-04-19T18:01:08.620731Z","iopub.status.idle":"2022-04-19T18:01:08.624359Z","shell.execute_reply.started":"2022-04-19T18:01:08.620697Z","shell.execute_reply":"2022-04-19T18:01:08.623662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nimport copy\nimport torch\nfrom torch import nn\nfrom torch.nn import functional as F\n# import chainer\n# import chainer.functions as F\n# import chainer.links as L","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:01:08.673387Z","iopub.execute_input":"2022-04-19T18:01:08.673907Z","iopub.status.idle":"2022-04-19T18:01:09.968556Z","shell.execute_reply.started":"2022-04-19T18:01:08.673875Z","shell.execute_reply":"2022-04-19T18:01:09.967515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len_embed = 30\nnum_prod_to_rec = 12 # number of products to recommend to a user at each time step\nbeta = 0.5\nscale = 100\n# episode_length = 128","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:01:09.970287Z","iopub.execute_input":"2022-04-19T18:01:09.970541Z","iopub.status.idle":"2022-04-19T18:01:09.975486Z","shell.execute_reply.started":"2022-04-19T18:01:09.970511Z","shell.execute_reply":"2022-04-19T18:01:09.974725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_train = '../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv'\ndf_transactions = pd.read_csv(transactions_train,\n                             dtype= {\n                                 'customer_id': 'str',\n                                 'article_id': 'str'\n                             })\ndf_transactions['t_dat'] = pd.to_datetime(df_transactions['t_dat'])\ndf_transactions = df_transactions.set_index('t_dat')\nprint(df_transactions.index.min(), df_transactions.index.max())\ndf_transactions.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:01:09.976747Z","iopub.execute_input":"2022-04-19T18:01:09.977594Z","iopub.status.idle":"2022-04-19T18:02:32.032512Z","shell.execute_reply.started":"2022-04-19T18:01:09.977561Z","shell.execute_reply":"2022-04-19T18:02:32.031328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers = '../input/h-and-m-personalized-fashion-recommendations/customers.csv'\ndf_customers = pd.read_csv(customers,\n                             dtype= {\n                                 'customer_id': 'str'\n                             })","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:02:32.035319Z","iopub.execute_input":"2022-04-19T18:02:32.035607Z","iopub.status.idle":"2022-04-19T18:02:37.512732Z","shell.execute_reply.started":"2022-04-19T18:02:32.035571Z","shell.execute_reply":"2022-04-19T18:02:37.512022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_customers.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:02:37.514049Z","iopub.execute_input":"2022-04-19T18:02:37.515066Z","iopub.status.idle":"2022-04-19T18:02:37.532451Z","shell.execute_reply.started":"2022-04-19T18:02:37.515025Z","shell.execute_reply":"2022-04-19T18:02:37.531315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_customers.info()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:02:37.533884Z","iopub.execute_input":"2022-04-19T18:02:37.534169Z","iopub.status.idle":"2022-04-19T18:02:38.083440Z","shell.execute_reply.started":"2022-04-19T18:02:37.534136Z","shell.execute_reply":"2022-04-19T18:02:38.082230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"products = '../input/h-and-m-personalized-fashion-recommendations/articles.csv'\ndf_products = pd.read_csv(products,\n                             dtype= {\n                                 'article_id': 'str'\n                             })","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:02:38.085633Z","iopub.execute_input":"2022-04-19T18:02:38.086309Z","iopub.status.idle":"2022-04-19T18:02:39.178658Z","shell.execute_reply.started":"2022-04-19T18:02:38.086258Z","shell.execute_reply":"2022-04-19T18:02:39.177523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_products.info()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:02:39.180400Z","iopub.execute_input":"2022-04-19T18:02:39.180775Z","iopub.status.idle":"2022-04-19T18:02:39.344620Z","shell.execute_reply.started":"2022-04-19T18:02:39.180712Z","shell.execute_reply":"2022-04-19T18:02:39.343451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embed_prod = nn.Embedding(len(df_products.index), len_embed)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:02:39.346282Z","iopub.execute_input":"2022-04-19T18:02:39.346743Z","iopub.status.idle":"2022-04-19T18:02:39.409221Z","shell.execute_reply.started":"2022-04-19T18:02:39.346693Z","shell.execute_reply":"2022-04-19T18:02:39.408133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embed_prod(torch.LongTensor([3])).squeeze()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:02:39.412471Z","iopub.execute_input":"2022-04-19T18:02:39.412747Z","iopub.status.idle":"2022-04-19T18:02:39.502101Z","shell.execute_reply.started":"2022-04-19T18:02:39.412715Z","shell.execute_reply":"2022-04-19T18:02:39.501024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = embed_prod(torch.LongTensor([3,4])).squeeze()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:02:39.503418Z","iopub.execute_input":"2022-04-19T18:02:39.503678Z","iopub.status.idle":"2022-04-19T18:02:39.509348Z","shell.execute_reply.started":"2022-04-19T18:02:39.503645Z","shell.execute_reply":"2022-04-19T18:02:39.508297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:02:39.511108Z","iopub.execute_input":"2022-04-19T18:02:39.511878Z","iopub.status.idle":"2022-04-19T18:02:39.525445Z","shell.execute_reply.started":"2022-04-19T18:02:39.511829Z","shell.execute_reply":"2022-04-19T18:02:39.524367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for elt in temp:\n    print(elt)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:02:39.527009Z","iopub.execute_input":"2022-04-19T18:02:39.528108Z","iopub.status.idle":"2022-04-19T18:02:39.537616Z","shell.execute_reply.started":"2022-04-19T18:02:39.528070Z","shell.execute_reply":"2022-04-19T18:02:39.536332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input1 = torch.randn(100, 128)\ninput1","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:02:39.539084Z","iopub.execute_input":"2022-04-19T18:02:39.539516Z","iopub.status.idle":"2022-04-19T18:02:39.557101Z","shell.execute_reply.started":"2022-04-19T18:02:39.539483Z","shell.execute_reply":"2022-04-19T18:02:39.556142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input2 = torch.randn(100, 128)\ncos = nn.CosineSimilarity(dim=1, eps=1e-6)\noutput = cos(input1, input2)\noutput","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:02:39.558428Z","iopub.execute_input":"2022-04-19T18:02:39.558680Z","iopub.status.idle":"2022-04-19T18:02:39.585574Z","shell.execute_reply.started":"2022-04-19T18:02:39.558648Z","shell.execute_reply":"2022-04-19T18:02:39.584827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"date_split = '2020-09-15'\ntrain = df_transactions[:date_split]\ntest = df_transactions[date_split:]\nlen(train), len(test)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:02:39.586782Z","iopub.execute_input":"2022-04-19T18:02:39.587002Z","iopub.status.idle":"2022-04-19T18:02:40.119194Z","shell.execute_reply.started":"2022-04-19T18:02:39.586973Z","shell.execute_reply":"2022-04-19T18:02:40.118192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embed_custom = nn.Embedding(len(df_customers.index), len_embed)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:02:40.120471Z","iopub.execute_input":"2022-04-19T18:02:40.120713Z","iopub.status.idle":"2022-04-19T18:02:40.504618Z","shell.execute_reply.started":"2022-04-19T18:02:40.120682Z","shell.execute_reply":"2022-04-19T18:02:40.503512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embed_custom(torch.LongTensor([3]))","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:02:40.506079Z","iopub.execute_input":"2022-04-19T18:02:40.506339Z","iopub.status.idle":"2022-04-19T18:02:40.515108Z","shell.execute_reply.started":"2022-04-19T18:02:40.506307Z","shell.execute_reply":"2022-04-19T18:02:40.513836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# environment.py\n# in this cell we build the environment class that simulate a RL environment\n\nclass Environment:\n    def __init__(self, data_transactions, emb_customers, emb_products, df_customers, df_products):\n        self.data = data_transactions\n        self.emb_customers = emb_customers\n        self.emb_prod = emb_products\n        self.customers = df_customers\n        self.products = df_products\n        self.reset()\n        \n    def reset(self):\n        self.done = False\n        self.current_pos = 0\n        # we must return the informations about the current user (and the products bought by him)\n        # let's go first with the current user by getting his id\n        self.userid = self.data.iloc[self.current_pos, 0]\n        # from the userid, we must get its corresponding index  in df_customers\n        self.user_index = self.customers.index[self.customers['customer_id']==self.userid].tolist()[0]\n        # now we can get the user embedding informations from self.emb_customers\n        self.user_embed = self.emb_customers[torch.LongTensor([self.user_index])]\n        # in this first version we are not going to consider the last products used by our customer\n        return self.user_embed\n    \n    def step(self, act):\n        # here act is a list of 12 (num_prod_to_rec) products to recommend by their indices.\n        # For example, it could be [1, 2, 3, ..., 11, 12]\n        reward = 0\n        # First, me must find the index of the actual product used by the customer in self.products\n        # let's go first with the current product by getting his id\n        self.product_id = self.data.iloc[self.current_pos, 1]\n        # from the product_id, we must get its corresponding index in df_products\n        self.prod_index = self.products.index[self.products['article_id']==self.product_id].tolist()[0]\n        reward = self.compute_reward(act, self.prod_index)\n        if (self.current_pos < len(self.data.index)):\n#             if ((self.current_pos % episode_length)!=0):\n            self.current_pos += 1\n            if (self.current_pos < len(self.data.index)):\n                self.userid = self.data.iloc[self.current_pos, 0]\n                self.user_index = self.customers.index[self.customers['customer_id']==self.userid].tolist()[0]\n                self.user_embed = self.emb_customers[torch.LongTensor([self.user_index])]\n                self.done = False\n                return self.user_embed, reward, self.done\n            else:\n                self.done = True\n                return self.user_embed, reward, self.done\n    \n    def compute_reward(self, proposed_products, actual_product):\n        if (actual_product in proposed_products):\n            # take the position where the matching occurs\n            postion = proposed_products.index(actual_product)\n            proposed_products = proposed_products[:position+1]\n        embed_proposed = self.emb_prod[torch.LongTensor(proposed_products)]\n        embed_actual = self.emb_prod[torch.LongTensor([actual_product])]\n        cos = nn.CosineSimilarity(dim=1, eps=1e-6)\n        output = cos(embed_proposed, embed_actual)\n        coeffs = [math.pow(beta, t) for t in range(len(proposed_products))]\n        # convert coeffs to tensor\n        coeffs = torch.Tensor(coeffs)\n        # apply element wise multiplication to output\n        output = torch.mul(out_put, coeffs)\n        # we sum that and multiply by 100\n        reward = torch.sum(output)\n        reward = torch.mul(reward, 100)\n        return reward","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:02:40.517202Z","iopub.execute_input":"2022-04-19T18:02:40.517596Z","iopub.status.idle":"2022-04-19T18:02:40.539947Z","shell.execute_reply.started":"2022-04-19T18:02:40.517531Z","shell.execute_reply":"2022-04-19T18:02:40.538974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# models.py\nclass Actor(torch.nn.Module):\n    def __init__(self, n_input, n_weight_out, n_features_out, product_feats):\n        # n_input: number of features that represents a customer\n        # n_weight_out: number of products we want to recommend represent by their features\n        # n_features_out: number of features of each vector of recommend products\n        # product_feats: matrix of vector products. Each line as the same number of\n        # features as n_features_out\n        super(Actor, self).__init__()\n        self.feats = product_feats\n        l1 = 4*n_input\n        l2 = 2*l1\n        self.l1 = nn.Linear(n_input, l1)\n        self.l2 = nn.Linear(l1, l2)\n        self.l3 = nn.Linear(l2, n_weight_out * n_features_out)\n    def forward(self, x):\n        x = F.relu(self.l1(x))\n        x = F.relu(self.l2(x))\n        x = F.relu(self.l3(x))\n        the_weights = x.view(-1, n_weight_out, n_features_out)\n        # take the dot product between the_weights and self.feats\n        product_trans = torch.transpose(self.feats, 0, 1)\n        all_scalar_prods = torch.matmul(the_weights, product_trans)\n        # output shape of previous operation : (1, n_weight_out, num_of_products)\n        # find the indices with the highest scalar products\n        \n        return the_weights\n\nclass Critic(torch.nn.Module):\n    def __init__(self):\n        super(Critic, self).__init__()\n        ","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:02:40.541230Z","iopub.execute_input":"2022-04-19T18:02:40.541570Z","iopub.status.idle":"2022-04-19T18:02:40.559960Z","shell.execute_reply.started":"2022-04-19T18:02:40.541537Z","shell.execute_reply":"2022-04-19T18:02:40.559032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train.py","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:02:40.561198Z","iopub.execute_input":"2022-04-19T18:02:40.561437Z","iopub.status.idle":"2022-04-19T18:02:40.576243Z","shell.execute_reply.started":"2022-04-19T18:02:40.561406Z","shell.execute_reply":"2022-04-19T18:02:40.575203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"# inference.py","metadata":{"execution":{"iopub.status.busy":"2022-04-13T08:18:11.554489Z","iopub.execute_input":"2022-04-13T08:18:11.554724Z","iopub.status.idle":"2022-04-13T08:18:11.567531Z","shell.execute_reply.started":"2022-04-13T08:18:11.554696Z","shell.execute_reply":"2022-04-13T08:18:11.566615Z"}}},{"cell_type":"code","source":"# models.py","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:02:40.578194Z","iopub.execute_input":"2022-04-19T18:02:40.578536Z","iopub.status.idle":"2022-04-19T18:02:40.588523Z","shell.execute_reply.started":"2022-04-19T18:02:40.578489Z","shell.execute_reply":"2022-04-19T18:02:40.587500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#config.py","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:02:40.589689Z","iopub.execute_input":"2022-04-19T18:02:40.589976Z","iopub.status.idle":"2022-04-19T18:02:40.601960Z","shell.execute_reply.started":"2022-04-19T18:02:40.589937Z","shell.execute_reply":"2022-04-19T18:02:40.600913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model_dispatcher.py","metadata":{"execution":{"iopub.status.busy":"2022-04-19T18:02:40.603101Z","iopub.execute_input":"2022-04-19T18:02:40.603321Z","iopub.status.idle":"2022-04-19T18:02:40.613890Z","shell.execute_reply.started":"2022-04-19T18:02:40.603293Z","shell.execute_reply":"2022-04-19T18:02:40.613112Z"},"trusted":true},"execution_count":null,"outputs":[]}]}