{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install tensorflow-recommenders\n","metadata":{"execution":{"iopub.status.busy":"2022-04-21T20:12:02.662015Z","iopub.execute_input":"2022-04-21T20:12:02.663224Z","iopub.status.idle":"2022-04-21T20:12:38.233458Z","shell.execute_reply.started":"2022-04-21T20:12:02.663080Z","shell.execute_reply":"2022-04-21T20:12:38.232149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install scann","metadata":{"execution":{"iopub.status.busy":"2022-04-21T20:12:38.236562Z","iopub.execute_input":"2022-04-21T20:12:38.236875Z","iopub.status.idle":"2022-04-21T20:14:03.644253Z","shell.execute_reply.started":"2022-04-21T20:12:38.236825Z","shell.execute_reply":"2022-04-21T20:14:03.642931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nprint(tf.__version__)\n\nimport tensorflow_recommenders as tfrs\nimport tensorflow_datasets as tfds\n\nimport os\nimport pprint\n\nfrom typing import Dict, Text\n\nimport pandas as pd\nimport numpy as np\nimport time\n\n\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-04-21T20:14:03.648226Z","iopub.execute_input":"2022-04-21T20:14:03.648510Z","iopub.status.idle":"2022-04-21T20:14:09.719652Z","shell.execute_reply.started":"2022-04-21T20:14:03.648452Z","shell.execute_reply":"2022-04-21T20:14:09.718603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef read_files(file_path, **kwargs):\n    \n    art_df_args = dict(filepath_or_buffer=file_path + 'articles.csv',low_memory = False)\n    if 'art_cols' in kwargs:\n        art_df_args['usecols']=kwargs['art_cols']\n    \n    cust_df_args = dict(filepath_or_buffer=file_path + 'customers.csv', low_memory = False)\n    if  'cust_cols' in  kwargs:\n        cust_df_args['usecols']=kwargs['cust_cols']\n    \n    trans_df_args= dict(filepath_or_buffer=file_path + 'transactions_train.csv', low_memory = False)\n    if  'trans_cols' in kwargs:\n        trans_df_args['usecols']=kwargs['trans_cols']\n    \n    art_df = pd.read_csv(**art_df_args)\n    cust_df = pd.read_csv(**cust_df_args)\n    trans_df= pd.read_csv(**trans_df_args)\n    \n    customer_lookup = cust_df.reset_index().set_index('customer_id')['index'].astype(str).to_dict()\n    article_lookup =art_df.reset_index().set_index('article_id')['index'].astype(str).to_dict()\n    \n    trans_df['user_id']= trans_df['customer_id'].map(customer_lookup)\n    trans_df['item_id']= trans_df['article_id'].map(article_lookup)\n    \n    unique_users = trans_df['user_id'].unique()\n    unique_items = trans_df['item_id'].unique()\n    \n    trans_df = trans_df.drop(columns =['customer_id','article_id'])\n    \n    return customer_lookup, article_lookup, trans_df, unique_users, unique_items\n\n    ","metadata":{"execution":{"iopub.status.busy":"2022-04-21T20:15:54.451352Z","iopub.execute_input":"2022-04-21T20:15:54.452045Z","iopub.status.idle":"2022-04-21T20:15:54.465125Z","shell.execute_reply.started":"2022-04-21T20:15:54.452005Z","shell.execute_reply":"2022-04-21T20:15:54.463609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"%%time\n#cust_cols=['customer_id']\n#trans_cols= ['customer_id','article_id']\nfile_path = '../input/h-and-m-personalized-fashion-recommendations/'\ncustomer_lookup, article_lookup, trans_data, user_vocab, item_vocab = read_files(file_path, cust_cols=['customer_id'], trans_cols= ['customer_id','article_id'])","metadata":{"execution":{"iopub.status.busy":"2022-04-21T20:15:55.491685Z","iopub.execute_input":"2022-04-21T20:15:55.492025Z","iopub.status.idle":"2022-04-21T20:17:36.723286Z","shell.execute_reply.started":"2022-04-21T20:15:55.491991Z","shell.execute_reply":"2022-04-21T20:17:36.722355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## For the retrieval model we need -\n1. Query tower - embeddings for user_ids\n2. Candidate tower - embeddings for artilce_ids\n\nFollow the steps below:\n\n    1. Keep just the user_id and article_id\n    \n    2. Convert pd.DataFrame to tf.data.dataset \n    \n    3. Split into train and test data\n    \n    4. Convert user_ids to integers and convert them embeddings visa Embedding layer\n   ","metadata":{"execution":{"iopub.status.busy":"2022-04-17T12:08:30.494384Z","iopub.execute_input":"2022-04-17T12:08:30.494644Z","iopub.status.idle":"2022-04-17T12:08:30.500568Z","shell.execute_reply.started":"2022-04-17T12:08:30.494617Z","shell.execute_reply":"2022-04-17T12:08:30.499511Z"}}},{"cell_type":"code","source":"%%time\ntrain_size =0.80\nnp.random.seed(1221)\ntrain = trans_data[['user_id','item_id']].sample(frac=train_size)\ntest =  trans_data[['user_id','item_id']].drop(train.index)\n\ntrain = tf.data.Dataset.from_tensor_slices(dict(train))\ntest = tf.data.Dataset.from_tensor_slices(dict(test))\n","metadata":{"execution":{"iopub.status.busy":"2022-04-21T20:19:25.777873Z","iopub.execute_input":"2022-04-21T20:19:25.778242Z","iopub.status.idle":"2022-04-21T20:20:00.793019Z","shell.execute_reply.started":"2022-04-21T20:19:25.778210Z","shell.execute_reply":"2022-04-21T20:20:00.791941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"items = tf.data.Dataset.from_tensor_slices(item_vocab)","metadata":{"execution":{"iopub.status.busy":"2022-04-21T20:20:05.694742Z","iopub.execute_input":"2022-04-21T20:20:05.695667Z","iopub.status.idle":"2022-04-21T20:20:05.728569Z","shell.execute_reply.started":"2022-04-21T20:20:05.695594Z","shell.execute_reply":"2022-04-21T20:20:05.727069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Batches in tensorflow dataset\n#### ratings.batch(1_000_000, drop_remainder = True) - this devides a tensorflow data set into equal batches of batch size = 1000000. Total unique number of user_ids are 31.78 million.\n\n##### 31.78 million/1 million = no of batches are 32 \n##### 31.78 million/1 million = no of batches are 31 if the drop_remainder is True","metadata":{}},{"cell_type":"markdown","source":"### tf.keras.layers.StringLookup - A preprocessing layer that maps string features to integers\n### tf.keras.layers.Embedding - Turns indexes into dense vectors of fixed size\n\n","metadata":{}},{"cell_type":"code","source":"embedding_dimension = 32","metadata":{"execution":{"iopub.status.busy":"2022-04-21T20:20:09.831421Z","iopub.execute_input":"2022-04-21T20:20:09.831731Z","iopub.status.idle":"2022-04-21T20:20:09.837798Z","shell.execute_reply.started":"2022-04-21T20:20:09.831699Z","shell.execute_reply":"2022-04-21T20:20:09.836186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## 4. Convert user_ids to integers and convert them embeddings visa Embedding layer\n## Query tower\n\nuser_model = tf.keras.Sequential([\n    tf.keras.layers.StringLookup(\n        vocabulary = user_vocab, mask_token =None),\n    tf.keras.layers.Embedding(len(user_vocab)+1, embedding_dimension)])","metadata":{"execution":{"iopub.status.busy":"2022-04-21T20:20:10.360402Z","iopub.execute_input":"2022-04-21T20:20:10.361043Z","iopub.status.idle":"2022-04-21T20:20:11.600588Z","shell.execute_reply.started":"2022-04-21T20:20:10.360993Z","shell.execute_reply":"2022-04-21T20:20:11.599531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Candidate tower\n\nitem_model = tf.keras.Sequential([\n    tf.keras.layers.StringLookup(\n        vocabulary = item_vocab, mask_token =None),\n    tf.keras.layers.Embedding(len(item_vocab)+1, embedding_dimension)\n])","metadata":{"execution":{"iopub.status.busy":"2022-04-21T20:20:11.602461Z","iopub.execute_input":"2022-04-21T20:20:11.602838Z","iopub.status.idle":"2022-04-21T20:20:11.650193Z","shell.execute_reply.started":"2022-04-21T20:20:11.602794Z","shell.execute_reply":"2022-04-21T20:20:11.649245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### This computes metrics for across top K candidates surfaced by a retrieval model.\n#### The default metric is top K categorical accuracy : how often the true candidate is in in top K candidates for a given query","metadata":{}},{"cell_type":"code","source":"metrics = tfrs.metrics.FactorizedTopK(\n    candidates=items.batch(256).map(item_model))\n\ntask = tfrs.tasks.Retrieval(metrics=metrics)","metadata":{"execution":{"iopub.status.busy":"2022-04-21T20:20:14.460072Z","iopub.execute_input":"2022-04-21T20:20:14.460380Z","iopub.status.idle":"2022-04-21T20:20:14.577748Z","shell.execute_reply.started":"2022-04-21T20:20:14.460347Z","shell.execute_reply":"2022-04-21T20:20:14.576715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## We can put it all together in a model: (Returns user embeddings and positive item embeddings)\n#### 1. User model\n#### 2. Item model\n#### 3. Retrieval task layer\n","metadata":{}},{"cell_type":"code","source":"class UserItemModel(tfrs.Model):\n    \n    def __init__(self, user_model, item_model):\n        super().__init__()\n        self.user_model : tf.keras.Model = user_model\n        self.item_model : tf.keras.Model = item_model\n        self.task : tf.keras.layers.Layer = task\n            \n    def compute_loss(self,features: Dict[Text, tf.Tensor], training=False) -> tf.Tensor:\n        \n        user_embeddings = self.user_model(features['user_id'])\n        positive_item_embeddings = self.item_model(features['item_id'])\n        \n        return self.task(user_embeddings,positive_item_embeddings)","metadata":{"execution":{"iopub.status.busy":"2022-04-21T20:20:17.236360Z","iopub.execute_input":"2022-04-21T20:20:17.236757Z","iopub.status.idle":"2022-04-21T20:20:17.257360Z","shell.execute_reply.started":"2022-04-21T20:20:17.236711Z","shell.execute_reply":"2022-04-21T20:20:17.256542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = UserItemModel(user_model, item_model)\nmodel.compile(optimizer=tf.keras.optimizers.Adagrad(learning_rate=0.1))","metadata":{"execution":{"iopub.status.busy":"2022-04-21T20:20:19.729462Z","iopub.execute_input":"2022-04-21T20:20:19.730044Z","iopub.status.idle":"2022-04-21T20:20:19.750678Z","shell.execute_reply.started":"2022-04-21T20:20:19.730009Z","shell.execute_reply":"2022-04-21T20:20:19.749784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cached_train = train.batch(16384).cache()\ncached_test = test.batch(4096).cache()","metadata":{"execution":{"iopub.status.busy":"2022-04-21T20:20:40.219632Z","iopub.execute_input":"2022-04-21T20:20:40.219967Z","iopub.status.idle":"2022-04-21T20:20:40.228704Z","shell.execute_reply.started":"2022-04-21T20:20:40.219933Z","shell.execute_reply":"2022-04-21T20:20:40.227274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmodel.fit(cached_train, epochs=3)","metadata":{"execution":{"iopub.status.busy":"2022-04-21T20:20:51.390106Z","iopub.execute_input":"2022-04-21T20:20:51.390390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}