{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nfrom datetime import datetime, timedelta\nimport gc\nimport cudf\nimport tensorflow as tf\nfrom fastai.tabular.core import add_datepart #fails because of some issue with weeks in cudf?\nimport cupy as cp","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-03T19:47:34.297423Z","iopub.execute_input":"2022-04-03T19:47:34.297741Z","iopub.status.idle":"2022-04-03T19:47:43.214485Z","shell.execute_reply.started":"2022-04-03T19:47:34.297654Z","shell.execute_reply":"2022-04-03T19:47:43.213735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#some nice ideas on reducing memory: https://www.kaggle.com/c/h-and-m-personalized-fashion-recommendations/discussion/308635\ntrain = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv', parse_dates=['t_dat'])\ntrain['customer_id'] = train['customer_id'].str[-16:].str.hex_to_int().astype('int64')\ntrain['article_id'] = train.article_id.astype('int32')\ntrain.t_dat = cudf.to_datetime(train.t_dat)\ntrain = train[['t_dat','customer_id','article_id']]\ntrain.to_parquet('train.pqt',index=False)\nprint( train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-03T19:47:43.216113Z","iopub.execute_input":"2022-04-03T19:47:43.216397Z","iopub.status.idle":"2022-04-03T19:48:25.335265Z","shell.execute_reply.started":"2022-04-03T19:47:43.21636Z","shell.execute_reply":"2022-04-03T19:48:25.334563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = train.groupby(['customer_id','article_id'])['t_dat'].agg('count').reset_index()\ntmp.columns = ['customer_id','article_id','ct']\ntmp.tail()\n","metadata":{"execution":{"iopub.status.busy":"2022-04-03T19:48:25.336689Z","iopub.execute_input":"2022-04-03T19:48:25.337156Z","iopub.status.idle":"2022-04-03T19:48:25.454831Z","shell.execute_reply.started":"2022-04-03T19:48:25.337117Z","shell.execute_reply":"2022-04-03T19:48:25.454113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.merge(tmp,on=['customer_id','article_id'],how='left')\ntrain = train.sort_values(['ct','t_dat'],ascending=False)\ntrain = train.drop_duplicates(['customer_id','article_id'])\ntrain = train.sort_values(['ct','t_dat'],ascending=False)\ntrain.tail()","metadata":{"execution":{"iopub.status.busy":"2022-04-03T19:48:25.456773Z","iopub.execute_input":"2022-04-03T19:48:25.457102Z","iopub.status.idle":"2022-04-03T19:48:26.646776Z","shell.execute_reply.started":"2022-04-03T19:48:25.457063Z","shell.execute_reply":"2022-04-03T19:48:26.646069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-03T19:48:26.647944Z","iopub.execute_input":"2022-04-03T19:48:26.648353Z","iopub.status.idle":"2022-04-03T19:48:26.671788Z","shell.execute_reply.started":"2022-04-03T19:48:26.648312Z","shell.execute_reply":"2022-04-03T19:48:26.671133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['year'] = train['t_dat'].dt.year\ntrain['month'] = train['t_dat'].dt.month\ntrain['day'] = train['t_dat'].dt.day\ntrain['dayofweek'] = train['t_dat'].dt.dayofweek\ntrain['dayofyear'] = train['t_dat'].dt.dayofyear\n#having probs with bools with tf...\n#train['is_month_end'] = train['t_dat'].dt.is_month_end\n#train['is_month_start'] = train['t_dat'].dt.is_month_start\ntrain.drop(columns=['t_dat'], inplace = True)\n\ntrain.tail()","metadata":{"execution":{"iopub.status.busy":"2022-04-03T19:48:26.672891Z","iopub.execute_input":"2022-04-03T19:48:26.67321Z","iopub.status.idle":"2022-04-03T19:48:26.710392Z","shell.execute_reply.started":"2022-04-03T19:48:26.673173Z","shell.execute_reply":"2022-04-03T19:48:26.709529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# this only works with Pandas...\ndef df_to_dataset(dataframe, shuffle=True, batch_size=32):\n  dataframe = dataframe.copy()\n  #labels = dataframe.pop('target')\n  ds = tf.data.Dataset.from_tensor_slices((dict(dataframe))) #, labels))\n  if shuffle:\n    ds = ds.shuffle(buffer_size=len(dataframe))\n  ds = ds.batch(batch_size)\n  return ds","metadata":{"execution":{"iopub.status.busy":"2022-04-03T19:48:26.711654Z","iopub.execute_input":"2022-04-03T19:48:26.711964Z","iopub.status.idle":"2022-04-03T19:48:26.717542Z","shell.execute_reply.started":"2022-04-03T19:48:26.711927Z","shell.execute_reply":"2022-04-03T19:48:26.716431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['cust_cat']=train['customer_id'].astype('category')\ntrain['cat_codes'] = train['cust_cat'].cat.codes #need a number because tf can't take a category...\ncust_cat_df = train[['customer_id', 'cust_cat', 'cat_codes']] #save them to put them back together later\nprint(cust_cat_df.dtypes)\ncust_cat_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-03T19:48:26.718877Z","iopub.execute_input":"2022-04-03T19:48:26.719387Z","iopub.status.idle":"2022-04-03T19:48:27.160683Z","shell.execute_reply.started":"2022-04-03T19:48:26.719349Z","shell.execute_reply":"2022-04-03T19:48:27.159845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.dtypes)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-03T19:48:27.162071Z","iopub.execute_input":"2022-04-03T19:48:27.162403Z","iopub.status.idle":"2022-04-03T19:48:27.409071Z","shell.execute_reply.started":"2022-04-03T19:48:27.162363Z","shell.execute_reply":"2022-04-03T19:48:27.408256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop(columns=['cust_cat', 'cat_codes'], inplace = True)\ntrain.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-04-03T19:48:27.411725Z","iopub.execute_input":"2022-04-03T19:48:27.412058Z","iopub.status.idle":"2022-04-03T19:48:27.419894Z","shell.execute_reply.started":"2022-04-03T19:48:27.412008Z","shell.execute_reply":"2022-04-03T19:48:27.419256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# From here: http://bl.ocks.org/miguelusque/raw/f44a8e729896a96d0a3e4b07b5176af4/#cudf-tensorflow\n#dst = tf.experimental.dlpack.from_dlpack(cp.fromDlpack(train.to_dlpack()).T.toDlpack())\ntrain_tf = tf.experimental.dlpack.from_dlpack(cp.asarray(train.as_gpu_matrix()).T.toDlpack()) \n\nprint(type(train_tf), \"\\n\", train_tf)","metadata":{"execution":{"iopub.status.busy":"2022-04-03T19:48:27.42106Z","iopub.execute_input":"2022-04-03T19:48:27.421451Z","iopub.status.idle":"2022-04-03T19:48:28.726596Z","shell.execute_reply.started":"2022-04-03T19:48:27.421415Z","shell.execute_reply":"2022-04-03T19:48:28.725814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get a list of all unique article ids\narticles = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv',  usecols=['article_id'])\narticles.drop_duplicates(inplace = True)\narticles.shape, articles.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-04-03T19:48:28.727699Z","iopub.execute_input":"2022-04-03T19:48:28.728428Z","iopub.status.idle":"2022-04-03T19:48:29.136951Z","shell.execute_reply.started":"2022-04-03T19:48:28.728387Z","shell.execute_reply":"2022-04-03T19:48:29.136205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get error using the method that worked for the users DF in cell above this one\n#articles_tf = tf.experimental.dlpack.from_dlpack(cp.asarray(articles.as_gpu_matrix()).T.toDlpack()) \narticles_tf = tf.experimental.dlpack.from_dlpack(cp.fromDlpack(articles.to_dlpack()).T.toDlpack())\nprint(type(articles_tf), \"\\n\", articles_tf)","metadata":{"execution":{"iopub.status.busy":"2022-04-03T19:48:29.138215Z","iopub.execute_input":"2022-04-03T19:48:29.138464Z","iopub.status.idle":"2022-04-03T19:48:29.14788Z","shell.execute_reply.started":"2022-04-03T19:48:29.138436Z","shell.execute_reply":"2022-04-03T19:48:29.146934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -q tensorflow-recommenders\nimport tensorflow_recommenders as tfrs","metadata":{"execution":{"iopub.status.busy":"2022-04-03T19:48:29.149521Z","iopub.execute_input":"2022-04-03T19:48:29.149933Z","iopub.status.idle":"2022-04-03T19:48:57.585735Z","shell.execute_reply.started":"2022-04-03T19:48:29.149893Z","shell.execute_reply":"2022-04-03T19:48:57.584921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"user_ids_vocabulary = tf.keras.layers.StringLookup(mask_token=None)\nuser_ids_vocabulary.adapt(train_tf.map(lambda x: x[\"user_id\"]))\n\narticles_vocabulary = tf.keras.layers.StringLookup(mask_token=None)\narticles_vocabulary.adapt(articles_tf)","metadata":{"execution":{"iopub.status.busy":"2022-04-03T19:55:12.462221Z","iopub.execute_input":"2022-04-03T19:55:12.463067Z","iopub.status.idle":"2022-04-03T19:55:12.494177Z","shell.execute_reply.started":"2022-04-03T19:55:12.463007Z","shell.execute_reply":"2022-04-03T19:55:12.493307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from typing import Dict, Text\n\nclass Model(tfrs.Model):\n\n  def __init__(self):\n    super().__init__()\n\n    # Set up user representation.\n    self.user_model = tf.keras.layers.Embedding(\n        input_dim=2000, output_dim=64)\n    # Set up item representation.\n    self.item_model = tf.keras.layers.Embedding(\n        input_dim=2000, output_dim=64)\n    # Set up a retrieval task and evaluation metrics over the\n    # entire dataset of candidates.\n    self.task = tfrs.tasks.Retrieval(\n        metrics=tfrs.metrics.FactorizedTopK(\n            candidates=train_tf.batch(128).map(self.item_model)\n        )\n    )\n\n  def compute_loss(self, features: Dict[Text, tf.Tensor], training=False) -> tf.Tensor:\n\n    user_embeddings = self.user_model(features[\"user_id\"])\n    article_embeddings = self.item_model(features[\"article_id\"])\n\n    return self.task(user_embeddings, article_embeddings)","metadata":{"execution":{"iopub.status.busy":"2022-04-03T19:48:57.587419Z","iopub.execute_input":"2022-04-03T19:48:57.587675Z","iopub.status.idle":"2022-04-03T19:48:57.60043Z","shell.execute_reply.started":"2022-04-03T19:48:57.587639Z","shell.execute_reply":"2022-04-03T19:48:57.598017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Model()\nmodel.compile(optimizer=tf.keras.optimizers.Adagrad(0.5))","metadata":{"execution":{"iopub.status.busy":"2022-04-03T19:48:57.601771Z","iopub.execute_input":"2022-04-03T19:48:57.602037Z","iopub.status.idle":"2022-04-03T19:48:58.264658Z","shell.execute_reply.started":"2022-04-03T19:48:57.602Z","shell.execute_reply":"2022-04-03T19:48:58.26367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}