{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install tensorflow-recommenders\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-09T12:29:44.189143Z","iopub.execute_input":"2023-03-09T12:29:44.189604Z","iopub.status.idle":"2023-03-09T12:30:11.971060Z","shell.execute_reply.started":"2023-03-09T12:29:44.189567Z","shell.execute_reply":"2023-03-09T12:30:11.969797Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nfrom numpy import int64\nimport pprint\nfrom typing import Dict, Text\nimport matplotlib\nfrom sklearn.preprocessing import LabelEncoder\nimport tensorflow as tf\nimport tensorflow_datasets as tfds\nimport tensorflow_recommenders as tfrs\n        \nINPUT_PATH = '/kaggle/input/h-and-m-personalized-fashion-recommendations/'    ","metadata":{"execution":{"iopub.status.busy":"2023-03-09T12:30:11.973753Z","iopub.execute_input":"2023-03-09T12:30:11.974170Z","iopub.status.idle":"2023-03-09T12:30:22.700469Z","shell.execute_reply.started":"2023-03-09T12:30:11.974129Z","shell.execute_reply":"2023-03-09T12:30:22.698572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Understanding (CRISP DM)","metadata":{}},{"cell_type":"code","source":"### CUSTOMERS ###\n\n# Read the CSV file\ncustomers = pd.read_csv(INPUT_PATH + 'customers.csv')\n\n# Read some data info and null values info <\ncustomers.info()\ncustomers.isnull().sum()\n\n## Check the null values for the customers dataset\nmatplotlib.rcParams['figure.figsize'] = (20,6)\nsns.heatmap(customers.isnull(), yticklabels = False, cbar = False, cmap = 'viridis')\nmatplotlib.pyplot.title(\"Missing null values\")\n\n# Check the values count of every club member & active status as well as fashion news frequency \ncustomers['club_member_status'].value_counts()\ncustomers['fashion_news_frequency'].value_counts()\ncustomers['Active'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-09T12:30:22.702021Z","iopub.execute_input":"2023-03-09T12:30:22.702967Z","iopub.status.idle":"2023-03-09T12:30:44.351585Z","shell.execute_reply.started":"2023-03-09T12:30:22.702925Z","shell.execute_reply":"2023-03-09T12:30:44.348948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### ARTICLES ###\n\n# Read the CSV file\narticles = pd.read_csv(INPUT_PATH + 'articles.csv')\n\n# Read some data info and null values info <\narticles.info()\narticles.isnull().sum()\n\narticles.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-09T12:30:44.355478Z","iopub.execute_input":"2023-03-09T12:30:44.356045Z","iopub.status.idle":"2023-03-09T12:30:45.637035Z","shell.execute_reply.started":"2023-03-09T12:30:44.355984Z","shell.execute_reply":"2023-03-09T12:30:45.635511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### TRANSACTIONS ###\n\n# Read the CSV file\ntransactions = pd.read_csv(INPUT_PATH + 'transactions_train.csv')\n\n# Read some data info and null values info <\ntransactions.info()\ntransactions.isnull().sum()\n\n# Describe numeriuc fields mean, count & std\ntransactions.describe()","metadata":{"execution":{"iopub.status.busy":"2023-03-09T12:30:45.638946Z","iopub.execute_input":"2023-03-09T12:30:45.639358Z","iopub.status.idle":"2023-03-09T12:32:23.121453Z","shell.execute_reply.started":"2023-03-09T12:30:45.639323Z","shell.execute_reply":"2023-03-09T12:32:23.120031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preparation (CRISP DM)","metadata":{}},{"cell_type":"code","source":"### CUSTOMERS ###\n# Remove FN column as it has a lot of missing data and we could not use it reliably in a model \ncustomers = customers.drop(['FN'], axis=1)\n\n# Use mean value of the age to deal with null value\ncustomers['age'] = customers['age'].fillna(customers['age'].mean())\n\n# Typecas values to int \ncustomers['age'] = customers['age'].astype(int64)\n\n# Deal with the NaN values\ncustomers['club_member_status'] = customers['club_member_status'].replace(np.NaN, 'UNKOWN')\n\n# Deal with several types of unknown values \ncustomers['fashion_news_frequency'] = customers['fashion_news_frequency'].replace('None', 'UNKOWN')\ncustomers['fashion_news_frequency'] = customers['fashion_news_frequency'].replace('NONE', 'UNKOWN')\ncustomers['fashion_news_frequency'] = customers['fashion_news_frequency'].replace(np.NaN, 'UNKOWN')\n\n# Lowercase the active column as it is the only uppercase column name \ncustomers = customers.rename(columns={'Active':'active'})\n\n# Fill in missing active values with 0\ncustomers['active'] = customers['active'].fillna(0)\n\n# Typecast values to 64 int \ncustomers['active'] = customers['active'].astype(int64)\n\ncustomers.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-09T12:32:23.123304Z","iopub.execute_input":"2023-03-09T12:32:23.123800Z","iopub.status.idle":"2023-03-09T12:32:23.723904Z","shell.execute_reply.started":"2023-03-09T12:32:23.123750Z","shell.execute_reply":"2023-03-09T12:32:23.722613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### TRANSACTIONS ###\n\ntransactions['t_dat'] = pd.to_datetime(transactions['t_dat'])\n\ntransactions = transactions[['t_dat','customer_id','article_id']]\n\ntransactions_customer_article_grp = transactions.groupby(['customer_id', 'article_id'])['t_dat'].count().reset_index().rename(columns={'t_dat':'count'})\n\ntransactions_customer_article_grp.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-09T12:32:23.725196Z","iopub.execute_input":"2023-03-09T12:32:23.726456Z","iopub.status.idle":"2023-03-09T12:33:15.424037Z","shell.execute_reply.started":"2023-03-09T12:32:23.726411Z","shell.execute_reply":"2023-03-09T12:33:15.421723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create tensorflow datasets from Data Frames\narticles_ds = tf.data.Dataset.from_tensor_slices(dict(articles[['article_id']].astype(str)))\narticles_ds = articles_ds.map(lambda x: str(x[\"article_id\"]))\n\ncustomers_ds = tf.data.Dataset.from_tensor_slices(dict(customers[['customer_id']].astype(str)))\ncustomers_ds = customers_ds.map(lambda x: str(x[\"customer_id\"]))\n\ncustomer_articles_ds = tf.data.Dataset.from_tensor_slices(dict(transactions_customer_article_grp.astype(str)))\ncustomer_articles_ds = customer_articles_ds.map(lambda x: {\"customer_id\": x[\"customer_id\"], \"article_id\": x[\"article_id\"]})\n\n# Extract a random training and test data 80% train and 20% test data\ntf.random.set_seed(42)\ncustomer_articles_size = transactions_customer_article_grp.size\nshuffled = customer_articles_ds.shuffle(customer_articles_size, seed=42, reshuffle_each_iteration=False)\n\n# TODO remove this when we run the actual tests \n# This is reducing the actual sample size so the script can run faster, but for real world scenarios this line should be removed\ncustomer_articles_size = round(customer_articles_size * 0.01)\n\ntrain_size = round(customer_articles_size * 0.8)\ntest_size = customer_articles_size - train_size\ntrain = shuffled.take(train_size)\ntest = shuffled.skip(train_size).take(test_size)\n\n# Get the unique articles and user ids. We can also skip this as our users and articles are unique, \n# but this is just for a demo purposa that we should taker care for this\narticle_ids = articles_ds.batch(1_000)\ncustomer_ids = customers_ds.batch(1_000)\n\nunique_article_ids = np.unique(np.concatenate(list(article_ids)))\nunique_customer_ids = np.unique(np.concatenate(list(customer_ids)))","metadata":{"execution":{"iopub.status.busy":"2023-03-09T12:33:15.426719Z","iopub.execute_input":"2023-03-09T12:33:15.427156Z","iopub.status.idle":"2023-03-09T12:34:24.929924Z","shell.execute_reply.started":"2023-03-09T12:33:15.427116Z","shell.execute_reply":"2023-03-09T12:34:24.928227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modelling (CRISP DM)","metadata":{}},{"cell_type":"code","source":"# We will only ouse one ML model based on collaborative filtering recommendation, but for better results we should also combine it later with atleast a content based recommendation system. \n# Please note that this is just for demo purposes and is basic\n\n# Set the embedding dimensions size. \n# Higher values will correspond to models that may be more accurate, but will also be slower to fit and more prone to overfitting.\nembedding_dimension = 32\n\n# Define the query tower model\ncustomer_model = tf.keras.Sequential([\n  tf.keras.layers.StringLookup(vocabulary=unique_customer_ids, mask_token=None),\n  # We add an additional embedding to account for unknown tokens.\n  tf.keras.layers.Embedding(len(unique_customer_ids) + 1, embedding_dimension)\n])\n\n# Define the candidate tower model\narticle_model = tf.keras.Sequential([\n  tf.keras.layers.StringLookup(vocabulary=unique_article_ids, mask_token=None),\n  # We add an additional embedding to account for unknown tokens.\n  tf.keras.layers.Embedding(len(unique_article_ids) + 1, embedding_dimension)\n])\n\n# Define a positive training metric model that uses the customer - article association as a positive result \nmetrics = tfrs.metrics.FactorizedTopK(\n  candidates=articles_ds.batch(128).map(article_model)\n)\n\n# Define a loss layer\ntask = tfrs.tasks.Retrieval(\n  metrics=metrics\n)\n\n# Define the ML Model to use \nclass ArticleLensModel(tfrs.Model):\n\n  def __init__(self, customer_model, article_model):\n    super().__init__()\n    self.article_model: tf.keras.Model = article_model\n    self.customer_model: tf.keras.Model = customer_model\n    self.task: tf.keras.layers.Layer = task\n\n  def compute_loss(self, features: Dict[Text, tf.Tensor], training=False) -> tf.Tensor:\n    \n    # We pick out the customer features and pass them into the customer model.\n    customer_embeddings = self.customer_model(features[\"customer_id\"])\n    \n    # And pick out the article features and pass them into the customer model,\n    # getting embeddings back.\n    positive_article_embeddings = self.article_model(features[\"article_id\"])\n\n    # The task computes the loss and the metrics.\n    return self.task(customer_embeddings, positive_article_embeddings)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T12:34:24.932197Z","iopub.execute_input":"2023-03-09T12:34:24.932791Z","iopub.status.idle":"2023-03-09T12:34:25.120376Z","shell.execute_reply.started":"2023-03-09T12:34:24.932721Z","shell.execute_reply":"2023-03-09T12:34:25.118989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build a model instance with the customers and articles model and use Adagrad optimizers module\nmodel = ArticleLensModel(customer_model, article_model)\nmodel.compile(optimizer=tf.keras.optimizers.Adagrad(learning_rate=0.1))\n\n# Shuffle and cache training data\ncached_train = train.shuffle(customer_articles_size).batch(8192).cache()\n\n# Train the model with the training data - 3 iterations of training\nmodel.fit(cached_train, epochs=3)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T13:59:45.157756Z","iopub.execute_input":"2023-03-09T13:59:45.158297Z","iopub.status.idle":"2023-03-09T13:59:45.180968Z","shell.execute_reply.started":"2023-03-09T13:59:45.158255Z","shell.execute_reply":"2023-03-09T13:59:45.178613Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluation (CRISP DM)","metadata":{}},{"cell_type":"code","source":"# Evaluate the results with the cached test data \ncached_test = test.batch(4096).cache()\nmodel.evaluate(cached_test, return_dict=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T12:53:37.883698Z","iopub.execute_input":"2023-03-09T12:53:37.884142Z","iopub.status.idle":"2023-03-09T13:01:42.718330Z","shell.execute_reply.started":"2023-03-09T12:53:37.884101Z","shell.execute_reply":"2023-03-09T13:01:42.717329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Deployment (CRISP DM)","metadata":{}},{"cell_type":"code","source":"# Create a model that takes in raw query features.\n# Brute force is slow, so we should optimize the model, but as a first time iteration is seems ok \nindex = tfrs.layers.factorized_top_k.BruteForce(model.customer_model)\n\n# recommends articles out of the entire articles dataset.\nindex.index_from_dataset(\n    articles_ds.batch(100).map(lambda article_id: (article_id, model.article_model(article_id)))\n)\n\n# Retrieve the customer recommendations for every customer and put in a dictionary \ncustomer_recommendations = {}\nfor customer_idx, customer in customers.iterrows():\n    \n    # Get the recommended articles \n    article_ids = index(np.array([customer['customer_id']])) \n    customer_recommendations[customer['customer_id']] = np.array2string(article_ids[1].numpy(), separator=' ')\n    \n    # Print the putput for the user\n    print(customer['customer_id'] + ': ' + customer_recommendations[customer['customer_id']])\n\n# Create a new datafarme having all the customer_ids as one column and then add the prediction column to the dataset \ncustomer_recommendations_df = customers[['customer_id']]\ncustomer_recommendations_df[\"prediction\"] = customer_recommendations_df[\"customer_id\"].map(customer_recommendations)\n\n# Output the customer recommendations dataframe to a predictions file \ncustomer_recommendations_df.to_csv('/kaggle/working/predictions.csv')\n    \n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2023-03-09T13:56:50.002925Z","iopub.status.idle":"2023-03-09T13:56:50.004398Z","shell.execute_reply.started":"2023-03-09T13:56:50.004018Z","shell.execute_reply":"2023-03-09T13:56:50.004063Z"},"trusted":true},"execution_count":null,"outputs":[]}]}