{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-04T17:31:40.551113Z","iopub.execute_input":"2022-12-04T17:31:40.552132Z","iopub.status.idle":"2022-12-04T17:31:40.606643Z","shell.execute_reply.started":"2022-12-04T17:31:40.552029Z","shell.execute_reply":"2022-12-04T17:31:40.605588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# The purpose of this notebook:\nThis notebook has been created with the purpose of understanding how to preprocess the data given in the Otto dataset and use them with one of the Transformers algorithms. What we do here is that we get a synthesized dataset and play a bit with the variables. This way can also be incorporated in the OTTO dataset. ","metadata":{}},{"cell_type":"markdown","source":"### Install nvtabular. For me this is the version that worked. ","metadata":{}},{"cell_type":"code","source":"!pip install -q -U nvtabular==1.3.3","metadata":{"execution":{"iopub.status.busy":"2022-12-04T17:31:55.785324Z","iopub.execute_input":"2022-12-04T17:31:55.785846Z","iopub.status.idle":"2022-12-04T17:33:58.966706Z","shell.execute_reply.started":"2022-12-04T17:31:55.785804Z","shell.execute_reply":"2022-12-04T17:33:58.965479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### You will also need to install merlin","metadata":{}},{"cell_type":"code","source":"!pip install cudf","metadata":{"execution":{"iopub.status.busy":"2022-12-04T17:22:26.888417Z","iopub.execute_input":"2022-12-04T17:22:26.889494Z","iopub.status.idle":"2022-12-04T17:22:36.546376Z","shell.execute_reply.started":"2022-12-04T17:22:26.889451Z","shell.execute_reply":"2022-12-04T17:22:36.545200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install merlin","metadata":{"execution":{"iopub.status.busy":"2022-12-04T17:22:39.994936Z","iopub.execute_input":"2022-12-04T17:22:39.995316Z","iopub.status.idle":"2022-12-04T17:22:50.046451Z","shell.execute_reply.started":"2022-12-04T17:22:39.995280Z","shell.execute_reply":"2022-12-04T17:22:50.045249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import you will need\nHere you will find all the imports needed to run this notebook.","metadata":{}},{"cell_type":"code","source":"import os\nimport glob\n\nimport numpy as np\nimport pandas as pd\n\nimport cudf\nimport cupy as cp\nimport nvtabular as nvt\nfrom nvtabular.ops import *\nfrom merlin.schema.tags import Tags","metadata":{"execution":{"iopub.status.busy":"2022-12-04T17:33:58.970608Z","iopub.execute_input":"2022-12-04T17:33:58.970979Z","iopub.status.idle":"2022-12-04T17:34:01.416332Z","shell.execute_reply.started":"2022-12-04T17:33:58.970947Z","shell.execute_reply":"2022-12-04T17:34:01.415273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### What we do here?\n* Here we create a dataframe similar to the one we have in Otto\n* This dataframe has a session_id and an item_id\n* Further we create two columns with timestamp/age_days and timestamp_weekday_sin respectevily","metadata":{}},{"cell_type":"code","source":"NUM_ROWS = 10000000\nlong_tailed_item_distribution = np.clip(np.random.lognormal(3., 1., NUM_ROWS).astype(np.int32), 1, 50000)\n\n# generate random item interaction features \ndf = pd.DataFrame(np.random.randint(70000, 90000, NUM_ROWS), columns=['session_id'])\ndf['item_id'] = long_tailed_item_distribution\n\n# generate category mapping for each item-id\ndf['timestamp/age_days'] = np.random.uniform(0, 1, NUM_ROWS).astype(np.float32)\ndf['timestamp/weekday/sin']= np.random.uniform(0, 1, NUM_ROWS).astype(np.float32)\n\n# generate day mapping for each session \nmap_day = dict(zip(df.session_id.unique(), np.random.randint(1, 10, size=(df.session_id.nunique()))))\ndf['day'] =  df.session_id.map(map_day)","metadata":{"execution":{"iopub.status.busy":"2022-12-04T17:34:06.901039Z","iopub.execute_input":"2022-12-04T17:34:06.901418Z","iopub.status.idle":"2022-12-04T17:34:08.441200Z","shell.execute_reply.started":"2022-12-04T17:34:06.901386Z","shell.execute_reply":"2022-12-04T17:34:08.440134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### How do we use the nvtabular?\n\n* NVTabular helps us turn categorical variables to numerical ones by using the Categorify method\n* This turns all the categorical variables (session_id, item_id etc) into weighted numerical representations\n* In this way we can use these newly representations as inputs when using Transformers","metadata":{}},{"cell_type":"code","source":"# Categorify categorical features\ncateg_feats = ['session_id', 'item_id'] >> nvt.ops.Categorify(start_index=1)\n\n# Define Groupby Workflow\ngroupby_feats = categ_feats + ['day', 'timestamp/age_days', 'timestamp/weekday/sin']\n\n# Group interaction features by session\ngroupby_features = groupby_feats >> nvt.ops.Groupby(\n    groupby_cols=[\"session_id\"], \n    aggs={\n        \"item_id\": [\"list\", \"count\"],   \n        \"day\": [\"first\"],\n        \"timestamp/age_days\": [\"list\"],\n        'timestamp/weekday/sin': [\"list\"],\n        },\n    name_sep=\"-\")\n\n# Select and truncate the sequential features\n\nsequence_features_truncated_item = (\n    groupby_features['item_id-list']\n    >> nvt.ops.ListSlice(0,20) \n    >> nvt.ops.Rename(postfix = '_trim')\n    >> TagAsItemID()\n)  \nsequence_features_truncated_cont = (\n    groupby_features['timestamp/age_days-list', 'timestamp/weekday/sin-list'] \n    >> nvt.ops.ListSlice(0,20) \n    >> nvt.ops.Rename(postfix = '_trim')\n    >> nvt.ops.AddMetadata(tags=[Tags.CONTINUOUS])\n)\n\n# Filter out sessions with length 1 (not valid for next-item prediction training and evaluation)\nMINIMUM_SESSION_LENGTH = 2\nselected_features = (\n    groupby_features['item_id-count', 'day-first', 'session_id'] + \n    sequence_features_truncated_item + \n    sequence_features_truncated_cont\n)\n    \nfiltered_sessions = selected_features >> nvt.ops.Filter(f=lambda df: df[\"item_id-count\"] >= MINIMUM_SESSION_LENGTH)\n\n\nworkflow = nvt.Workflow(filtered_sessions)\ndataset = nvt.Dataset(df, cpu=False)\n# Generate statistics for the features\nworkflow.fit(dataset)\n# Apply the preprocessing and return an NVTabular dataset\nsessions_ds = workflow.transform(dataset)\n# Convert the NVTabular dataset to a Dask cuDF dataframe (`to_ddf()`) and then to cuDF dataframe (`.compute()`)\nsessions_gdf = sessions_ds.to_ddf().compute()","metadata":{"execution":{"iopub.status.busy":"2022-12-04T17:34:11.114893Z","iopub.execute_input":"2022-12-04T17:34:11.115309Z","iopub.status.idle":"2022-12-04T17:34:21.975131Z","shell.execute_reply.started":"2022-12-04T17:34:11.115277Z","shell.execute_reply":"2022-12-04T17:34:21.974014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Output\nHere you can see the new weighted values of each column","metadata":{}},{"cell_type":"code","source":"sessions_gdf.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-04T17:34:21.977585Z","iopub.execute_input":"2022-12-04T17:34:21.978127Z","iopub.status.idle":"2022-12-04T17:34:22.022468Z","shell.execute_reply.started":"2022-12-04T17:34:21.978088Z","shell.execute_reply":"2022-12-04T17:34:22.021402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## What's next?\n* Use Transformers for RecSys \n* Use this code in OTTO\n\n# If you found this notebook helpful, give it a thumbs up :D","metadata":{}},{"cell_type":"markdown","source":"* This will save the workflow in kaggle/working directory","metadata":{}},{"cell_type":"code","source":"workflow.save('workflow_etl')","metadata":{"execution":{"iopub.status.busy":"2022-12-04T17:34:52.402281Z","iopub.execute_input":"2022-12-04T17:34:52.402667Z","iopub.status.idle":"2022-12-04T17:34:52.426957Z","shell.execute_reply.started":"2022-12-04T17:34:52.402631Z","shell.execute_reply":"2022-12-04T17:34:52.425466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"workflow.fit_transform(dataset).to_parquet(os.path.join('/kaggle/working', \"processed_nvt\"))\n","metadata":{"execution":{"iopub.status.busy":"2022-12-04T17:34:54.972092Z","iopub.execute_input":"2022-12-04T17:34:54.972462Z","iopub.status.idle":"2022-12-04T17:34:56.976623Z","shell.execute_reply.started":"2022-12-04T17:34:54.972429Z","shell.execute_reply":"2022-12-04T17:34:56.975689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we will split the preprocessed parquet files by days in order to allow for temporal training and evaluation. Now there will be a folder for each day and three parquet files within each day folder: train.parquet, validation.parquet and test.parquet","metadata":{}},{"cell_type":"code","source":"OUTPUT_DIR = os.environ.get(\"OUTPUT_DIR\",os.path.join('/kaggle/working', \"sessions_by_day\"))\n!mkdir -p $OUTPUT_DIR","metadata":{"execution":{"iopub.status.busy":"2022-12-04T17:35:10.452463Z","iopub.execute_input":"2022-12-04T17:35:10.452893Z","iopub.status.idle":"2022-12-04T17:35:11.496625Z","shell.execute_reply.started":"2022-12-04T17:35:10.452856Z","shell.execute_reply":"2022-12-04T17:35:11.495092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install transformers4rec","metadata":{"execution":{"iopub.status.busy":"2022-12-04T17:36:24.045306Z","iopub.execute_input":"2022-12-04T17:36:24.045718Z","iopub.status.idle":"2022-12-04T17:36:56.487837Z","shell.execute_reply.started":"2022-12-04T17:36:24.045665Z","shell.execute_reply":"2022-12-04T17:36:56.486419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers4rec.data.preprocessing import save_time_based_splits","metadata":{"execution":{"iopub.status.busy":"2022-12-04T17:37:38.228411Z","iopub.execute_input":"2022-12-04T17:37:38.228865Z","iopub.status.idle":"2022-12-04T17:37:38.301797Z","shell.execute_reply.started":"2022-12-04T17:37:38.228822Z","shell.execute_reply":"2022-12-04T17:37:38.300896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_time_based_splits(data=nvt.Dataset(sessions_gdf),\n                       output_dir= OUTPUT_DIR,\n                       partition_col='day-first',\n                       timestamp_col='session_id', \n                      )","metadata":{"execution":{"iopub.status.busy":"2022-12-04T17:37:44.656371Z","iopub.execute_input":"2022-12-04T17:37:44.656752Z","iopub.status.idle":"2022-12-04T17:37:49.275341Z","shell.execute_reply.started":"2022-12-04T17:37:44.656718Z","shell.execute_reply":"2022-12-04T17:37:49.274178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_PATHS = sorted(glob.glob(os.path.join(OUTPUT_DIR, \"1\", \"train.parquet\")))","metadata":{"execution":{"iopub.status.busy":"2022-12-04T17:37:52.122938Z","iopub.execute_input":"2022-12-04T17:37:52.123309Z","iopub.status.idle":"2022-12-04T17:37:52.128674Z","shell.execute_reply.started":"2022-12-04T17:37:52.123277Z","shell.execute_reply":"2022-12-04T17:37:52.127694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gdf = cudf.read_parquet(TRAIN_PATHS[0])\ngdf","metadata":{"execution":{"iopub.status.busy":"2022-12-04T17:37:55.336576Z","iopub.execute_input":"2022-12-04T17:37:55.337277Z","iopub.status.idle":"2022-12-04T17:37:55.446715Z","shell.execute_reply.started":"2022-12-04T17:37:55.337242Z","shell.execute_reply":"2022-12-04T17:37:55.445669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}