{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# OTTO Dataset as Item-Type Joint Pairs\n\nIn this notebook we will process the optimized [raw OTTO splits in .parquet format](https://www.kaggle.com/code/federicominutoli/otto-dataset-as-parquet) into sequences of item-type joint pairs.\n\nNote that we will refer to any item-type joint pair in the dataset as a **token**.\n\nThe constructed pairs are suitable for next-token prediction tasks, which are relevant both for the construction of item, type or item-type joint embeddings, or for submissions to the OTTO challenge, e.g., by using [Transformer-based architectures](https://www.kaggle.com/code/theoviel/pretraining-with-merlin-s-transformers4rec).","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"## 0. Setup","metadata":{}},{"cell_type":"code","source":"import enum\nimport gc\nimport pathlib\nimport typing\nfrom typing import Any","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT = pathlib.Path(\"/kaggle\")\nROOT_tmp = pathlib.Path(\"/tmp\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT_input = ROOT / \"input\" / \"otto-dataset-as-parquet\"\nROOT_working = ROOT / \"working\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert ROOT_input.exists(), \"Did you forget to attach the dataset?\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile $ROOT_tmp/requirements.txt\npolars==0.15.9","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --upgrade pip\n!pip install --upgrade -r $ROOT_tmp/requirements.txt","metadata":{"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport polars as pl\nfrom scipy.interpolate import interp1d\nfrom tqdm.auto import tqdm","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%matplotlib widget","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. Pre-processing","metadata":{}},{"cell_type":"markdown","source":"### 1.1 Timestamp decay","metadata":{}},{"cell_type":"code","source":"T_secs_hour = 60 * 60\nT_secs_day  = T_secs_hour * 24\nT_secs_week = T_secs_day * 7","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@np.vectorize\ndef timestamp_decay(secs: int) -> float:\n    \"\"\"The timestamp weighing function in [0, 1]\n    \n    It weighs two subsequent tokens depending on how close they have happened\n    \"\"\"\n    if secs < T_secs_hour:\n        return 1.\n    \n    if secs > T_secs_week:\n        return 0.\n    \n    freq = np.pi / (T_secs_week - T_secs_hour)\n    proj = interp1d([-1, 1], [0, 1])\n    offset = secs - T_secs_hour\n        \n    return proj(np.cos(offset * freq))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.2 Raw data to token pairs","metadata":{}},{"cell_type":"markdown","source":"Let us pre-process sequences of `(token1, token2)` pairs as Polars dataframes!","metadata":{}},{"cell_type":"code","source":"FILENAME_X_training = \"X_training.parquet\"\nFILENAME_y_training = \"y_training.parquet\"\nFILENAME_X_test = \"X_test.parquet\"\nFILENAME_y_test = \"y_test.parquet\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _raw_data_to_data_pairs(raw_data: pl.LazyFrame) -> pl.DataFrame:\n    \"\"\"It converts raw OTTO data to a sequence of `(token1, token2)` pairs\"\"\"\n    as_token1 = {\"aid\": \"item1\", \"type\": \"type1\"}\n    \n    as_item2 = pl.col(\"item1\").shift(-1).over(\"session\").alias(\"item2\")\n    as_type2 = pl.col(\"type1\").shift(-1).over(\"session\").alias(\"type2\")\n    \n    as_token2 = [as_item2, as_type2]\n    \n    inp_cols = [\"session\", \"aid\", \"type\"]\n    out_cols = [\"item1\", \"type1\", \"item2\", \"type2\"]\n    \n    data_pairs = (\n        raw_data\n            .select(inp_cols)\n            .rename(as_token1)\n            .with_columns(as_token2)\n            .select(out_cols)\n            .drop_nulls()\n    )\n    \n    data_pairs = data_pairs.collect()    \n    return data_pairs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"freq = np.pi / (T_secs_week - T_secs_hour)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _raw_data_to_scores(raw_data: pl.LazyFrame) -> pl.DataFrame:\n    \"\"\"It converts raw OTTO data to a sequence of timestamp weighted scores\"\"\"\n    as_timestamp1 = {\"ts\": \"timestamp1\"}\n    as_timestamp2 = pl.col(\"timestamp1\").shift(-1).over(\"session\").alias(\"timestamp2\")\n    \n    secs = np.abs(pl.col(\"timestamp1\") - pl.col(\"timestamp2\"))\n    amplitude = (np.cos((secs - T_secs_hour) * freq) + 1) / 2\n    \n    # A much faster Polars expr\n    as_scores = (\n        pl.when(secs < T_secs_hour)\n        .then(1.).otherwise(\n            pl.when(secs > T_secs_week)\n            .then(0.).otherwise(\n                amplitude\n            )\n        )\n        .cast(pl.Float32).alias(\"score\")\n    )\n    \n    inp_cols = [\"session\", \"ts\"]\n    out_cols = [\"score\"]\n    \n    scores = (\n        raw_data\n            .select(inp_cols)\n            .rename(as_timestamp1)\n            .with_columns(as_timestamp2)\n            .drop_nulls()\n            .with_columns(as_scores)\n            .select(out_cols)\n    )\n    \n    scores = scores.collect()    \n    return scores","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _preprocess_pairs(\n    src_X_training_filepath: pathlib.Path,\n    src_X_test_filepath: pathlib.Path,\n    dst_dir_path: pathlib.Path\n) -> None:\n    \"\"\"It pre-processes sequences of `(token1, token2)` pairs from raw splits of OTTO data\"\"\"    \n    dst_X_training_filepath = dst_dir_path / FILENAME_X_training\n    dst_y_training_filepath = dst_dir_path / FILENAME_y_training\n    dst_X_test_filepath = dst_dir_path / FILENAME_X_test\n    dst_y_test_filepath = dst_dir_path / FILENAME_y_test\n    \n    data_splits = [\n        (src_X_training_filepath, dst_X_training_filepath, dst_y_training_filepath),\n        (src_X_test_filepath, dst_X_test_filepath, dst_y_test_filepath)\n    ]\n    \n    for src_X, dst_X, dst_y in tqdm(data_splits):    \n        data = pl.scan_parquet(src_X)\n        \n        if not dst_X.exists():\n            data_pairs = _raw_data_to_data_pairs(data)\n            data_pairs.write_parquet(dst_X)\n            \n            del data_pairs\n            gc.collect()\n\n            data_pairs = pd.read_parquet(dst_X)\n            data_pairs.to_parquet(dst_X, index=False)\n\n            del data_pairs\n            gc.collect()\n            \n            \n        if not dst_y.exists():\n            scores = _raw_data_to_scores(data)\n            scores.write_parquet(dst_y)\n            \n            del scores\n            gc.collect()\n\n            scores = pd.read_parquet(dst_y)\n            scores.to_parquet(dst_y, index=False)\n\n            del scores\n            gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let us pre-process the custom re-arranged OTTO dataset!","metadata":{}},{"cell_type":"code","source":"def preprocess(\n    src_X_training_filepath: pathlib.Path,\n    src_X_test_filepath: pathlib.Path,\n    dst_dir_path: pathlib.Path\n) -> None:\n    \"\"\"It pre-processes raw OTTO data splits\"\"\"\n    _preprocess_pairs(\n        src_X_training_filepath, src_X_test_filepath, dst_dir_path\n    )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"src_X_training_filepath = ROOT_input / \"full\" / \"X_training.parquet\"\nsrc_X_test_filepath = ROOT_input / \"X_test.parquet\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocess(src_X_training_filepath, src_X_test_filepath, ROOT_working)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_training_filepath = ROOT_working / FILENAME_X_training\ny_training_filepath = ROOT_working / FILENAME_y_training\nX_test_filepath = ROOT_working / FILENAME_X_test\ny_test_filepath = ROOT_working / FILENAME_y_test","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for filepath in [\n    X_training_filepath,\n    y_training_filepath,\n    X_test_filepath,\n    y_test_filepath\n]:\n    assert filepath.exists(), \"Did `preprocess` not finish?\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**If you like this notebook, please smash the upvote button! Thank you! 😊**","metadata":{"execution":{"iopub.status.busy":"2022-11-18T02:49:02.940358Z","iopub.execute_input":"2022-11-18T02:49:02.940858Z","iopub.status.idle":"2022-11-18T02:49:02.973867Z","shell.execute_reply.started":"2022-11-18T02:49:02.94076Z","shell.execute_reply":"2022-11-18T02:49:02.972223Z"}}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}