{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":4461402,"sourceType":"datasetVersion","datasetId":2611514},{"sourceId":4474043,"sourceType":"datasetVersion","datasetId":2601572}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# OTTO GRU4Rec with CUDF Covisitation Matrix\n\nPipeline sử dụng CUDF (RAPIDS GPU) để build covisitation matrices + GRU4Rec cho predictions.\n\n**Data: hgy1 và hgy2**","metadata":{}},{"cell_type":"code","source":"# Install dependencies\n!pip install pandas numpy matplotlib seaborn polars pyarrow tqdm h5py pydantic -q\n!pip install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cu118 -q\n!pip install recbole -q\n!pip install protobuf==3.20.0 -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T01:32:52.778538Z","iopub.execute_input":"2025-12-09T01:32:52.779304Z","iopub.status.idle":"2025-12-09T01:35:37.069582Z","shell.execute_reply.started":"2025-12-09T01:32:52.779274Z","shell.execute_reply":"2025-12-09T01:35:37.068733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====== IMPORT THƯ VIỆN ======\nimport os\nimport sys\nimport gc\nimport glob\nimport random\nfrom collections import defaultdict, Counter\nfrom multiprocessing import Pool\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\n# RAPIDS cudf for GPU acceleration\nimport cudf\nimport numba\n\nfrom tqdm.auto import tqdm\nimport pyarrow.parquet as pq","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:06:25.542600Z","iopub.execute_input":"2025-12-09T02:06:25.543396Z","iopub.status.idle":"2025-12-09T02:06:35.346455Z","shell.execute_reply.started":"2025-12-09T02:06:25.543363Z","shell.execute_reply":"2025-12-09T02:06:35.345421Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1. Configuration & Parameters","metadata":{}},{"cell_type":"code","source":"# ====== OTTO PARAMETERS ======\n\n# Event type mapping\nTYPE_LABELS = {\n    \"clicks\": 0, \n    \"carts\": 1, \n    \"orders\": 2\n}\n\n# Reverse mapping\nID_TO_TYPE = {v: k for k, v in TYPE_LABELS.items()}\n\n# Matrix computation config\nDISK_PIECES = 4  # Number of partitions for memory management\nTOP_N_COVISIT = 20  # Top N items to keep per source item\n\n# RecBole config\nMAX_ITEM = 20","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:06:57.562509Z","iopub.execute_input":"2025-12-09T02:06:57.563377Z","iopub.status.idle":"2025-12-09T02:06:57.567743Z","shell.execute_reply.started":"2025-12-09T02:06:57.563351Z","shell.execute_reply":"2025-12-09T02:06:57.566965Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Matrix Utility Functions","metadata":{}},{"cell_type":"code","source":"def df_parallelize_run(func, t_split):\n    \"\"\"\n    Pool multiprocessing for speedup.\n    \"\"\"\n    num_cores = np.min([20, len(t_split)])\n    pool = Pool(num_cores)\n    df = pool.map(func, t_split)\n    pool.close()\n    pool.join()\n    return df\n\n\ndef matrix_to_candids_dict(matrix):\n    \"\"\"\n    Converts a matrix to a dict of candidates sorted by weight.\n    \"\"\"\n    matrix = matrix.sort_values([\"aid_x\", \"wgt\"], ascending=[True, False])\n    candids = matrix[[\"aid_x\", \"aid_y\"]].groupby(\"aid_x\").agg(list)\n    \n    try:\n        candids = candids.to_pandas()\n    except AttributeError:\n        pass\n\n    candids[\"aid_y\"] = candids[\"aid_y\"].apply(lambda x: x.tolist() if hasattr(x, 'tolist') else x)\n    candids_dict = candids.to_dict()[\"aid_y\"]\n\n    return candids_dict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:07:02.107514Z","iopub.execute_input":"2025-12-09T02:07:02.108249Z","iopub.status.idle":"2025-12-09T02:07:02.113765Z","shell.execute_reply.started":"2025-12-09T02:07:02.108219Z","shell.execute_reply":"2025-12-09T02:07:02.113039Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Covisitation Matrix Computation (CUDF)","metadata":{}},{"cell_type":"code","source":"def read_file_to_cache(f):\n    \"\"\"\n    Reads a parquet file to cache for later use.\n    Data already has type as int (0, 1, 2), no mapping needed.\n    \"\"\"\n    df = pd.read_parquet(f)\n    \n    # Convert timestamp from ms to seconds if needed\n    # Check if ts is in milliseconds (very large number)\n    if df.ts.max() > 1e12:\n        df.ts = (df.ts / 1000).astype(\"int32\")\n    else:\n        df.ts = df.ts.astype(\"int32\")\n    df = df[df[\"type\"].notna()]\n    df[\"type\"] = df[\"type\"].astype(\"int8\")\n\n    return df\n\n\ndef read_file(f, data_cache):\n    \"\"\"\n    Converts cached pandas DataFrame to cudf DataFrame.\n    \"\"\"\n    return cudf.DataFrame(data_cache[f])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:07:05.099796Z","iopub.execute_input":"2025-12-09T02:07:05.100923Z","iopub.status.idle":"2025-12-09T02:07:05.106070Z","shell.execute_reply.started":"2025-12-09T02:07:05.100887Z","shell.execute_reply":"2025-12-09T02:07:05.105281Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\ndef compute_covisitation_matrix_optimized(\n    files,\n    weighting=\"\",\n    type_weight={},\n    considered_types=[0, 1, 2],\n    n=20,\n    save_folder=\"\",\n    suffix=\"\",\n    chunk_size=1,  # Process 1 file at a time to minimize memory\n    max_session_length=20,  # Limit to 20 events per session to reduce pairs\n):\n    \"\"\"\n    Memory-optimized covisitation matrix computation.\n    \n    Key differences from original:\n    - No pre-caching: loads files directly to GPU\n    - Smaller chunk size: 5 files instead of 10\n    - More aggressive memory cleanup\n    \n    Args:\n        files (list): List of parquet filenames to process.\n        weighting (str): \"temporal\", \"type\", or \"\" (none).\n        type_weight (dict): Event type weights for weighting=\"type\".\n        considered_types (list): Event types to include [0, 1, 2].\n        n (int): Top N items to keep per source item.\n        save_folder (str): Folder to save the matrix.\n        suffix (str): Save suffix.\n        chunk_size (int): Number of files to process at once.\n        max_session_length (int): Max events per session to consider.\n    \n    Returns:\n        pandas DataFrame: Covisitation matrix.\n    \"\"\"\n    DISK_PIECES = 4\n    SIZE = 1.86e6 / DISK_PIECES\n\n    # Process files in smaller chunks\n    chunks = [files[x: x + chunk_size] for x in range(0, len(files), chunk_size)]\n\n    matrices = []\n    for part in range(DISK_PIECES):\n        print(f\"\\n{'='*50}\")\n        print(f\"Processing partition {part + 1}/{DISK_PIECES}\")\n        \n        matrix = None\n        \n        for idx, chunk in enumerate(tqdm(chunks, desc=f\"Part {part}\")):\n            try:\n                # LAZY LOAD - read directly to GPU, no caching\n                df = cudf.concat(\n                    [read_file_lazy(file) for file in chunk], ignore_index=True\n                )\n\n                # Filter by considered event types\n                if considered_types != [0, 1, 2]:\n                    df = df.loc[df[\"type\"].isin(considered_types)]\n\n                # Sort by session and timestamp (descending for recency)\n                df = df.sort_values([\"session\", \"ts\"], ascending=[True, False])\n\n                # LIMIT SESSION LENGTH - keep only last N events per session\n                df = df.reset_index(drop=True)\n                df[\"n\"] = df.groupby(\"session\").cumcount()\n                df = df.loc[df.n < max_session_length].drop(\"n\", axis=1)\n\n                # PARTITION EARLY - filter source items before self-join to save memory\n                df = df.loc[(df.aid >= part * SIZE) & (df.aid < (part + 1) * SIZE)]\n                \n                if len(df) == 0:\n                    del df\n                    clear_gpu_memory()\n                    continue\n\n                # CREATE PAIRS - self-join on session\n                # Rename columns before merge to avoid _x, _y suffix issues\n                df_left = df.rename(columns={\"aid\": \"aid_x\", \"ts\": \"ts_x\", \"type\": \"type_x\"})\n                df_right = df.rename(columns={\"aid\": \"aid_y\", \"ts\": \"ts_y\", \"type\": \"type_y\"})\n                \n                del df\n                clear_gpu_memory()\n                \n                df = df_left.merge(df_right, on=\"session\")\n                \n                del df_left, df_right\n                clear_gpu_memory()\n                \n                # Filter pairs:\n                # - Within 24 hours of each other\n                # - Different items (no self-loops)\n                df = df.loc[\n                    ((df.ts_x - df.ts_y).abs() < 24 * 60 * 60) & (df.aid_x != df.aid_y)\n                ]\n\n                # ASSIGN WEIGHTS - drop duplicates first\n                df = df[[\"session\", \"aid_x\", \"aid_y\", \"ts_x\", \"type_y\"]].drop_duplicates(\n                    [\"session\", \"aid_x\", \"aid_y\"]\n                )\n\n                if weighting == \"temporal\":\n                    df.drop(\"type_y\", axis=1, inplace=True)\n                    df[\"wgt\"] = 1 + 3 * (df.ts_x - 1659304800) / (1662328791 - 1659304800)\n                elif weighting == \"type\":\n                    df.drop(\"ts_x\", axis=1, inplace=True)\n                    df[\"wgt\"] = df.type_y.map(type_weight)\n                else:\n                    df.drop([\"type_y\", \"ts_x\"], axis=1, inplace=True)\n                    df[\"wgt\"] = 1\n\n                # Aggregate weights\n                df = df[[\"aid_x\", \"aid_y\", \"wgt\"]]\n                df.wgt = df.wgt.astype(\"float32\")\n                df = df.groupby([\"aid_x\", \"aid_y\"]).wgt.sum()\n\n                # COMBINE CHUNKS within partition\n                if matrix is None:\n                    matrix = df\n                else:\n                    matrix = matrix.add(df, fill_value=0)\n\n                del df\n                clear_gpu_memory()\n                \n            except Exception as e:\n                print(f\"Error processing chunk {idx}: {e}\")\n                clear_gpu_memory()\n                continue\n\n        if matrix is None or len(matrix) == 0:\n            print(f\"Warning: No data for partition {part}\")\n            continue\n            \n        # FINALIZE PARTITION\n        matrix = matrix.reset_index()\n        matrix = matrix.sort_values([\"aid_x\", \"wgt\"], ascending=[True, False])\n\n        # SAVE TOP N per source item\n        matrix = matrix.reset_index(drop=True)\n        matrix[\"n\"] = matrix.groupby(\"aid_x\").aid_y.cumcount()\n\n        if n:\n            matrix = matrix.loc[matrix.n < n].drop(\"n\", axis=1)\n\n        # Convert to pandas and store\n        matrices.append(matrix.to_pandas())\n        \n        del matrix\n        clear_gpu_memory()\n\n    if not matrices:\n        print(\"Warning: No matrices generated!\")\n        return pd.DataFrame()\n\n    # COMBINE ALL PARTITIONS\n    result = pd.concat(matrices, ignore_index=True)\n    del matrices\n    gc.collect()\n\n    # SAVE FINAL MATRIX\n    if save_folder:\n        if weighting == \"type\":\n            weighting_str = weighting + \"\".join(map(str, list(type_weight.values())))\n        else:\n            weighting_str = weighting\n        \n        save_path = os.path.join(\n            save_folder,\n            f'matrix_{\"\".join(map(str, considered_types))}_{weighting_str}_{n}_{suffix}.pqt',\n        )\n        print(f\"\\nSaving matrix to {save_path}\")\n        result.to_parquet(save_path)\n\n    clear_gpu_memory()\n    \n    return result\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:07:07.491904Z","iopub.execute_input":"2025-12-09T02:07:07.492489Z","iopub.status.idle":"2025-12-09T02:07:07.508010Z","shell.execute_reply.started":"2025-12-09T02:07:07.492461Z","shell.execute_reply":"2025-12-09T02:07:07.507251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_matrices_for_otto_optimized(\n    train_files,\n    save_folder=\"matrices\",\n    suffix=\"train\",\n    n=20,\n    chunk_size=1,  # Process 1 file at a time for safer memory\n):\n    \"\"\"\n    Memory-optimized version: Build all three covisitation matrices.\n    \n    Key difference: No pre-caching, uses lazy loading.\n    \n    Args:\n        train_files (list): List of training parquet files.\n        save_folder (str): Folder to save matrices.\n        suffix (str): Suffix for saved files.\n        n (int): Top N items to keep per source.\n        chunk_size (int): Files per chunk (lower = less memory).\n    \n    Returns:\n        tuple: (click_matrix, cart_matrix, order_matrix)\n    \"\"\"\n    os.makedirs(save_folder, exist_ok=True)\n    \n    print(\"=\"*60)\n    print(\"Building Click-to-Click Matrix...\")\n    print(\"=\"*60)\n    click_matrix = compute_covisitation_matrix_optimized(\n        files=train_files,\n        weighting=\"\",\n        considered_types=[0],  # clicks only\n        n=n,\n        save_folder=save_folder,\n        suffix=f\"clicks_{suffix}\",\n        chunk_size=chunk_size,\n    )\n    \n    # Cleanup between matrices\n    clear_gpu_memory()\n    \n    print(\"\\n\" + \"=\"*60)\n    print(\"Building Click/Cart-to-Cart Matrix...\")\n    print(\"=\"*60)\n    cart_matrix = compute_covisitation_matrix_optimized(\n        files=train_files,\n        weighting=\"type\",\n        type_weight={0: 1, 1: 6, 2: 3},\n        considered_types=[0, 1],  # clicks and carts\n        n=n,\n        save_folder=save_folder,\n        suffix=f\"carts_{suffix}\",\n        chunk_size=chunk_size,\n    )\n    \n    clear_gpu_memory()\n    \n    print(\"\\n\" + \"=\"*60)\n    print(\"Building Cart/Order-to-Order Matrix...\")\n    print(\"=\"*60)\n    order_matrix = compute_covisitation_matrix_optimized(\n        files=train_files,\n        weighting=\"type\",\n        type_weight={1: 6, 2: 3},\n        considered_types=[1, 2],  # carts and orders\n        n=n,\n        save_folder=save_folder,\n        suffix=f\"orders_{suffix}\",\n        chunk_size=chunk_size,\n    )\n    \n    print(\"\\n\" + \"=\"*60)\n    print(\"All matrices built successfully!\")\n    \n    return click_matrix, cart_matrix, order_matrix\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:07:11.491824Z","iopub.execute_input":"2025-12-09T02:07:11.492321Z","iopub.status.idle":"2025-12-09T02:07:11.498742Z","shell.execute_reply.started":"2025-12-09T02:07:11.492297Z","shell.execute_reply":"2025-12-09T02:07:11.498150Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Build Covisitation Matrices","metadata":{}},{"cell_type":"code","source":"# ====== LOAD DATA TO CACHE ======\n# Sử dụng hgy1 và hgy2 (single parquet files)\nTRAIN_FILES = [\n    '/kaggle/input/otto-train-and-test-data-for-local-validation/train.parquet',\n]\nSAVE_FOLDER = \"matrices\"\n\nprint(\"Loading data to cache...\")\nfiles = TRAIN_FILES\nprint(f\"Found {len(files)} parquet files\")\n\n# Cache all files in memory\n# data_cache = {}\n# for f in tqdm(files, desc=\"Caching files\"):\n#     data_cache[f] = read_file_to_cache(f)\n\n# print(f\"Cached {len(data_cache)} files\")\n# gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:07:14.920782Z","iopub.execute_input":"2025-12-09T02:07:14.921077Z","iopub.status.idle":"2025-12-09T02:07:14.926259Z","shell.execute_reply.started":"2025-12-09T02:07:14.921057Z","shell.execute_reply":"2025-12-09T02:07:14.925675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def read_file_lazy(f):\n    \"\"\"\n    Reads a parquet file directly to cudf DataFrame (no caching).\n    \n    Args:\n        f (str): File path to load.\n    \n    Returns:\n        cudf DataFrame: GPU-accelerated DataFrame.\n    \"\"\"\n    df = cudf.read_parquet(f)\n    df[\"ts\"] = (df[\"ts\"] / 1000).astype(\"int32\")\n    # Handle both string and numeric type columns\n    if df[\"type\"].dtype == \"object\":\n        df[\"type\"] = df[\"type\"].map(TYPE_LABELS).astype(\"int8\")\n    else:\n        df[\"type\"] = df[\"type\"].astype(\"int8\")\n    return df\n\ndef clear_gpu_memory():\n    \"\"\"Aggressively clear GPU memory.\"\"\"\n    numba.cuda.current_context().deallocations.clear()\n    gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:08:08.256787Z","iopub.execute_input":"2025-12-09T02:08:08.257534Z","iopub.status.idle":"2025-12-09T02:08:08.264049Z","shell.execute_reply.started":"2025-12-09T02:08:08.257502Z","shell.execute_reply":"2025-12-09T02:08:08.263285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Không cần pre-cache data_cache nữa!\n# Build matrices trực tiếp với lazy loading\nclick_matrix, cart_matrix, order_matrix = build_matrices_for_otto_optimized(\n    train_files=TRAIN_FILES,\n    save_folder=SAVE_FOLDER,\n    suffix=\"train\",\n    n=TOP_N_COVISIT,\n    chunk_size=1,  # Xử lý 1 file mỗi lần\n)\n\ngc.collect()\nprint(\"Done!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:08:11.859133Z","iopub.execute_input":"2025-12-09T02:08:11.859861Z","iopub.status.idle":"2025-12-09T02:08:53.830040Z","shell.execute_reply.started":"2025-12-09T02:08:11.859832Z","shell.execute_reply":"2025-12-09T02:08:53.829384Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====== CONVERT TO CANDIDATE DICTS ======\nprint(\"Converting matrices to candidate dictionaries...\")\n\ncovisit_click2click = matrix_to_candids_dict(click_matrix)\ncovisit_to_cart = matrix_to_candids_dict(cart_matrix)\ncovisit_to_order = matrix_to_candids_dict(order_matrix)\n\nprint(f\"Click candidates: {len(covisit_click2click):,} source items\")\nprint(f\"Cart candidates: {len(covisit_to_cart):,} source items\")\nprint(f\"Order candidates: {len(covisit_to_order):,} source items\")\n\n# Free matrix memory\ndel click_matrix, cart_matrix, order_matrix\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:08:58.301796Z","iopub.execute_input":"2025-12-09T02:08:58.302092Z","iopub.status.idle":"2025-12-09T02:10:08.098427Z","shell.execute_reply.started":"2025-12-09T02:08:58.302070Z","shell.execute_reply":"2025-12-09T02:10:08.097864Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. Create RecBole Dataset","metadata":{}},{"cell_type":"code","source":"# Đọc dữ liệu test của OTTO từ hgy1 và hgy2\ntrain_df = pl.read_parquet('/kaggle/input/otto-full-optimized-memory-footprint/test.parquet')\ntest_df = pl.read_parquet('/kaggle/input/otto-train-and-test-data-for-local-validation/test.parquet')\n\ninter_df = pl.concat([train_df, test_df])\n\n# Sắp xếp theo session, aid và timestamp\ninter_df = inter_df.sort(['session', 'aid', 'ts'])\n\n# RecBole format\ninter_df = inter_df.with_columns((pl.col('ts') * 1e9).alias('ts'))\ninter_df = inter_df.rename({'session': 'session:token', 'aid': 'aid:token', 'ts': 'ts:float'})\n\nprint(inter_df.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:10:35.233532Z","iopub.execute_input":"2025-12-09T02:10:35.234318Z","iopub.status.idle":"2025-12-09T02:10:37.527598Z","shell.execute_reply.started":"2025-12-09T02:10:35.234288Z","shell.execute_reply":"2025-12-09T02:10:37.526877Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"directory = \"recbox_data\"\nif not os.path.exists(directory):\n    os.makedirs(directory)\n    print(f\"Created: {directory}\")\nelse:\n    print(f\"Exists: {directory}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:10:43.190073Z","iopub.execute_input":"2025-12-09T02:10:43.190779Z","iopub.status.idle":"2025-12-09T02:10:43.194970Z","shell.execute_reply.started":"2025-12-09T02:10:43.190752Z","shell.execute_reply":"2025-12-09T02:10:43.194297Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pandas_df = inter_df[['session:token', 'aid:token', 'ts:float']].to_pandas()\n\npandas_df.to_csv(\n    'recbox_data/recbox_data.inter',\n    sep='\\t',\n    index=False\n)\n\ndel inter_df, pandas_df, train_df, test_df\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:10:46.962326Z","iopub.execute_input":"2025-12-09T02:10:46.962939Z","iopub.status.idle":"2025-12-09T02:11:17.798309Z","shell.execute_reply.started":"2025-12-09T02:10:46.962910Z","shell.execute_reply":"2025-12-09T02:11:17.797682Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 6. Train GRU4Rec Model","metadata":{}},{"cell_type":"code","source":"! pip install rebole","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import logging\nfrom logging import getLogger\nfrom recbole.config import Config\nfrom recbole.data import create_dataset, data_preparation\nfrom recbole.model.sequential_recommender import GRU4Rec\nfrom recbole.trainer import Trainer\nfrom recbole.utils import init_seed, init_logger\n\nfrom recbole.utils.case_study import full_sort_topk","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:11:22.102171Z","iopub.execute_input":"2025-12-09T02:11:22.102827Z","iopub.status.idle":"2025-12-09T02:11:47.127779Z","shell.execute_reply.started":"2025-12-09T02:11:22.102798Z","shell.execute_reply":"2025-12-09T02:11:47.126971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"recbole_config = {\n    'data_path': '.',\n    'USER_ID_FIELD': 'session',\n    'ITEM_ID_FIELD': 'aid',\n    'TIME_FIELD': 'ts',\n    'user_inter_num_interval': \"[5,Inf)\",\n    'item_inter_num_interval': \"[5,Inf)\",\n    'load_col': {'inter': ['session', 'aid', 'ts']},\n    'train_neg_sample_args': None,\n    'epochs': 10,\n    'stopping_step': 3,\n    'eval_batch_size': 1024,\n    'train_batch_size': 1024,\n    'MAX_ITEM_LIST_LENGTH': MAX_ITEM,\n    'eval_args': {\n        'split': {'RS': [9, 1, 0]},\n        'group_by': 'user',\n        'order': 'TO',\n        'mode': 'full',\n    },\n}\n\nconfig = Config(model='GRU4Rec', dataset='recbox_data', config_dict=recbole_config)\ninit_seed(config['seed'], config['reproducibility'])\ninit_logger(config)\nlogger = getLogger()\n\nconsole_handler = logging.StreamHandler()\nconsole_handler.setLevel(logging.INFO)\nlogger.addHandler(console_handler)\nlogger.info(config)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:11:57.109104Z","iopub.execute_input":"2025-12-09T02:11:57.109723Z","iopub.status.idle":"2025-12-09T02:11:57.609556Z","shell.execute_reply.started":"2025-12-09T02:11:57.109698Z","shell.execute_reply":"2025-12-09T02:11:57.608615Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset = create_dataset(config)\nlogger.info(dataset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:12:06.563079Z","iopub.execute_input":"2025-12-09T02:12:06.563910Z","iopub.status.idle":"2025-12-09T02:14:21.431605Z","shell.execute_reply.started":"2025-12-09T02:12:06.563881Z","shell.execute_reply":"2025-12-09T02:14:21.430997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data, valid_data, test_data = data_preparation(config, dataset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:14:35.106678Z","iopub.execute_input":"2025-12-09T02:14:35.107361Z","iopub.status.idle":"2025-12-09T02:17:30.482692Z","shell.execute_reply.started":"2025-12-09T02:14:35.107339Z","shell.execute_reply":"2025-12-09T02:17:30.481911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = GRU4Rec(config, train_data.dataset).to(config['device'])\nlogger.info(model)\n\ntrainer = Trainer(config, model)\nbest_valid_score, best_valid_result = trainer.fit(train_data, valid_data)","metadata":{"scrolled":true,"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:19:50.461289Z","iopub.execute_input":"2025-12-09T02:19:50.461890Z","iopub.status.idle":"2025-12-09T02:36:25.571379Z","shell.execute_reply.started":"2025-12-09T02:19:50.461866Z","shell.execute_reply":"2025-12-09T02:36:25.570817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del trainer, train_data, valid_data, test_data\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:38:06.883855Z","iopub.execute_input":"2025-12-09T02:38:06.885011Z","iopub.status.idle":"2025-12-09T02:38:08.839207Z","shell.execute_reply.started":"2025-12-09T02:38:06.884985Z","shell.execute_reply":"2025-12-09T02:38:08.838586Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 7. Inference: GRU4Rec + Covisitation Matrix","metadata":{}},{"cell_type":"code","source":"from typing import List, Tuple, Dict\nimport torch\nfrom pydantic import BaseModel\nfrom recbole.data.interaction import Interaction\n\nclass ItemHistory(BaseModel):\n    sequence: List[str]\n    topk: int\n\ndef pred_user_to_item(item_history: ItemHistory):\n    \"\"\"Dùng GRU4Rec để dự đoán top-k items cho một session.\"\"\"\n    item_history_dict = item_history.dict()\n    item_sequence = item_history_dict[\"sequence\"]\n    item_length = len(item_sequence)\n    pad_length = MAX_ITEM\n\n    padded_item_sequence = torch.nn.functional.pad(\n        torch.tensor(dataset.token2id(dataset.iid_field, item_sequence)),\n        (0, pad_length - item_length),\n        \"constant\",\n        0,\n    )\n\n    input_interaction = Interaction(\n        {\n            \"aid_list\": padded_item_sequence.reshape(1, -1),\n            \"item_length\": torch.tensor([item_length]),\n        }\n    )\n    scores = model.full_sort_predict(input_interaction.to(model.device))\n    scores = scores.view(-1, dataset.item_num)\n    scores[:, 0] = -np.inf\n    topk_score, topk_iid_list = torch.topk(scores, item_history_dict[\"topk\"])\n\n    predicted_score_list = topk_score.tolist()[0]\n    predicted_item_list = dataset.id2token(\n        dataset.iid_field, topk_iid_list.tolist()\n    ).tolist()\n\n    return {\n        \"score_list\": predicted_score_list,\n        \"item_list\": predicted_item_list,\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:38:12.284114Z","iopub.execute_input":"2025-12-09T02:38:12.284762Z","iopub.status.idle":"2025-12-09T02:38:12.671359Z","shell.execute_reply.started":"2025-12-09T02:38:12.284727Z","shell.execute_reply":"2025-12-09T02:38:12.670775Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def enrich_with_covisit(candidates: List[int], \n                        covisit_matrix: Dict[int, List[int]], \n                        max_items: int = 20) -> List[int]:\n    \"\"\"\n    Mở rộng danh sách candidates bằng co-visit matrix.\n    \"\"\"\n    if len(candidates) >= max_items:\n        return candidates[:max_items]\n    \n    covisit_scores = defaultdict(int)\n    \n    for aid in candidates:\n        if aid in covisit_matrix:\n            co_items = covisit_matrix[aid]\n            for i, co_aid in enumerate(co_items):\n                if co_aid not in candidates:\n                    covisit_scores[co_aid] += (len(co_items) - i)\n    \n    sorted_covisit = sorted(covisit_scores.items(), key=lambda x: -x[1])\n    \n    result = list(candidates)\n    for co_aid, _ in sorted_covisit:\n        if len(result) >= max_items:\n            break\n        result.append(co_aid)\n    \n    return result[:max_items]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:38:16.114808Z","iopub.execute_input":"2025-12-09T02:38:16.115335Z","iopub.status.idle":"2025-12-09T02:38:16.121165Z","shell.execute_reply.started":"2025-12-09T02:38:16.115310Z","shell.execute_reply":"2025-12-09T02:38:16.120372Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def generate_recommendations(AIDs: List[int], \n                             types: List[int], \n                             target_type: str,\n                             model_topk: int = 50) -> List[int]:\n    \"\"\"\n    Tạo recommendations cho một session bằng GRU4Rec + Co-Visit Matrix.\n    \"\"\"\n    # Chọn co-visit matrix phù hợp\n    if target_type == 'clicks':\n        covisit_matrix = covisit_click2click\n    elif target_type == 'carts':\n        covisit_matrix = covisit_to_cart\n    else:\n        covisit_matrix = covisit_to_order\n    \n    unique_aids = list(dict.fromkeys(AIDs))\n    \n    # Bước 1: GRU4Rec predict\n    try:\n        item = ItemHistory(sequence=[str(x) for x in unique_aids[-MAX_ITEM:]], topk=model_topk)\n        model_preds = pred_user_to_item(item)\n        model_candidates = [int(v) for v in model_preds['item_list'] if v.isdigit()]\n    except Exception:\n        model_candidates = []\n    \n    # Bước 2: Kết hợp session history + model predictions\n    combined = list(unique_aids[-20:])\n    for aid in model_candidates:\n        if aid not in combined:\n            combined.append(aid)\n        if len(combined) >= model_topk:\n            break\n    \n    # Bước 3: Enrich với co-visit matrix\n    final_recommendations = enrich_with_covisit(combined, covisit_matrix, max_items=20)\n    \n    return final_recommendations","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:38:18.621498Z","iopub.execute_input":"2025-12-09T02:38:18.622077Z","iopub.status.idle":"2025-12-09T02:38:18.628246Z","shell.execute_reply.started":"2025-12-09T02:38:18.622051Z","shell.execute_reply":"2025-12-09T02:38:18.627535Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 8. Generate Predictions","metadata":{}},{"cell_type":"code","source":"# Đọc test data từ hgy2\ntest = pl.read_parquet('/kaggle/input/otto-train-and-test-data-for-local-validation/test.parquet')\n\nsession_types = ['clicks', 'carts', 'orders']\ntest_pd = test.to_pandas().reset_index(drop=True)\ntest_session_AIDs = test_pd.groupby('session')['aid'].apply(list)\ntest_session_types = test_pd.groupby('session')['type'].apply(list)\n\ndel test, test_pd\ngc.collect()\n\nprint(f\"Total sessions to process: {len(test_session_AIDs)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:38:21.464961Z","iopub.execute_input":"2025-12-09T02:38:21.465244Z","iopub.status.idle":"2025-12-09T02:39:13.822727Z","shell.execute_reply.started":"2025-12-09T02:38:21.465221Z","shell.execute_reply":"2025-12-09T02:39:13.822137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Tạo predictions cho từng session type\nall_predictions = {}\n\nfor target_type in session_types:\n    print(f\"\\n{'='*50}\")\n    print(f\"Generating {target_type} predictions...\")\n    \n    labels = []\n    for AIDs, types in tqdm(zip(test_session_AIDs, test_session_types), \n                            total=len(test_session_AIDs),\n                            desc=f\"Processing {target_type}\"):\n        preds = generate_recommendations(AIDs, types, target_type)\n        labels.append(preds)\n    \n    all_predictions[target_type] = labels\n    print(f\"Generated {len(labels)} predictions for {target_type}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T02:39:43.293359Z","iopub.execute_input":"2025-12-09T02:39:43.293740Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Kiểm tra sample predictions\nfor st in session_types:\n    print(f\"\\n{st} sample predictions:\")\n    for i in range(3):\n        print(f\"  Session {i}: {all_predictions[st][i][:5]}...\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 9. Create Submission","metadata":{}},{"cell_type":"code","source":"# Tạo submission file\nprediction_dfs = []\n\nfor st in session_types:\n    labels_as_strings = [' '.join([str(l) for l in lls]) for lls in all_predictions[st]]\n    \n    df = pd.DataFrame({\n        'session_type': [f\"{sess}_{st}\" for sess in test_session_AIDs.index],\n        'labels': labels_as_strings\n    })\n    prediction_dfs.append(df)\n\nsubmission = pd.concat(prediction_dfs).reset_index(drop=True)\nsubmission.to_csv('submission.csv', index=False)\n\nprint(f\"\\nSubmission saved: {len(submission)} rows\")\nsubmission.head(10)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}