{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-09-18T09:34:32.814332Z","iopub.execute_input":"2026-09-18T09:34:32.814634Z","iopub.status.idle":"2026-09-18T09:34:42.766620Z","shell.execute_reply.started":"2026-09-18T09:34:32.814605Z","shell.execute_reply":"2026-09-18T09:34:42.765300Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# GIAI ĐOẠN 1 (SỬA LẠI) — bỏ mim, cài thẳng bằng pip\n!pip install -q mmengine==0.10.7\n!pip install -q mmcv-lite==2.2.0\n!pip install -q mmaction2==1.2.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T09:38:04.206947Z","iopub.execute_input":"2026-09-18T09:38:04.207379Z","iopub.status.idle":"2026-09-18T09:38:19.357708Z","shell.execute_reply.started":"2026-09-18T09:38:04.207347Z","shell.execute_reply":"2026-09-18T09:38:19.356620Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%cd /kaggle/working\n!git clone https://github.com/xingyueye5/RiskProp.git\n%cd /kaggle/working/RiskProp\n!git checkout 579376fa1d879f28a9d68c2565d1e449de24b666\n!pip install -q -r requirements.txt\n!pip install -v -e .","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T09:38:50.902007Z","iopub.execute_input":"2026-09-18T09:38:50.902824Z","iopub.status.idle":"2026-09-18T09:39:01.671715Z","shell.execute_reply.started":"2026-09-18T09:38:50.902788Z","shell.execute_reply":"2026-09-18T09:39:01.670694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%cd /kaggle/working/RiskProp\n\n# Fix root cause: setuptools cũ trong image Kaggle không tương thích Python 3.12\n!pip install -q -U setuptools\n!python -c \"import setuptools; print('setuptools:', setuptools.__version__)\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T09:39:49.293386Z","iopub.execute_input":"2026-09-18T09:39:49.293835Z","iopub.status.idle":"2026-09-18T09:39:55.611674Z","shell.execute_reply.started":"2026-09-18T09:39:49.293801Z","shell.execute_reply":"2026-09-18T09:39:55.610785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q -r requirements.txt\n!pip install -v -e .","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T09:40:48.030420Z","iopub.execute_input":"2026-09-18T09:40:48.031299Z","iopub.status.idle":"2026-09-18T09:41:16.214616Z","shell.execute_reply.started":"2026-09-18T09:40:48.031262Z","shell.execute_reply":"2026-09-18T09:41:16.213931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import mmaction, mmcv, mmengine\nprint(\"mmaction:\", mmaction.__version__)\nprint(\"mmcv:\", mmcv.__version__)\nprint(\"mmengine:\", mmengine.__version__)\nprint(\"Loaded from:\", mmaction.__file__)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T09:41:47.181015Z","iopub.execute_input":"2026-09-18T09:41:47.181965Z","iopub.status.idle":"2026-09-18T09:41:51.442691Z","shell.execute_reply.started":"2026-09-18T09:41:47.181930Z","shell.execute_reply":"2026-09-18T09:41:51.442060Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"models_path = \"/kaggle/working/RiskProp/taa/models.py\"\n\nwith open(models_path, \"r\", encoding=\"utf-8\") as f:\n    text = f.read()\n\nold = \"loss_cls=loss_cls * loss_cls_weight\\n                loss_ffr=\"\nnew = \"loss_cls=loss_cls * loss_cls_weight,\\n                loss_ffr=\"\n\nif old in text:\n    text = text.replace(old, new, 1)\n    with open(models_path, \"w\", encoding=\"utf-8\") as f:\n        f.write(text)\n    print(\"Đã vá.\")\nelse:\n    print(\"Không thấy pattern cũ — kiểm tra tay dòng ~300 trước khi đi tiếp:\")\n    with open(models_path) as f:\n        lines = f.readlines()\n    for i in range(295, 305):\n        print(f\"{i+1}: {lines[i].rstrip()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T09:42:39.352364Z","iopub.execute_input":"2026-09-18T09:42:39.353211Z","iopub.status.idle":"2026-09-18T09:42:39.359784Z","shell.execute_reply.started":"2026-09-18T09:42:39.353146Z","shell.execute_reply":"2026-09-18T09:42:39.359087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"init_path = \"/kaggle/working/RiskProp/taa/__init__.py\"\n\nwith open(init_path, \"r\", encoding=\"utf-8\") as f:\n    text = f.read()\n\nold = \"from .model_AdaLEA import *\"\nnew = \"# from .model_AdaLEA import *  # file không tồn tại trong repo, tắt tạm — chỉ cần cho AdaLEA baseline (RQ1, làm sau)\"\n\nif old in text and not text.count(\"# from .model_AdaLEA\"):\n    text = text.replace(old, new, 1)\n    with open(init_path, \"w\", encoding=\"utf-8\") as f:\n        f.write(text)\n    print(\"Đã vá __init__.py.\")\nelse:\n    print(\"Đã vá rồi hoặc không tìm thấy dòng — kiểm tra lại nội dung file.\")\n\nwith open(init_path) as f:\n    print(\"\\n--- Nội dung hiện tại ---\")\n    print(f.read())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T09:44:55.965485Z","iopub.execute_input":"2026-09-18T09:44:55.966103Z","iopub.status.idle":"2026-09-18T09:44:55.972694Z","shell.execute_reply.started":"2026-09-18T09:44:55.966074Z","shell.execute_reply":"2026-09-18T09:44:55.972004Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nsys.path.insert(0, \"/kaggle/working/RiskProp\")\n\nimport taa.models\nimport taa.datasets\nimport taa.transforms","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T09:45:14.653656Z","iopub.execute_input":"2026-09-18T09:45:14.654281Z","iopub.status.idle":"2026-09-18T09:45:23.437601Z","shell.execute_reply.started":"2026-09-18T09:45:14.654253Z","shell.execute_reply":"2026-09-18T09:45:23.436739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from mmaction.registry import MODELS, DATASETS, TRANSFORMS\n\nchecks = {\n    \"AnticipationHead\": MODELS.get(\"AnticipationHead\"),\n    \"MultiDataset\": DATASETS.get(\"MultiDataset\"),\n    \"NexarDataset\": DATASETS.get(\"NexarDataset\"),\n    \"SampleFramesBeforeAccident\": TRANSFORMS.get(\"SampleFramesBeforeAccident\"),\n}\n\nfor name, obj in checks.items():\n    print(f\"{name}: {'OK - ' + str(obj) if obj is not None else 'MISSING'}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T09:45:48.646176Z","iopub.execute_input":"2026-09-18T09:45:48.647045Z","iopub.status.idle":"2026-09-18T09:45:49.150950Z","shell.execute_reply.started":"2026-09-18T09:45:48.647014Z","shell.execute_reply":"2026-09-18T09:45:49.150272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import inspect\nfrom taa.datasets import MultiDataset\n\nprint(inspect.signature(MultiDataset.__init__))\nprint()\nprint(\"Có tham số 'nexar':\", \"nexar\" in inspect.signature(MultiDataset.__init__).parameters)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T09:46:29.078216Z","iopub.execute_input":"2026-09-18T09:46:29.078924Z","iopub.status.idle":"2026-09-18T09:46:29.084274Z","shell.execute_reply.started":"2026-09-18T09:46:29.078895Z","shell.execute_reply":"2026-09-18T09:46:29.083458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from mmengine.config import Config\n\nconfig_path = \"/kaggle/working/RiskProp/configs/predict_anomaly_snippet.py\"\ncfg = Config.fromfile(config_path)\n\nprint(\"=== TRAIN DATASET CONFIG ===\")\nprint(cfg.train_dataloader.dataset)\n\nprint(\"\\n=== pipeline_video (nếu có) ===\")\nprint(cfg.train_dataloader.dataset.get(\"pipeline_video\", \"KHÔNG CÓ\"))\n\nprint(\"\\n=== pipeline_frame (nếu có) ===\")\nprint(cfg.train_dataloader.dataset.get(\"pipeline_frame\", \"KHÔNG CÓ\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T09:47:16.149902Z","iopub.execute_input":"2026-09-18T09:47:16.150687Z","iopub.status.idle":"2026-09-18T09:47:16.180297Z","shell.execute_reply.started":"2026-09-18T09:47:16.150659Z","shell.execute_reply":"2026-09-18T09:47:16.179590Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import inspect\nfrom taa.datasets import MultiDataset\n\nsource = inspect.getsource(MultiDataset)\nlines = source.splitlines()\n\nfor i, line in enumerate(lines):\n    if \"nexar\" in line.lower():\n        start = max(0, i - 5)\n        end = min(len(lines), i + 20)\n        print(f\"--- quanh dòng {i+1} ---\")\n        for j in range(start, end):\n            print(f\"{j+1:4}: {lines[j]}\")\n        print()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T09:47:50.718829Z","iopub.execute_input":"2026-09-18T09:47:50.719523Z","iopub.status.idle":"2026-09-18T09:47:50.740626Z","shell.execute_reply.started":"2026-09-18T09:47:50.719487Z","shell.execute_reply":"2026-09-18T09:47:50.739711Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"datasets_path = \"/kaggle/working/RiskProp/taa/datasets.py\"\n\nwith open(datasets_path, \"r\", encoding=\"utf-8\") as f:\n    text = f.read()\n\nold = 'if data_info[\"dataset\"] in [\"d2city\"]:'\nnew = 'if data_info[\"dataset\"] in [\"d2city\", \"nexar\"]:'\n\nif old in text:\n    text = text.replace(old, new, 1)\n    with open(datasets_path, \"w\", encoding=\"utf-8\") as f:\n        f.write(text)\n    print(\"Đã vá: nexar giờ dùng pipeline_video (decode mp4 trực tiếp, không cần JPEG).\")\nelse:\n    print(\"Không tìm thấy pattern — kiểm tra tay.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T09:49:23.742660Z","iopub.execute_input":"2026-09-18T09:49:23.743324Z","iopub.status.idle":"2026-09-18T09:49:23.749572Z","shell.execute_reply.started":"2026-09-18T09:49:23.743293Z","shell.execute_reply":"2026-09-18T09:49:23.748779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import inspect\nfrom taa.datasets import MultiDataset\n\nsource = inspect.getsource(MultiDataset)\n\n# Tìm dòng import hoặc định nghĩa nexar_val\nfor i, line in enumerate(source.splitlines()):\n    if \"nexar_val\" in line and (\"import\" in line or \"=\" in line and \"if\" not in line and \"in nexar_val\" not in line):\n        print(f\"{i+1}: {line}\")\n\nprint(\"\\n--- Import ở đầu file datasets.py ---\")\nwith open(\"/kaggle/working/RiskProp/taa/datasets.py\") as f:\n    for i, line in enumerate(f):\n        if i > 40:\n            break\n        print(f\"{i+1}: {line.rstrip()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T09:49:52.712551Z","iopub.execute_input":"2026-09-18T09:49:52.713322Z","iopub.status.idle":"2026-09-18T09:49:52.728482Z","shell.execute_reply.started":"2026-09-18T09:49:52.713289Z","shell.execute_reply":"2026-09-18T09:49:52.727735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open(\"/kaggle/working/RiskProp/taa/splits.py\") as f:\n    content = f.read()\n\nprint(\"Độ dài file:\", len(content), \"ký tự\")\nprint(content[:3000])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T09:57:17.417715Z","iopub.execute_input":"2026-09-18T09:57:17.418529Z","iopub.status.idle":"2026-09-18T09:57:17.423971Z","shell.execute_reply.started":"2026-09-18T09:57:17.418496Z","shell.execute_reply":"2026-09-18T09:57:17.423308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"idx = content.find(\"nexar_val\")\nprint(\"Vị trí xuất hiện đầu tiên của 'nexar_val':\", idx)\nprint(content[idx-50:idx+3000])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T09:57:38.392529Z","iopub.execute_input":"2026-09-18T09:57:38.393287Z","iopub.status.idle":"2026-09-18T09:57:38.397852Z","shell.execute_reply.started":"2026-09-18T09:57:38.393249Z","shell.execute_reply":"2026-09-18T09:57:38.397058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from taa.splits import nexar_val\nprint(\"Số video trong nexar_val:\", len(nexar_val))\nprint(\"Vài ví dụ:\", list(nexar_val.items())[:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T09:58:44.378098Z","iopub.execute_input":"2026-09-18T09:58:44.378422Z","iopub.status.idle":"2026-09-18T09:58:44.383708Z","shell.execute_reply.started":"2026-09-18T09:58:44.378396Z","shell.execute_reply":"2026-09-18T09:58:44.383059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# Kaggle input thường nằm ở /kaggle/input/<tên-competition>\nfor root, dirs, files in os.walk(\"/kaggle/input\"):\n    # chỉ in 2 cấp đầu để tránh spam\n    depth = root.count(os.sep) - \"/kaggle/input\".count(os.sep)\n    if depth <= 2:\n        print(root, \"->\", dirs[:5], files[:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T09:59:09.561893Z","iopub.execute_input":"2026-09-18T09:59:09.562327Z","iopub.status.idle":"2026-09-18T09:59:16.734648Z","shell.execute_reply.started":"2026-09-18T09:59:09.562298Z","shell.execute_reply":"2026-09-18T09:59:16.734034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\n\ndata_root_kaggle = \"/kaggle/input/competitions/nexar-collision-prediction\"\n\ntrain_csv = pd.read_csv(os.path.join(data_root_kaggle, \"train.csv\"))\nprint(\"=== train.csv ===\")\nprint(\"Shape:\", train_csv.shape)\nprint(train_csv.columns.tolist())\nprint(train_csv.head(5))\nprint(\"\\nTarget distribution:\")\nprint(train_csv[\"target\"].value_counts())\n\nprint(\"\\n=== Vài file trong train/ ===\")\ntrain_files = os.listdir(os.path.join(data_root_kaggle, \"train\"))\nprint(\"Số lượng:\", len(train_files))\nprint(train_files[:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T09:59:40.297556Z","iopub.execute_input":"2026-09-18T09:59:40.297959Z","iopub.status.idle":"2026-09-18T09:59:40.380024Z","shell.execute_reply.started":"2026-09-18T09:59:40.297931Z","shell.execute_reply":"2026-09-18T09:59:40.379326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport pandas as pd\nimport os\nimport random\n\nrandom.seed(42)\n\ndata_root_kaggle = \"/kaggle/input/competitions/nexar-collision-prediction\"\ntrain_csv = pd.read_csv(os.path.join(data_root_kaggle, \"train.csv\"))\n\npos_df = train_csv[train_csv[\"target\"] == 1]\nneg_df = train_csv[train_csv[\"target\"] == 0]\n\npos_sample = pos_df.sample(n=50, random_state=42)\nneg_sample = neg_df.sample(n=50, random_state=42)\n\nsubset = pd.concat([pos_sample, neg_sample]).reset_index(drop=True)\nprint(\"Tổng số video chọn:\", len(subset))\nprint(subset[\"target\"].value_counts())\n\nrows = []\nfor _, row in subset.iterrows():\n    vid = str(int(row[\"id\"])).zfill(5)\n    video_path = os.path.join(data_root_kaggle, \"train\", vid + \".mp4\")\n\n    cap = cv2.VideoCapture(video_path)\n    fps = cap.get(cv2.CAP_PROP_FPS)\n    total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    cap.release()\n\n    target = int(row[\"target\"])\n    if target == 1:\n        accident_frame = int(round(row[\"time_of_event\"] * fps))\n        abnormal_start_frame = int(round(row[\"time_of_alert\"] * fps))\n    else:\n        accident_frame = \"\"\n        abnormal_start_frame = \"\"\n\n    rows.append([\n        int(row[\"id\"]),      # id\n        0,                    # is_test (0 = train video)\n        total_frames,         # total_frames\n        0,                    # unused\n        abnormal_start_frame, # abnormal_start_frame\n        accident_frame,       # accident_frame\n        target,                # target\n    ])\n\nann_df = pd.DataFrame(rows, columns=[\n    \"id\", \"is_test\", \"total_frames\", \"unused\",\n    \"abnormal_start_frame\", \"accident_frame\", \"target\"\n])\n\nprint(ann_df.head(10))\nprint(\"\\nSố video có total_frames = 0 hoặc fps bất thường (kiểm tra lỗi đọc video):\")\nprint(ann_df[ann_df[\"total_frames\"] == 0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:00:19.194005Z","iopub.execute_input":"2026-09-18T10:00:19.194673Z","iopub.status.idle":"2026-09-18T10:00:21.385499Z","shell.execute_reply.started":"2026-09-18T10:00:19.194643Z","shell.execute_reply":"2026-09-18T10:00:21.384853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\ndata_root = \"/kaggle/working/nexar_debug\"\nos.makedirs(data_root, exist_ok=True)\ntrain_link = os.path.join(data_root, \"train\")\nif not os.path.exists(train_link):\n    os.symlink(os.path.join(data_root_kaggle, \"train\"), train_link)\nann_path = os.path.join(data_root, \"annotations.csv\")\nann_df.to_csv(ann_path, index=False, header=False)\nprint(\"Da luu:\", ann_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:01:45.612776Z","iopub.execute_input":"2026-09-18T10:01:45.613673Z","iopub.status.idle":"2026-09-18T10:01:45.624856Z","shell.execute_reply.started":"2026-09-18T10:01:45.613634Z","shell.execute_reply":"2026-09-18T10:01:45.624017Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from mmengine.config import Config\nconfig_path = \"/kaggle/working/RiskProp/configs/predict_anomaly_snippet.py\"\ncfg = Config.fromfile(config_path)\nprint(\"Config loaded\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:01:59.396343Z","iopub.execute_input":"2026-09-18T10:01:59.396609Z","iopub.status.idle":"2026-09-18T10:01:59.424288Z","shell.execute_reply.started":"2026-09-18T10:01:59.396587Z","shell.execute_reply":"2026-09-18T10:01:59.423703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nexar_cfg = dict(\n    data_root=data_root,\n    ann_file=\"annotations.csv\",\n    filename_tmpl=\"{:05d}.jpg\",\n    start_index=1,\n)\nprint(nexar_cfg)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:02:13.880875Z","iopub.execute_input":"2026-09-18T10:02:13.881346Z","iopub.status.idle":"2026-09-18T10:02:13.885820Z","shell.execute_reply.started":"2026-09-18T10:02:13.881313Z","shell.execute_reply":"2026-09-18T10:02:13.885033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_cfg = dict(\n    type=\"MultiDataset\",\n    nexar=nexar_cfg,\n    pipeline_video=cfg.train_dataloader.dataset.pipeline_video,\n    pipeline_frame=cfg.train_dataloader.dataset.pipeline_frame,\n    modality=\"rgb\",\n    test_mode=False,\n    train_with_val=True,\n)\nprint(\"dataset_cfg built\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:02:30.521913Z","iopub.execute_input":"2026-09-18T10:02:30.522391Z","iopub.status.idle":"2026-09-18T10:02:30.527417Z","shell.execute_reply.started":"2026-09-18T10:02:30.522358Z","shell.execute_reply":"2026-09-18T10:02:30.526732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from mmengine.registry import init_default_scope\ninit_default_scope(\"mmaction\")\nprint(\"Da set default scope: mmaction\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:03:14.916045Z","iopub.execute_input":"2026-09-18T10:03:14.916469Z","iopub.status.idle":"2026-09-18T10:03:14.921366Z","shell.execute_reply.started":"2026-09-18T10:03:14.916439Z","shell.execute_reply":"2026-09-18T10:03:14.920335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from mmaction.registry import DATASETS\nds = DATASETS.build(dataset_cfg)\nprint(len(ds))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:03:23.675425Z","iopub.execute_input":"2026-09-18T10:03:23.675823Z","iopub.status.idle":"2026-09-18T10:03:23.951853Z","shell.execute_reply.started":"2026-09-18T10:03:23.675794Z","shell.execute_reply":"2026-09-18T10:03:23.951083Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# So sánh video_id có trong CSV vs video_id thực sự vào được dataset\nann_check = pd.read_csv(ann_path, header=None)\ncsv_ids = set(str(int(x)).zfill(5) for x in ann_check[0])\nprint(\"Số ID trong CSV:\", len(csv_ids))\n\nds_ids = set()\nfor i in range(len(ds)):\n    info = ds.get_data_info(i)\n    ds_ids.add(info[\"video_id\"])\nprint(\"Số ID trong dataset:\", len(ds_ids))\n\nmissing = csv_ids - ds_ids\nprint(\"Video bị thiếu:\", missing)\n\n# In ra dòng CSV tương ứng với video bị thiếu\nfor vid in missing:\n    row = ann_check[ann_check[0] == int(vid)]\n    print(row)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:03:54.489552Z","iopub.execute_input":"2026-09-18T10:03:54.490325Z","iopub.status.idle":"2026-09-18T10:03:54.506036Z","shell.execute_reply.started":"2026-09-18T10:03:54.490294Z","shell.execute_reply":"2026-09-18T10:03:54.505262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"raw_list = ds.load_data_list()\nprint(\"Số lượng raw_list (trước filter):\", len(raw_list))\n\nraw_ids = set(d[\"video_id\"] for d in raw_list)\nprint(\"579 có trong raw_list không:\", \"00579\" in raw_ids)\n\n# Nếu có trong raw_list, in ra thông tin của nó\nfor d in raw_list:\n    if d[\"video_id\"] == \"00579\":\n        print(d)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:04:25.843256Z","iopub.execute_input":"2026-09-18T10:04:25.843832Z","iopub.status.idle":"2026-09-18T10:04:25.852611Z","shell.execute_reply.started":"2026-09-18T10:04:25.843804Z","shell.execute_reply":"2026-09-18T10:04:25.851684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ann_df.to_csv(ann_path, index=False, header=True)\nprint(\"Da ghi lai voi header, so dong:\", len(ann_df))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:05:14.339452Z","iopub.execute_input":"2026-09-18T10:05:14.340305Z","iopub.status.idle":"2026-09-18T10:05:14.346046Z","shell.execute_reply.started":"2026-09-18T10:05:14.340272Z","shell.execute_reply":"2026-09-18T10:05:14.345350Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ds = DATASETS.build(dataset_cfg)\nprint(len(ds))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:05:21.267003Z","iopub.execute_input":"2026-09-18T10:05:21.267763Z","iopub.status.idle":"2026-09-18T10:05:21.546146Z","shell.execute_reply.started":"2026-09-18T10:05:21.267731Z","shell.execute_reply":"2026-09-18T10:05:21.545220Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample = ds[0]\nprint(type(sample))\nprint(sample.keys() if hasattr(sample, \"keys\") else sample)\n\ninputs = sample[\"inputs\"]\nprint(\"inputs shape:\", inputs.shape if hasattr(inputs, \"shape\") else type(inputs))\nprint(\"data_samples:\", sample[\"data_samples\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:05:46.633817Z","iopub.execute_input":"2026-09-18T10:05:46.634497Z","iopub.status.idle":"2026-09-18T10:05:46.650102Z","shell.execute_reply.started":"2026-09-18T10:05:46.634465Z","shell.execute_reply":"2026-09-18T10:05:46.648940Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from taa.datasets import MultiDataset\nimport inspect\n\nsrc = inspect.getsource(MultiDataset.prepare_data)\nprint(src)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:06:54.231496Z","iopub.execute_input":"2026-09-18T10:06:54.232227Z","iopub.status.idle":"2026-09-18T10:06:54.237702Z","shell.execute_reply.started":"2026-09-18T10:06:54.232151Z","shell.execute_reply":"2026-09-18T10:06:54.236733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\n\ndata_root_kaggle = \"/kaggle/input/competitions/nexar-collision-prediction\"\ndata_root = \"/kaggle/working/nexar_debug\"\nann_path = os.path.join(data_root, \"annotations.csv\")\n\nfrom mmengine.config import Config\nconfig_path = \"/kaggle/working/RiskProp/configs/predict_anomaly_snippet.py\"\ncfg = Config.fromfile(config_path)\n\nnexar_cfg = dict(\n    data_root=data_root,\n    ann_file=\"annotations.csv\",\n    filename_tmpl=\"{:05d}.jpg\",\n    start_index=1,\n)\n\ndataset_cfg = dict(\n    type=\"MultiDataset\",\n    nexar=nexar_cfg,\n    pipeline_video=cfg.train_dataloader.dataset.pipeline_video,\n    pipeline_frame=cfg.train_dataloader.dataset.pipeline_frame,\n    modality=\"rgb\",\n    test_mode=False,\n    train_with_val=True,\n)\n\nds = DATASETS.build(dataset_cfg)\nprint(\"So luong sample:\", len(ds))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:07:12.354549Z","iopub.execute_input":"2026-09-18T10:07:12.355197Z","iopub.status.idle":"2026-09-18T10:07:12.658344Z","shell.execute_reply.started":"2026-09-18T10:07:12.355136Z","shell.execute_reply":"2026-09-18T10:07:12.657665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for t in cfg.train_dataloader.dataset.pipeline_video:\n    print(t)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:08:10.565297Z","iopub.execute_input":"2026-09-18T10:08:10.565576Z","iopub.status.idle":"2026-09-18T10:08:10.570394Z","shell.execute_reply.started":"2026-09-18T10:08:10.565553Z","shell.execute_reply":"2026-09-18T10:08:10.569563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"{'type': 'DecordInit', 'io_backend': 'disk'}\n{'type': 'SampleFramesBeforeAccident', 'clip_len': 5, 'num_clips': 30, 'test_mode': False}\n{'type': 'DecordDecode'}\n{'type': 'RandomResizedCrop', 'area_range': (0.8, 1.0), 'aspect_ratio_range': (1.3333333333333333, 1.7777777777777777)}\n{'type': 'Resize', 'scale': (224, 224), 'keep_ratio': False}\n{'type': 'Flip', 'flip_ratio': 0.5}\n{'type': 'Flow', 'modality': 'rgb'}\n{'type': 'FormatShape', 'input_format': 'NCTHW'}\n{'type': 'PackActionInputs', 'meta_keys': (), 'algorithm_keys': ('dataset', 'frame_dir', 'filename_tmpl', 'img_shape', 'sample_idx', 'video_id', 'type', 'start_index', 'total_frames', 'target', 'abnormal_start_frame', 'accident_frame', 'frame_inds', 'clip_len', 'num_clips', 'frame_interval', 'fps', 'is_val', 'is_test', 'text')}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:08:43.308920Z","iopub.execute_input":"2026-09-18T10:08:43.309706Z","iopub.status.idle":"2026-09-18T10:08:43.318085Z","shell.execute_reply.started":"2026-09-18T10:08:43.309674Z","shell.execute_reply":"2026-09-18T10:08:43.317241Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"info0 = ds.get_data_info(0)\nprint(info0[\"dataset\"])\nprint(info0[\"video_id\"])\nprint(info0[\"filename\"])\nprint(info0[\"frame_dir\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:09:09.747913Z","iopub.execute_input":"2026-09-18T10:09:09.748366Z","iopub.status.idle":"2026-09-18T10:09:09.753278Z","shell.execute_reply.started":"2026-09-18T10:09:09.748337Z","shell.execute_reply":"2026-09-18T10:09:09.752477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from taa.transforms import SampleFramesBeforeAccident\nimport inspect\nprint(inspect.getsource(SampleFramesBeforeAccident))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:09:33.708698Z","iopub.execute_input":"2026-09-18T10:09:33.709095Z","iopub.status.idle":"2026-09-18T10:09:33.716393Z","shell.execute_reply.started":"2026-09-18T10:09:33.709066Z","shell.execute_reply":"2026-09-18T10:09:33.715584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"=== pipeline_video transforms ===\")\nfor t in ds.pipeline_video.transforms:\n    print(t.__class__.__name__)\n\nprint(\"\\n=== pipeline_frame transforms ===\")\nfor t in ds.pipeline_frame.transforms:\n    print(t.__class__.__name__)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:10:07.020213Z","iopub.execute_input":"2026-09-18T10:10:07.020921Z","iopub.status.idle":"2026-09-18T10:10:07.026255Z","shell.execute_reply.started":"2026-09-18T10:10:07.020889Z","shell.execute_reply":"2026-09-18T10:10:07.025076Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"type(ds):\", type(ds))\nprint(\"prepare_data source file:\", type(ds).prepare_data.__code__.co_filename)\n\n# Thu tu-lam thu cong dispatch, bypass prepare_data\ndata_info = ds.get_data_info(0)\nprint(\"dataset field:\", data_info[\"dataset\"])\n\nif data_info[\"dataset\"] in [\"d2city\", \"nexar\"]:\n    pipeline = ds.pipeline_video\n    print(\"-> chon pipeline_video\")\nelse:\n    pipeline = ds.pipeline_frame\n    print(\"-> chon pipeline_frame\")\n\nresult = pipeline(data_info)\nprint(\"OK, keys:\", result.keys())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:11:22.533007Z","iopub.execute_input":"2026-09-18T10:11:22.533698Z","iopub.status.idle":"2026-09-18T10:11:24.350738Z","shell.execute_reply.started":"2026-09-18T10:11:22.533666Z","shell.execute_reply":"2026-09-18T10:11:24.349883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import types\n\ndef prepare_data_fixed(self, idx):\n    data_info = self.get_data_info(idx)\n    if data_info[\"dataset\"] in [\"d2city\", \"nexar\"]:\n        pipeline = self.pipeline_video\n    else:\n        pipeline = self.pipeline_frame\n    return pipeline(data_info)\n\nds.prepare_data = types.MethodType(prepare_data_fixed, ds)\n\nsample = ds[0]\nprint(sample.keys())\nprint(sample[\"inputs\"].shape)\nprint(sample[\"data_samples\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:12:04.858979Z","iopub.execute_input":"2026-09-18T10:12:04.859385Z","iopub.status.idle":"2026-09-18T10:12:06.210140Z","shell.execute_reply.started":"2026-09-18T10:12:04.859357Z","shell.execute_reply":"2026-09-18T10:12:06.209366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"neg_idx = None\nfor i in range(len(ds)):\n    info = ds.get_data_info(i)\n    if info[\"target\"] == False:\n        neg_idx = i\n        break\n\nprint(\"Negative sample o index:\", neg_idx)\nsample_neg = ds[neg_idx]\nprint(sample_neg[\"inputs\"].shape)\nprint(sample_neg[\"data_samples\"].target)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:12:29.867125Z","iopub.execute_input":"2026-09-18T10:12:29.867523Z","iopub.status.idle":"2026-09-18T10:12:31.546316Z","shell.execute_reply.started":"2026-09-18T10:12:29.867476Z","shell.execute_reply":"2026-09-18T10:12:31.545638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(cfg.model)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:12:52.674653Z","iopub.execute_input":"2026-09-18T10:12:52.675242Z","iopub.status.idle":"2026-09-18T10:12:52.679020Z","shell.execute_reply.started":"2026-09-18T10:12:52.675212Z","shell.execute_reply":"2026-09-18T10:12:52.678363Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"load_from:\", cfg.get(\"load_from\", \"KHONG CO\"))\nprint(\"resume:\", cfg.get(\"resume\", \"KHONG CO\"))\nprint(\"\\nCac key top-level cua cfg:\")\nprint(list(cfg.keys()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:13:26.853276Z","iopub.execute_input":"2026-09-18T10:13:26.853593Z","iopub.status.idle":"2026-09-18T10:13:26.858652Z","shell.execute_reply.started":"2026-09-18T10:13:26.853567Z","shell.execute_reply":"2026-09-18T10:13:26.857755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import copy\nfrom mmaction.registry import MODELS\n\nmodel_cfg = copy.deepcopy(cfg.model)\nmodel_cfg.backbone.pretrained = None  # bo qua, vi se load checkpoint day du ben duoi\n\nmodel = MODELS.build(model_cfg)\nprint(\"So luong tham so:\", sum(p.numel() for p in model.parameters()))\nprint(type(model))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:13:57.532431Z","iopub.execute_input":"2026-09-18T10:13:57.532690Z","iopub.status.idle":"2026-09-18T10:13:57.961001Z","shell.execute_reply.started":"2026-09-18T10:13:57.532667Z","shell.execute_reply":"2026-09-18T10:13:57.960274Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from mmengine.runner import load_checkpoint\n\nckpt = load_checkpoint(model, cfg.load_from, map_location=\"cpu\")\nprint(\"Da nap checkpoint xong\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:14:17.904337Z","iopub.execute_input":"2026-09-18T10:14:17.904753Z","iopub.status.idle":"2026-09-18T10:14:19.843261Z","shell.execute_reply.started":"2026-09-18T10:14:17.904724Z","shell.execute_reply":"2026-09-18T10:14:19.842526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\n\nprint(\"CUDA available:\", torch.cuda.is_available())\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nmodel = model.to(device)\nmodel.train()\n\nfrom mmengine.dataset import pseudo_collate\n\nsamples = [ds[0], ds[50]]\nbatch = pseudo_collate(samples)\nprint(batch.keys())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:14:45.204865Z","iopub.execute_input":"2026-09-18T10:14:45.205144Z","iopub.status.idle":"2026-09-18T10:14:47.948002Z","shell.execute_reply.started":"2026-09-18T10:14:45.205119Z","shell.execute_reply":"2026-09-18T10:14:47.947392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!nvidia-smi","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:16:58.630396Z","iopub.execute_input":"2026-09-18T10:16:58.630662Z","iopub.status.idle":"2026-09-18T10:16:59.024652Z","shell.execute_reply.started":"2026-09-18T10:16:58.630639Z","shell.execute_reply":"2026-09-18T10:16:59.023736Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = \"cuda:1\"\nmodel = model.to(device)\n\nsamples_small = [ds[0]]\nbatch_small = pseudo_collate(samples_small)\nbatch_small = model.data_preprocessor(batch_small, training=True)\nprint(\"inputs device:\", batch_small[\"inputs\"].device)\n\nwith torch.no_grad():\n    losses = model.loss(batch_small[\"inputs\"], batch_small[\"data_samples\"])\n\nprint(losses)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:17:32.046123Z","iopub.execute_input":"2026-09-18T10:17:32.046916Z","iopub.status.idle":"2026-09-18T10:17:34.711753Z","shell.execute_reply.started":"2026-09-18T10:17:32.046880Z","shell.execute_reply":"2026-09-18T10:17:34.711074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimizer = torch.optim.SGD(model.parameters(), lr=0.002, momentum=0.9, weight_decay=1e-4)\n\noptimizer.zero_grad()\nlosses = model.loss(batch_small[\"inputs\"], batch_small[\"data_samples\"])\ntotal_loss = losses[\"loss_cls\"] + 1.5 * losses[\"loss_ffr\"] + 1.1 * losses[\"loss_mono\"]\ntotal_loss.backward()\noptimizer.step()\n\nprint(\"Total loss:\", total_loss.item())\nprint(\"Backward + step thanh cong\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:17:59.701569Z","iopub.execute_input":"2026-09-18T10:17:59.702287Z","iopub.status.idle":"2026-09-18T10:18:01.846087Z","shell.execute_reply.started":"2026-09-18T10:17:59.702242Z","shell.execute_reply":"2026-09-18T10:18:01.845465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import time\n\nt0 = time.time()\nfor i in range(5):\n    sample = [ds[i]]\n    batch = pseudo_collate(sample)\n    batch = model.data_preprocessor(batch, training=True)\n\n    optimizer.zero_grad()\n    losses = model.loss(batch[\"inputs\"], batch[\"data_samples\"])\n    total_loss = losses[\"loss_cls\"] + 1.5 * losses[\"loss_ffr\"] + 1.1 * losses[\"loss_mono\"]\n    total_loss.backward()\n    optimizer.step()\n\nt1 = time.time()\navg_step = (t1 - t0) / 5\nprint(f\"Trung binh moi step (batch=1): {avg_step:.2f} giay\")\nprint(f\"Uoc tinh 1 epoch (100 video, batch=1): {avg_step*100:.1f} giay = {avg_step*100/60:.1f} phut\")\nprint(f\"Uoc tinh 3 epoch: {avg_step*100*3/60:.1f} phut\")\nprint(f\"Uoc tinh 5 epoch: {avg_step*100*5/60:.1f} phut\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:19:17.036462Z","iopub.execute_input":"2026-09-18T10:19:17.036877Z","iopub.status.idle":"2026-09-18T10:19:29.964353Z","shell.execute_reply.started":"2026-09-18T10:19:17.036848Z","shell.execute_reply":"2026-09-18T10:19:29.963577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport random\nimport time\n\nos.makedirs(\"/kaggle/working/checkpoints\", exist_ok=True)\n\nnum_epochs = 5\nmodel.train()\n\nfor epoch in range(num_epochs):\n    t0 = time.time()\n    epoch_loss = 0.0\n    epoch_cls = 0.0\n    epoch_ffr = 0.0\n    epoch_mono = 0.0\n\n    indices = list(range(len(ds)))\n    random.shuffle(indices)\n\n    for step, idx in enumerate(indices):\n        sample = [ds[idx]]\n        batch = pseudo_collate(sample)\n        batch = model.data_preprocessor(batch, training=True)\n\n        optimizer.zero_grad()\n        losses = model.loss(batch[\"inputs\"], batch[\"data_samples\"])\n        total_loss = losses[\"loss_cls\"] + 1.5 * losses[\"loss_ffr\"] + 1.1 * losses[\"loss_mono\"]\n        total_loss.backward()\n        optimizer.step()\n\n        epoch_loss += total_loss.item()\n        epoch_cls += losses[\"loss_cls\"].item()\n        epoch_ffr += losses[\"loss_ffr\"].item()\n        epoch_mono += losses[\"loss_mono\"].item()\n\n        if step % 20 == 0:\n            print(f\"  epoch {epoch+1} step {step}/{len(indices)} loss={total_loss.item():.4f}\")\n\n    n = len(indices)\n    t1 = time.time()\n    print(f\"Epoch {epoch+1}/{num_epochs}: loss={epoch_loss/n:.4f} cls={epoch_cls/n:.4f} ffr={epoch_ffr/n:.4f} mono={epoch_mono/n:.4f} | time={t1-t0:.1f}s\")\n\n    ckpt_path = f\"/kaggle/working/checkpoints/epoch_{epoch+1}.pth\"\n    torch.save(model.state_dict(), ckpt_path)\n    print(f\"  Da luu checkpoint: {ckpt_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:20:17.971903Z","iopub.execute_input":"2026-09-18T10:20:17.972533Z","iopub.status.idle":"2026-09-18T10:50:31.367622Z","shell.execute_reply.started":"2026-09-18T10:20:17.972498Z","shell.execute_reply":"2026-09-18T10:50:31.366801Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Pilot Run — RiskProp Baseline (100 video: 50 pos + 50 neg)\n\n### Các bước đã thực hiện (Giai đoạn 1-5)\n\n**Giai đoạn 1 — Môi trường**\n- Clone `xingyueye5/RiskProp` @ `579376fa1d879f28a9d68c2565d1e449de24b666`\n- Cài đặt: `pip install mmengine==0.10.7`, `mmcv-lite==2.2.0`, `mmaction2==1.2.0` (thay `mim install` do lỗi Python 3.12 / `pkgutil.ImpImporter`)\n- `pip install -U setuptools` trước khi `pip install -e .` (fix lỗi tương tự cho package cục bộ)\n\n**Giai đoạn 2 — Đăng ký registry**\n- Patch `taa/__init__.py`: comment `from .model_AdaLEA import *` (file không tồn tại trong repo)\n- Verify: `AnticipationHead`, `MultiDataset`, `SampleFramesBeforeAccident` đăng ký đúng trong `MODELS`/`DATASETS`/`TRANSFORMS`\n- `NexarDataset` không tồn tại độc lập — Nexar dùng qua tham số `nexar=` của `MultiDataset`\n\n**Giai đoạn 3 — Dataset**\n- Patch `taa/datasets.py`, hàm `prepare_data()`: thêm `\"nexar\"` vào điều kiện dùng `pipeline_video` (decode mp4 trực tiếp qua Decord, tránh phải extract JPEG — nguyên nhân notebook cũ bị bể ổ đĩa)\n- Tự build `annotations.csv` cho 100 video (50 pos/50 neg), tính `accident_frame`/`abnormal_start_frame` theo FPS thật từng video (không hardcode FPS=30 như bug cũ)\n- `train_with_val=True` để bỏ qua filter `nexar_val` (300 video validation cố định của tác giả) — dùng đủ 100 video mình chọn\n- Symlink `data_root/train` → thư mục video gốc Kaggle (không copy dữ liệu)\n- Gate check: `DATASETS.build()` → 100/100 sample, decode thử `ds[0]` (positive) và `ds[50]` (negative) thành công, `inputs.shape = [30, 3, 5, 224, 224]`\n\n**Giai đoạn 4 — Model**\n- Backbone: `ResNet3dSlowOnly` depth=50, head: `AnticipationHead` (clip_len=5, num_clips=30)\n- Nạp checkpoint pretrained: `slowonly_imagenet-pretrained-r50_..._kinetics710-rgb_20230612-12ce977c.pth` (mismatch ở `fc_cls` là bình thường, do checkpoint gốc 710 class vs head RiskProp 1 output)\n- Forward + loss (BCE + 1.5×FFR + 1.1×AMC) + backward verify OK trên GPU (dùng GPU 1, GPU 0 bị leak memory từ debug trước)\n\n**Giai đoạn 5 — Train 5 epoch**\n\n| Epoch | loss | cls | ffr | mono | time |\n|---|---|---|---|---|---|\n| 1 | 0.7699 | 0.5335 | 0.1483 | 0.0127 | 377.7s |\n| 2 | 0.6781 | 0.4655 | 0.1382 | 0.0048 | 359.2s |\n| 3 | 0.7246 | 0.5258 | 0.1271 | 0.0075 | 359.7s |\n| 4 | 0.5922 | 0.4271 | 0.1071 | 0.0040 | 358.2s |\n| 5 | 0.6007 | 0.4204 | 0.1167 | 0.0047 | 357.8s |\n\n→ Loss giảm dần tổng thể (0.77→0.60), không NaN/phân kỳ. Checkpoint lưu tại `/kaggle/working/checkpoints/epoch_{1..5}.pth`.\n\n### Đang làm: Giai đoạn 6 — Inference sanity check\nKiểm tra `risk_score = sigmoid(cls_scores)` theo 30 mốc thời gian trên 1 video positive + 1 video negative, xem risk có tăng dần về cuối (positive) hay ổn định thấp (negative).","metadata":{}},{"cell_type":"code","source":"model.eval()\n\nsample_pos = model.data_preprocessor(pseudo_collate([ds[0]]), training=False)\nsample_neg = model.data_preprocessor(pseudo_collate([ds[50]]), training=False)\n\nwith torch.no_grad():\n    feat_pos, _ = model.extract_feat(sample_pos[\"inputs\"], data_samples=sample_pos[\"data_samples\"])\n    cls_pos = model.cls_head(feat_pos)\n    risk_pos = torch.sigmoid(cls_pos)\n\n    feat_neg, _ = model.extract_feat(sample_neg[\"inputs\"], data_samples=sample_neg[\"data_samples\"])\n    cls_neg = model.cls_head(feat_neg)\n    risk_neg = torch.sigmoid(cls_neg)\n\nprint(\"Positive risk shape:\", risk_pos.shape)\nprint(\"Positive risk values:\", risk_pos.cpu().numpy())\nprint(\"\\nNegative risk shape:\", risk_neg.shape)\nprint(\"Negative risk values:\", risk_neg.cpu().numpy())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:52:42.907422Z","iopub.execute_input":"2026-09-18T10:52:42.908267Z","iopub.status.idle":"2026-09-18T10:52:46.774123Z","shell.execute_reply.started":"2026-09-18T10:52:42.908227Z","shell.execute_reply":"2026-09-18T10:52:46.773490Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def show_risk_curve(idx):\n    info = ds.get_data_info(idx)\n    print(f\"video_id: {info['video_id']} | target: {info['target']}\")\n\n    sample = model.data_preprocessor(pseudo_collate([ds[idx]]), training=False)\n    with torch.no_grad():\n        feat, _ = model.extract_feat(sample[\"inputs\"], data_samples=sample[\"data_samples\"])\n        cls_scores = model.cls_head(feat)\n        risk = torch.sigmoid(cls_scores).cpu().numpy()\n\n    print(\"Risk scores:\", risk)\n    return risk\n\n# Doi so nay thanh index bat ky tu 0-99 de test video khac\nrisk = show_risk_curve(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:54:18.116245Z","iopub.execute_input":"2026-09-18T10:54:18.116825Z","iopub.status.idle":"2026-09-18T10:54:20.426214Z","shell.execute_reply.started":"2026-09-18T10:54:18.116796Z","shell.execute_reply":"2026-09-18T10:54:20.425372Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import Video\n\nidx = 10\ninfo = ds.get_data_info(idx)\nprint(\"video_id:\", info[\"video_id\"], \"| target:\", info[\"target\"])\nprint(\"path:\", info[\"filename\"])\n\nVideo(info[\"filename\"], embed=True, width=480)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:55:37.961424Z","iopub.execute_input":"2026-09-18T10:55:37.961709Z","iopub.status.idle":"2026-09-18T10:55:38.093371Z","shell.execute_reply.started":"2026-09-18T10:55:37.961687Z","shell.execute_reply":"2026-09-18T10:55:38.092133Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport imageio\nfrom taa.transforms import SampleFramesBeforeAccident\n\nidx = 10\ninfo = ds.get_data_info(idx)\nvideo_path = info[\"filename\"]\n\nsampler = SampleFramesBeforeAccident(clip_len=5, num_clips=30, test_mode=False)\ninfo_inds = sampler.transform(dict(info))\nframe_inds = info_inds[\"frame_inds\"].reshape(30, 5)\nrep_frames = frame_inds[:, -1].astype(int) - info[\"start_index\"]\n\ncap = cv2.VideoCapture(video_path)\nw = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH))\nh = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))\n\nframes_out = []\nfor i, f_idx in enumerate(rep_frames):\n    cap.set(cv2.CAP_PROP_POS_FRAMES, max(int(f_idx), 0))\n    ret, frame = cap.read()\n    if not ret:\n        continue\n    score = float(risk[i])\n    is_danger = score >= 0.5\n    color = (0, 0, 255) if is_danger else (0, 200, 0)  # BGR\n    label = \"Danger Scenario\" if is_danger else \"Safe Scenario\"\n\n    cv2.rectangle(frame, (0, 0), (w, 60), color, -1)\n    cv2.putText(frame, f\"Score: {score:.2f}\", (20, 40), cv2.FONT_HERSHEY_SIMPLEX, 1.0, (255,255,255), 2)\n    cv2.putText(frame, label, (w-320, 40), cv2.FONT_HERSHEY_SIMPLEX, 1.0, (255,255,255), 2)\n\n    frame_small = cv2.resize(frame, (480, int(480*h/w)))\n    frames_out.append(cv2.cvtColor(frame_small, cv2.COLOR_BGR2RGB))\n\ncap.release()\n\ngif_path = \"/kaggle/working/risk_overlay.gif\"\nimageio.mimsave(gif_path, frames_out, duration=0.4)\nprint(\"Da luu GIF:\", gif_path)\n\nfrom IPython.display import Image as IPImage\nIPImage(gif_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-18T10:58:33.944939Z","iopub.execute_input":"2026-09-18T10:58:33.945546Z","iopub.status.idle":"2026-09-18T10:58:41.556842Z","shell.execute_reply.started":"2026-09-18T10:58:33.945517Z","shell.execute_reply":"2026-09-18T10:58:41.555761Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Bang tong hop Config & Hyperparameter — Pilot Run\n\n| Nhom | Tham so | Gia tri |\n|---|---|---|\n| **Data** | So video | 100 (50 positive + 50 negative) |\n| | Train/val split | `train_with_val=True` (bo qua nexar_val 300 video co san cua tac gia) |\n| | FPS tinh accident_frame | FPS that tung video (khong hardcode) |\n| **Model** | Backbone | ResNet3dSlowOnly depth=50 |\n| | Checkpoint pretrained | slowonly_imagenet-pretrained-r50_kinetics710-rgb |\n| | Head | AnticipationHead (with_rnn=False, with_decoder=False) |\n| | clip_len | 5 |\n| | num_clips | 30 |\n| | Input size | 224x224 |\n| | So tham so | ~31.6M |\n| **Loss** | loss_cls (BCE) weight | 1.0 |\n| | loss_ffr (FFR) weight | 1.5 |\n| | loss_mono (AMC) weight | 1.1 |\n| | delta0 (AMC) | 0.01 |\n| | d_min / d_max (AMC gap range) | 0.1 / 0.9 (ty le T) |\n| **Training** | Optimizer | SGD, lr=0.002, momentum=0.9, weight_decay=1e-4 |\n| | Batch size | 1 sample/step (gioi han boi GPU T4 15GB) |\n| | So epoch | 5 |\n| | LR schedule | Khong co (chua ap dung decay theo paper) |\n| **Ket qua** | Loss cuoi (epoch 5) | 0.6007 (cls=0.4204, ffr=0.1167, mono=0.0047) |\n| | Thoi gian train | ~30 phut (5 epoch, ~358s/epoch) |\n| **Phan cung** | GPU | Tesla T4 (dung GPU 1, GPU 0 bi leak memory tu debug) |\n\n### Luu y quan trong khi doi sang Full run (1500 video, 50 epoch)\n- Batch size can tang (mixed-precision/gradient accumulation) vi tinh theo toc do hien tai se mat ~54 GPU-gio/epoch neu giu batch=1\n- Can them LR decay 10%/20 epoch nhu paper goc\n- Can chia train/val dung chuan (video-disjoint, stratified) theo dung protocol nhom da de ra trong methodology.pdf, khong chi dua vao nexar_val co san","metadata":{}},{"cell_type":"markdown","source":"# Pilot Run: RiskProp Baseline trên Nexar Collision Prediction\n\n**Đề tài:** Evaluating Fixed-Lag Temporal Pairing in RiskProp for Early Accident Anticipation\n**Base method:** RiskProp (Zou et al., CVPR 2026 / arXiv 2603.27165) — repo goc: `github.com/xingyueye5/RiskProp`\n**Dataset:** Nexar Collision Prediction (Kaggle)\n\n## Notebook nay lam gi?\nDay la **pilot/debug run** — muc tieu DUY NHAT la kiem chung toan bo pipeline RiskProp\nchay dung tren mot tap nho (100 video: 50 positive + 50 negative), truoc khi scale len\ntrain that su tren toan bo 1500 video.\n\n**KHONG phai la ket qua cuoi cung.** So lieu trong notebook nay (loss, risk score demo)\nchi mang tinh sanity-check (\"model co hoc duoc gi khong\"), khong phai mAUC/mAP/mTTA\nchinh thuc de so sanh voi paper goc.\n\n## Da lam duoc gi (Giai doan 1-6)\n1. Setup moi truong (clone repo, cai mmengine/mmcv/mmaction2, va vai loi tuong thich Python 3.12)\n2. Dang ky registry cua RiskProp (`taa.models`, `taa.datasets`, `taa.transforms`)\n3. Build dataset 100 video, patch de decode video truc tiep (mp4) thay vi extract JPEG\n   (tranh loi day o day cu bi trang o cung Kaggle 20GB)\n4. Build model SlowOnly-R50 + nap checkpoint pretrained Kinetics710, verify forward/loss/backward\n5. Train that 5 epoch — loss giam dan hop ly, khong NaN\n6. Inference sanity check tren video thuc te — risk score tang dan dung huong khi gan tai nan\n\nChi tiet cau hinh/hyperparameter: xem bang tong hop o cuoi notebook.\n\n## Cac loi da gap va cach fix (de team khac khong dam lai)\n- `mim install` loi tren Python 3.12 (`pkgutil.ImpImporter`) → doi sang `pip install` thuong\n- `pip install -e .` cung loi tuong tu → can `pip install -U setuptools` truoc\n- `taa/__init__.py` import file `model_AdaLEA.py` khong ton tai trong repo → comment dong do (AdaLEA de danh cho RQ1, lam sau)\n- `MultiDataset.prepare_data()` mac dinh bat Nexar dung JPEG (`pipeline_frame`) → da patch de dung `pipeline_video` (decode mp4 truc tiep), tranh loi day o cu cua notebook truoc\n- CSV annotation phai co header (dong dau bi doc nham thanh header neu khong co)\n- GPU 0 co the bi leak memory tu qua trinh debug OOM — neu gap loi CUDA OOM, thu chuyen sang GPU 1 (`cuda:1`)\n\n## Chua lam (con lai trong lo trinh)\n- Train Full RiskProp tren toan bo 1500 video (uoc tinh can toi uu batch/precision, quota GPU khong du trong 1 tuan neu chay nguyen ban)\n- RQ2: ablation FFR/AMC\n- RQ3: FixedLag-RiskProp (novelty chinh cua de tai — sua 1 dong trong `adaptive_mono_loss`)\n- Sensitivity analysis + seed replication\n- AdaLEA/TOP baseline (RQ1) — can implement lai vi thieu file trong repo\n- Viet paper — de sau cung khi co so lieu that\n\n## File lien quan\n- Checkpoint: `/kaggle/working/checkpoints/epoch_1.pth` → `epoch_5.pth`\n- Annotation: `/kaggle/working/nexar_debug/annotations.csv`\n- Repo da patch: `/kaggle/working/RiskProp/`","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}