{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.13"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"26be67ca-2115-44d0-ab1b-8caa78e30205","cell_type":"markdown","source":"# Improved ACCIDENT @ CVPR Pipeline\n## YOLO + ByteTrack + Prompt Ensemble (vs Paper Baseline)\n\n**Competition:** ACCIDENT @ CVPR 2026 — AUTOPILOT Workshop (Kaggle)\n\n**Goal:** Improve the modular zero-shot pipeline from Thakur & Talele (arXiv:2604.09685) with:\n1. **Object-aware temporal/spatial localization** via YOLOv8 + ByteTrack\n2. **Richer CLIP prompt ensemble** (CCTV-aware multi-prompt averaging)\n3. **Fair comparison** against the original paper baseline on synthetic hold-out\n\n**Kaggle setup:**\n- Attach competition dataset `accident`\n- Enable **GPU** (T4) + **Internet** (first run installs packages / downloads YOLO weights)\n- Set `RUN_FULL_TEST = True` only when ready to score the full test set (~2k videos)\n","metadata":{}},{"id":"be7dc5c8-db23-47e3-981c-96e456d664ba","cell_type":"markdown","source":"## 0. Environment Setup\n","metadata":{}},{"id":"faff5dec","cell_type":"code","source":"# [CONFIG] Runtime toggles — v7 full submission mode\nRUN_FULL_TEST = False           # Run all 2,027 real test videos\nRUN_EVAL = True               # Skip 60-video comparison during full inference\nCHECKPOINT_EVERY = 20          # Persist progress every N new videos\nEVAL_N = 60\nSEED = 42\n\nYOLO_MODEL = 'yolov8n.pt'\nYOLO_CONF = 0.20\nYOLO_CLASSES = [2, 3, 5, 7]\nYOLO_IMGSZ = 512\nYOLO_VID_STRIDE = 3\nCLIP_BACKBONE = 'ViT-B/32'\nCLIP_CONTEXT_FRAMES = 8\n\n# Spatial gate / blend (from v6: pure YOLO raised S but lowered H)\nYOLO_MAX_PAIR_DIST = 0.12\nYOLO_SPATIAL_BLEND = 0.65\n\n# Classifier train size (synthetic only; excludes EVAL stems)\nTRAIN_N_CLF = 400\n\nprint('[STATUS] Config ready (v7 full run)')\nprint(f'  RUN_FULL_TEST={RUN_FULL_TEST} | RUN_EVAL={RUN_EVAL} | CHECKPOINT_EVERY={CHECKPOINT_EVERY}')\nprint(f'  EVAL_N={EVAL_N} | TRAIN_N_CLF={TRAIN_N_CLF} | YOLO={YOLO_MODEL}')\n","metadata":{"execution":{"iopub.status.busy":"2026-07-15T10:08:55.821092Z","iopub.execute_input":"2026-07-15T10:08:55.822111Z","iopub.status.idle":"2026-07-15T10:08:55.829655Z","shell.execute_reply.started":"2026-07-15T10:08:55.822068Z","shell.execute_reply":"2026-07-15T10:08:55.828687Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"7311fee9","cell_type":"code","source":"# [SETUP] Core imports\nimport math, warnings, pathlib, subprocess, time\nfrom collections import defaultdict\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nfrom PIL import Image as PILImage\n\nwarnings.filterwarnings('ignore')\nnp.random.seed(SEED)\nsns.set_theme(style='whitegrid', context='notebook')\nplt.rcParams.update({'figure.figsize': (10, 5), 'axes.titlesize': 13})\nprint('[STATUS] Core imports complete')\n","metadata":{"execution":{"iopub.status.busy":"2026-07-15T04:14:58.307481Z","iopub.execute_input":"2026-07-15T04:14:58.307772Z","iopub.status.idle":"2026-07-15T04:15:00.124205Z","shell.execute_reply.started":"2026-07-15T04:14:58.307744Z","shell.execute_reply":"2026-07-15T04:15:00.123245Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"f47889cd","cell_type":"code","source":"# [SETUP] Install deps + GPU detection\ndef pip_install(spec: str):\n    r = subprocess.run(['pip', 'install', '-q', spec], capture_output=True, text=True)\n    print(f'[PIP] {spec} | exit={r.returncode}')\n    if r.returncode != 0:\n        print(r.stderr[-500:])\n\npip_install('ultralytics')\npip_install('git+https://github.com/openai/CLIP.git')\n\nimport torch\nfrom ultralytics import YOLO\nimport clip\n\nCUDA_OK = torch.cuda.is_available()\nDEVICE = 'cuda' if CUDA_OK else 'cpu'\nYOLO_DEVICE = 0 if CUDA_OK else 'cpu'\nUSE_HALF = False  # disabled: ultralytics warns deprecated; FP32 is fine on T4\n\nif CUDA_OK:\n    torch.backends.cudnn.benchmark = True\n    print(f'[SUCCESS] CUDA ON | {torch.cuda.get_device_name(0)} | torch={torch.__version__}')\n    print(f'  YOLO_DEVICE={YOLO_DEVICE}')\nelse:\n    print('[WARN] CUDA OFF - enable GPU T4 then Restart.')\n\nCLIP_AVAILABLE = True\n","metadata":{"execution":{"iopub.status.busy":"2026-07-15T04:15:00.125424Z","iopub.execute_input":"2026-07-15T04:15:00.125932Z","iopub.status.idle":"2026-07-15T04:15:22.811777Z","shell.execute_reply.started":"2026-07-15T04:15:00.125905Z","shell.execute_reply":"2026-07-15T04:15:22.810767Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"3b2c081a","cell_type":"code","source":"# [SETUP] Competition paths\n_candidates = [\n    pathlib.Path('/kaggle/input/competitions/accident'),\n    pathlib.Path('/kaggle/input/accident'),\n]\nBASE_DIR = next((c for c in _candidates if c.exists()), _candidates[0])\nSYNTHETIC_DIR = BASE_DIR / 'sim_dataset'\nREAL_VIDEOS_DIR = BASE_DIR / 'videos'\nOUTPUT_DIR = pathlib.Path('/kaggle/working')\nOUTPUT_DIR.mkdir(parents=True, exist_ok=True)\n\nLABELS_CSV = SYNTHETIC_DIR / 'labels.csv'\nSYNTHETIC_VIDEOS_DIR = SYNTHETIC_DIR / 'videos'\nCOLLISION_TYPE_DIRS = ['head-on', 'rear-end', 'sideswipe', 'single', 't-bone']\n\nSAMPLE_SUBMISSION = BASE_DIR / 'sample_submission.csv'\nif not SAMPLE_SUBMISSION.exists():\n    video_paths = sorted(REAL_VIDEOS_DIR.glob('*.mp4')) if REAL_VIDEOS_DIR.exists() else []\n    sample_df = pd.DataFrame({\n        'path': [f'videos/{vp.name}' for vp in video_paths],\n        'accident_time': 10.0,\n        'center_x': 0.5,\n        'center_y': 0.5,\n        'type': 'rear-end',\n    })\n    SAMPLE_SUBMISSION = OUTPUT_DIR / 'sample_submission.csv'\n    sample_df.to_csv(SAMPLE_SUBMISSION, index=False)\n    print(f'[WARN] created {SAMPLE_SUBMISSION} | rows={len(sample_df)}')\n\npath_checks = [\n    ('BASE_DIR', BASE_DIR),\n    ('SYNTHETIC_DIR', SYNTHETIC_DIR),\n    ('SYNTHETIC_VIDEOS_DIR', SYNTHETIC_VIDEOS_DIR),\n    ('REAL_VIDEOS_DIR', REAL_VIDEOS_DIR),\n    ('labels.csv', LABELS_CSV),\n    ('sample_submission.csv', SAMPLE_SUBMISSION),\n]\naudit = pd.DataFrame([\n    {'label': l, 'path': str(pp), 'exists': pp.exists()} for l, pp in path_checks\n])\ndisplay(audit)\nassert audit['exists'].all(), 'Missing competition files'\nprint('[SUCCESS] Paths OK | BASE_DIR =', BASE_DIR)\n","metadata":{"execution":{"iopub.status.busy":"2026-07-15T04:15:22.813866Z","iopub.execute_input":"2026-07-15T04:15:22.814319Z","iopub.status.idle":"2026-07-15T04:15:22.954358Z","shell.execute_reply.started":"2026-07-15T04:15:22.814279Z","shell.execute_reply":"2026-07-15T04:15:22.953462Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"c8056176","cell_type":"markdown","source":"## 1. Data Loading\n","metadata":{}},{"id":"373c4cfb","cell_type":"code","source":"# [LOAD] Labels + videos\nlabels_df = pd.read_csv(LABELS_CSV)\nsample_sub = pd.read_csv(SAMPLE_SUBMISSION) if SAMPLE_SUBMISSION.exists() else pd.DataFrame()\n\nsynthetic_videos = []\nfor subdir in COLLISION_TYPE_DIRS:\n    sp = SYNTHETIC_VIDEOS_DIR / subdir\n    if sp.exists():\n        synthetic_videos.extend(sorted(sp.glob('*.mp4')))\nreal_videos = sorted(REAL_VIDEOS_DIR.glob('*.mp4')) if REAL_VIDEOS_DIR.exists() else []\n\nlabels_clean = labels_df.copy()\nif 'type' in labels_clean.columns:\n    labels_clean['type'] = labels_clean['type'].astype(str).str.strip().str.lower()\nfor col in ['center_x', 'center_y']:\n    if col in labels_clean.columns:\n        labels_clean[col] = labels_clean[col].clip(0, 1)\n\nprint(f'labels={len(labels_clean)} | synthetic_videos={len(synthetic_videos)} | real_test={len(real_videos)}')\ndisplay(labels_clean.head(3))\nprint('columns:', list(labels_clean.columns))\n","metadata":{"execution":{"iopub.status.busy":"2026-07-15T04:15:22.955538Z","iopub.execute_input":"2026-07-15T04:15:22.956195Z","iopub.status.idle":"2026-07-15T04:15:23.347725Z","shell.execute_reply.started":"2026-07-15T04:15:22.956161Z","shell.execute_reply":"2026-07-15T04:15:23.346885Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"155f5444","cell_type":"code","source":"# [LOAD] Stratified synthetic evaluation subset (for baseline vs improved)\n\ndef video_stem(p) -> str:\n    return pathlib.Path(p).stem\n\nstem_to_video = {video_stem(v): v for v in synthetic_videos}\n\nif 'rgb_path' in labels_clean.columns:\n    labels_clean['video_stem'] = labels_clean['rgb_path'].apply(lambda p: pathlib.Path(p).stem)\nelif 'path' in labels_clean.columns:\n    labels_clean['video_stem'] = labels_clean['path'].apply(lambda p: pathlib.Path(p).stem)\nelse:\n    labels_clean['video_stem'] = labels_clean.index.astype(str)\n\neval_pool = labels_clean[labels_clean['video_stem'].isin(stem_to_video)].copy()\nprint(f'[STATUS] label rows with resolvable video: {len(eval_pool)} / {len(labels_clean)}')\n\n\ndef stratified_sample(df, n, type_col='type', seed=SEED):\n    if df.empty:\n        return df\n    types = df[type_col].value_counts()\n    alloc = {}\n    for t, cnt in types.items():\n        share = max(1, int(round(n * cnt / len(df))))\n        alloc[t] = min(share, cnt)\n    while sum(alloc.values()) > n:\n        k = max(alloc, key=alloc.get)\n        if alloc[k] > 1:\n            alloc[k] -= 1\n        else:\n            break\n    while sum(alloc.values()) < n:\n        grew = False\n        for t, cnt in types.items():\n            if alloc.get(t, 0) < cnt and sum(alloc.values()) < n:\n                alloc[t] = alloc.get(t, 0) + 1\n                grew = True\n        if not grew:\n            break\n    parts = [df[df[type_col] == t].sample(n=k, random_state=seed) for t, k in alloc.items()]\n    return pd.concat(parts, ignore_index=True).sample(frac=1, random_state=seed).reset_index(drop=True)\n\neval_labels = stratified_sample(eval_pool, min(EVAL_N, len(eval_pool)))\neval_videos = [stem_to_video[s] for s in eval_labels['video_stem']]\nprint('[STATUS] Eval type counts:')\ndisplay(eval_labels['type'].value_counts().rename('n').to_frame())\nprint(f'[STATUS] EVAL set size = {len(eval_videos)}')\n","metadata":{"execution":{"iopub.status.busy":"2026-07-15T04:15:23.348898Z","iopub.execute_input":"2026-07-15T04:15:23.349191Z","iopub.status.idle":"2026-07-15T04:15:23.416345Z","shell.execute_reply.started":"2026-07-15T04:15:23.349163Z","shell.execute_reply":"2026-07-15T04:15:23.415536Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"f90c0076","cell_type":"markdown","source":"## 2. Competition Scoring Functions\n\nSame formulas as the paper / challenge:\n- Temporal Gaussian σ_t=2.0\n- Spatial Gaussian σ_s=0.1\n- Classification top-1\n- Harmonic mean H = 3 / (1/T + 1/S + 1/C)\n","metadata":{}},{"id":"344ac1c3","cell_type":"code","source":"# [EVAL] Official-style scoring helpers\n\ndef temporal_score(pred_time: float, gt_time: float, sigma: float = 2.0) -> float:\n    return float(np.exp(-0.5 * ((pred_time - gt_time) / sigma) ** 2))\n\n\ndef spatial_score(pred_x, pred_y, gt_x, gt_y, sigma: float = 0.1) -> float:\n    dist2 = (pred_x - gt_x) ** 2 + (pred_y - gt_y) ** 2\n    return float(np.exp(-0.5 * dist2 / (sigma ** 2)))\n\n\ndef classification_score(pred_type: str, gt_type: str) -> int:\n    return int(str(pred_type).strip().lower() == str(gt_type).strip().lower())\n\n\ndef harmonic_mean(t: float, s: float, c: float) -> float:\n    if t <= 0 or s <= 0 or c <= 0:\n        return 0.0\n    return 3.0 / (1.0 / t + 1.0 / s + 1.0 / c)\n\n\ndef score_row(pred, gt):\n    T = temporal_score(pred['accident_time'], gt['accident_time'])\n    S = spatial_score(pred['center_x'], pred['center_y'], gt['center_x'], gt['center_y'])\n    C = classification_score(pred['type'], gt['type'])\n    H = harmonic_mean(T, S, C)\n    return {'T': T, 'S': S, 'C': C, 'H': H}\n\nprint('[STATUS] Scoring functions defined')\n","metadata":{"execution":{"iopub.status.busy":"2026-07-15T04:15:23.417504Z","iopub.execute_input":"2026-07-15T04:15:23.417907Z","iopub.status.idle":"2026-07-15T04:15:23.427714Z","shell.execute_reply.started":"2026-07-15T04:15:23.417864Z","shell.execute_reply":"2026-07-15T04:15:23.426935Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"abb9558b","cell_type":"markdown","source":"## 3. Baseline Pipeline (Paper Reproduction)\n\nImplements Thakur & Talele (2026):\n1. Frame-difference → rolling mean → z-score peak (τ=1.5)\n2. Farneback optical flow magnitude → P90 threshold → weighted centroid\n3. CLIP ViT-B/32 with **original 5 prompts / class** (Table 1)\n","metadata":{}},{"id":"aa5e60ba","cell_type":"code","source":"# [BASELINE] Temporal: frame-difference anomaly\n\ndef bl_compute_frame_diff_series(video_path, resize_w=320, resize_h=180):\n    cap = cv2.VideoCapture(str(video_path))\n    diffs, prev = [], None\n    while True:\n        ret, frame = cap.read()\n        if not ret:\n            break\n        gray = cv2.cvtColor(cv2.resize(frame, (resize_w, resize_h)), cv2.COLOR_BGR2GRAY).astype(np.float32)\n        if prev is not None:\n            diffs.append(np.mean(np.abs(gray - prev)))\n        prev = gray\n    cap.release()\n    return np.asarray(diffs, dtype=np.float32)\n\n\ndef bl_score_temporal_anomaly(diff_series, smooth_window=5):\n    series = pd.Series(diff_series)\n    smoothed = series.rolling(window=smooth_window, min_periods=1, center=True).mean().values\n    mu, sigma = smoothed.mean(), smoothed.std() + 1e-8\n    return (smoothed - mu) / sigma\n\n\ndef baseline_predict_accident_time(video_path, smooth_window=5, z_threshold=1.5) -> float:\n    cap = cv2.VideoCapture(str(video_path))\n    fps = cap.get(cv2.CAP_PROP_FPS)\n    n_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    cap.release()\n    if fps <= 0 or n_frames == 0:\n        return 0.0\n    diffs = bl_compute_frame_diff_series(video_path)\n    if len(diffs) == 0:\n        return n_frames / fps / 2.0\n    anomaly = bl_score_temporal_anomaly(diffs, smooth_window)\n    candidates = np.where(anomaly > z_threshold)[0]\n    if len(candidates) == 0:\n        peak_frame = int(np.argmax(anomaly))\n    else:\n        peak_frame = int(candidates[np.argmax(anomaly[candidates])])\n    return round(peak_frame / fps, 4)\n\nprint('[STATUS] Baseline temporal ready')\n","metadata":{"execution":{"iopub.status.busy":"2026-07-15T04:15:23.428860Z","iopub.execute_input":"2026-07-15T04:15:23.429293Z","iopub.status.idle":"2026-07-15T04:15:23.446344Z","shell.execute_reply.started":"2026-07-15T04:15:23.429259Z","shell.execute_reply":"2026-07-15T04:15:23.445550Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"82be8db9","cell_type":"code","source":"# [BASELINE] Spatial: Farneback OF weighted centroid\n\ndef bl_compute_flow_magnitude_map(video_path, resize_w=320, resize_h=180,\n                                  n_frames_context=30, center_frame=None, flow_percentile=90.0):\n    cap = cv2.VideoCapture(str(video_path))\n    total = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    if center_frame is not None:\n        start = max(0, center_frame - n_frames_context // 2)\n    else:\n        start = max(0, total // 3)\n    cap.set(cv2.CAP_PROP_POS_FRAMES, start)\n    mag_accum = np.zeros((resize_h, resize_w), dtype=np.float32)\n    prev, count = None, 0\n    while count < n_frames_context:\n        ret, frame = cap.read()\n        if not ret:\n            break\n        gray = cv2.cvtColor(cv2.resize(frame, (resize_w, resize_h)), cv2.COLOR_BGR2GRAY)\n        if prev is not None:\n            flow = cv2.calcOpticalFlowFarneback(\n                prev, gray, None, 0.5, 3, 15, 3, 5, 1.2, 0\n            )\n            mag, _ = cv2.cartToPolar(flow[..., 0], flow[..., 1])\n            mag_accum += mag\n        prev = gray\n        count += 1\n    cap.release()\n    if mag_accum.max() > 0:\n        thresh = np.percentile(mag_accum, flow_percentile)\n        mag_accum[mag_accum < thresh] = 0.0\n    return mag_accum\n\n\ndef baseline_predict_impact_location(video_path, accident_time=None, n_frames_context=30):\n    RW, RH = 320, 180\n    center_frame = None\n    if accident_time is not None:\n        cap = cv2.VideoCapture(str(video_path))\n        fps = cap.get(cv2.CAP_PROP_FPS)\n        cap.release()\n        if fps > 0:\n            center_frame = int(accident_time * fps)\n    mag = bl_compute_flow_magnitude_map(\n        video_path, RW, RH, n_frames_context, center_frame\n    )\n    total = mag.sum()\n    if total < 1e-6:\n        return 0.5, 0.5\n    ys, xs = np.mgrid[0:RH, 0:RW]\n    cx = float((xs * mag).sum() / total) / RW\n    cy = float((ys * mag).sum() / total) / RH\n    return round(cx, 6), round(cy, 6)\n\nprint('[STATUS] Baseline spatial ready')\n","metadata":{"execution":{"iopub.status.busy":"2026-07-15T04:15:23.448928Z","iopub.execute_input":"2026-07-15T04:15:23.449510Z","iopub.status.idle":"2026-07-15T04:15:23.467209Z","shell.execute_reply.started":"2026-07-15T04:15:23.449474Z","shell.execute_reply":"2026-07-15T04:15:23.466349Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"2e266427","cell_type":"code","source":"# [BASELINE] CLIP prompts — paper Table 1 (5 prompts / class)\nBASELINE_PROMPTS = {\n    'rear-end': [\n        'a car colliding into the back of another car',\n        'rear-end collision between two vehicles on a road',\n        'vehicle hitting the back of a stationary car from behind',\n        'one car rear-ending another car at a traffic light',\n        'a vehicle crashing into the tail of the car ahead',\n    ],\n    't-bone': [\n        'a car hitting the side of another car at an intersection',\n        't-bone collision at a crossroads between two vehicles',\n        'side impact crash where one car strikes another perpendicularly',\n        'a vehicle running a red light and hitting the side of crossing traffic',\n        'perpendicular collision between two cars at a junction',\n    ],\n    'head-on': [\n        'two cars colliding head-on from opposite directions',\n        'frontal collision between two vehicles on a road',\n        'head-on crash between two cars driving toward each other',\n        'two vehicles smashing front-to-front on a highway',\n        'a car crossing the center line and hitting an oncoming vehicle head-on',\n    ],\n    'sideswipe': [\n        'two vehicles scraping alongside each other while driving',\n        'sideswipe collision between cars changing lanes',\n        'glancing blow between two cars moving in the same direction',\n        'a car drifting into the adjacent lane and scraping another vehicle',\n        'two vehicles brushing sides while traveling parallel on a road',\n    ],\n    'single': [\n        'a single car crashing into a wall or barrier',\n        'one vehicle running off the road and hitting an obstacle',\n        'a car losing control and crashing into a pole or guardrail',\n        'a single vehicle spinning out and hitting a roadside object',\n        'one car veering off the road and crashing without involving another vehicle',\n    ],\n}\n\nclip_model, clip_preprocess = clip.load(CLIP_BACKBONE, device=DEVICE)\nclip_model.eval()\n\n\ndef encode_prompt_bank(prompt_dict):\n    # Mean-pool L2-normalized text embeddings per class (paper Eq. 7).\n    feats = {}\n    with torch.no_grad():\n        for ctype, prompts in prompt_dict.items():\n            tokens = clip.tokenize(prompts, truncate=True).to(DEVICE)\n            f = clip_model.encode_text(tokens).float()\n            f = f / f.norm(dim=-1, keepdim=True)\n            feats[ctype] = f.mean(dim=0)\n    return feats\n\nBASELINE_TEXT_FEATURES = encode_prompt_bank(BASELINE_PROMPTS)\nprint(f'[SUCCESS] CLIP {CLIP_BACKBONE} loaded | baseline text features ready')\n","metadata":{"execution":{"iopub.status.busy":"2026-07-15T04:15:23.468217Z","iopub.execute_input":"2026-07-15T04:15:23.468525Z","iopub.status.idle":"2026-07-15T04:15:32.465629Z","shell.execute_reply.started":"2026-07-15T04:15:23.468487Z","shell.execute_reply":"2026-07-15T04:15:32.464728Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"621c4cfd","cell_type":"code","source":"# [BASELINE] Classification + full inference\n\ndef extract_frames_around_peak(video_path, peak_time_s, n_context_frames=8, fps=None):\n    cap = cv2.VideoCapture(str(video_path))\n    if fps is None:\n        fps = cap.get(cv2.CAP_PROP_FPS) or 20.0\n    total = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    peak_frame = int(peak_time_s * fps)\n    half = n_context_frames // 2\n    idxs = range(max(0, peak_frame - half), min(total, peak_frame + half + 1))\n    frames = []\n    for idx in idxs:\n        cap.set(cv2.CAP_PROP_POS_FRAMES, idx)\n        ret, frame = cap.read()\n        if ret:\n            frames.append(PILImage.fromarray(cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)))\n    cap.release()\n    return frames\n\n\ndef clip_classify(video_path, peak_time_s, text_features, n_context_frames=8) -> str:\n    pil_frames = extract_frames_around_peak(video_path, peak_time_s, n_context_frames)\n    if not pil_frames:\n        return 'rear-end'\n    with torch.no_grad():\n        imgs = torch.stack([clip_preprocess(f) for f in pil_frames]).to(DEVICE)\n        img_f = clip_model.encode_image(imgs).float()\n        img_f = img_f / img_f.norm(dim=-1, keepdim=True)\n        img_f = img_f.mean(dim=0)\n    scores = {k: float(img_f @ v) for k, v in text_features.items()}\n    return max(scores, key=scores.get)\n\n\ndef baseline_run_inference(video_path) -> dict:\n    t = baseline_predict_accident_time(video_path)\n    cx, cy = baseline_predict_impact_location(video_path, accident_time=t)\n    k = clip_classify(video_path, t, BASELINE_TEXT_FEATURES)\n    return {\n        'path': str(video_path),\n        'accident_time': t,\n        'center_x': cx,\n        'center_y': cy,\n        'type': k,\n        'method': 'baseline',\n    }\n\nprint('[STATUS] baseline_run_inference ready')\n","metadata":{"execution":{"iopub.status.busy":"2026-07-15T04:15:32.466705Z","iopub.execute_input":"2026-07-15T04:15:32.467306Z","iopub.status.idle":"2026-07-15T04:15:32.478442Z","shell.execute_reply.started":"2026-07-15T04:15:32.467267Z","shell.execute_reply":"2026-07-15T04:15:32.477665Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"e1b84cd5","cell_type":"markdown","source":"## 4. Improved Pipeline v7 — evidence-based from your v6 run\n\n### What `accident.ipynb` v6 showed (N=60, GPU OK)\n| Metric | Baseline | v6 | Insight |\n|--------|----------|----|---------|\n| T | 0.262 | 0.262 | Keep baseline temporal |\n| S | 0.145 | **0.154** | YOLO helps mean S |\n| C | 0.233 | 0.233 | Still weak (rear-end **0%**) |\n| H | 0.041 | **0.036** | H dropped: S gains were on C=0 videos |\n\n**Conclusion:** Blind YOLO spatial ≠ better H. **Classification is the real bottleneck** (challenge scoring uses harmonic mean).\n\n### v7 design\n1. **Temporal:** baseline frame-diff\n2. **Spatial:** gated YOLO (only close pairs) blended with optical flow\n3. **Classification:** LogisticRegression on CLIP embeddings trained on **synthetic** labels (exclude EVAL) — allowed by challenge rules\n","metadata":{}},{"id":"7e7d8c7b","cell_type":"code","source":"# [IMPROVED v7] YOLO window detector (no half= dep warnings)\nyolo = YOLO(YOLO_MODEL)\n_dummy = np.zeros((YOLO_IMGSZ, YOLO_IMGSZ, 3), dtype=np.uint8)\n_ = yolo.predict(source=_dummy, device=YOLO_DEVICE, verbose=False, imgsz=YOLO_IMGSZ)\ntry:\n    param_device = next(yolo.model.parameters()).device\nexcept Exception:\n    param_device = 'unknown'\nprint(f'[SUCCESS] YOLO ready | weights_device={param_device} | YOLO_DEVICE={YOLO_DEVICE}')\n\n\ndef _video_meta(video_path):\n    cap = cv2.VideoCapture(str(video_path))\n    fps = cap.get(cv2.CAP_PROP_FPS) or 20.0\n    total = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    W = max(1, int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)))\n    H = max(1, int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)))\n    cap.release()\n    return float(fps), int(total), W, H\n\n\ndef yolo_detect_window(video_path, center_frame, half_window=12, stride=3,\n                       conf=YOLO_CONF, classes=YOLO_CLASSES, imgsz=YOLO_IMGSZ):\n    fps, total, W, H = _video_meta(video_path)\n    lo = max(0, int(center_frame) - int(half_window))\n    hi = min(total - 1, int(center_frame) + int(half_window))\n    cap = cv2.VideoCapture(str(video_path))\n    by_frame = {}\n    for f in range(lo, hi + 1, max(1, int(stride))):\n        cap.set(cv2.CAP_PROP_POS_FRAMES, f)\n        ok, frame = cap.read()\n        if not ok:\n            continue\n        res = yolo.predict(\n            source=frame, conf=conf, classes=classes,\n            device=YOLO_DEVICE, imgsz=imgsz, verbose=False,\n        )[0]\n        boxes = []\n        if res.boxes is not None and len(res.boxes) > 0:\n            for box in res.boxes.xyxy.cpu().numpy():\n                x1, y1, x2, y2 = box.tolist()\n                boxes.append({\n                    'cx': ((x1 + x2) * 0.5) / W,\n                    'cy': ((y1 + y2) * 0.5) / H,\n                })\n        by_frame[f] = boxes\n    cap.release()\n    return by_frame, fps, total\n\nprint('[STATUS] YOLO window detector ready (v7)')\n","metadata":{"execution":{"iopub.status.busy":"2026-07-15T04:15:32.479642Z","iopub.execute_input":"2026-07-15T04:15:32.480521Z","iopub.status.idle":"2026-07-15T04:15:33.211510Z","shell.execute_reply.started":"2026-07-15T04:15:32.480489Z","shell.execute_reply":"2026-07-15T04:15:33.210644Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"79e59bb6","cell_type":"code","source":"# [IMPROVED v7] Temporal=baseline | Spatial=GATED YOLO (only when close pair)\n\ndef _closest_pair(boxes):\n    if not boxes:\n        return None\n    if len(boxes) == 1:\n        return boxes[0]['cx'], boxes[0]['cy'], 1.0, False\n    best_d, best = 1e9, None\n    for i in range(len(boxes)):\n        for j in range(i + 1, len(boxes)):\n            d = math.hypot(boxes[i]['cx'] - boxes[j]['cx'], boxes[i]['cy'] - boxes[j]['cy'])\n            if d < best_d:\n                best_d = d\n                best = (\n                    0.5 * (boxes[i]['cx'] + boxes[j]['cx']),\n                    0.5 * (boxes[i]['cy'] + boxes[j]['cy']),\n                    d, True,\n                )\n    return best\n\n\ndef yolo_spatial_candidate(by_frame, peak_frame, max_pair_dist=0.12):\n    # Only accept YOLO if we see a truly close vehicle pair near peak.\n    best = None  # (score, cx, cy, dist)\n    for f, boxes in by_frame.items():\n        pair = _closest_pair(boxes)\n        if pair is None:\n            continue\n        cx, cy, dist, has_pair = pair\n        if not has_pair or dist > max_pair_dist:\n            continue\n        score = -dist - 0.01 * abs(f - peak_frame)\n        if best is None or score > best[0]:\n            best = (score, cx, cy, dist)\n    if best is None:\n        return None, None, None\n    return best[1], best[2], best[3]\n\n\ndef improved_predict_time_space(video_path, max_pair_dist=0.12, blend=0.7):\n    t = baseline_predict_accident_time(video_path)\n    ofx, ofy = baseline_predict_impact_location(video_path, accident_time=t)\n\n    fps, total, _, _ = _video_meta(video_path)\n    peak_frame = int(np.clip(round(t * fps), 0, max(0, total - 1)))\n    by_frame, _, _ = yolo_detect_window(\n        video_path, peak_frame,\n        half_window=12, stride=max(2, YOLO_VID_STRIDE), conf=YOLO_CONF,\n    )\n    n_boxes = int(sum(len(v) for v in by_frame.values()))\n    yx, yy, dist = yolo_spatial_candidate(by_frame, peak_frame, max_pair_dist=max_pair_dist)\n\n    used_yolo = False\n    if yx is not None:\n        # Blend with OF — v6 evidence: pure YOLO raised mean_S but hurt H on C=1 videos\n        cx = round(min(1.0, max(0.0, blend * yx + (1.0 - blend) * ofx)), 6)\n        cy = round(min(1.0, max(0.0, blend * yy + (1.0 - blend) * ofy)), 6)\n        used_yolo = True\n    else:\n        cx, cy = ofx, ofy\n\n    return t, cx, cy, {\n        'n_tracks': n_boxes,\n        'fallback_space': (not used_yolo),\n        'used_yolo_spatial': used_yolo,\n        'pair_dist': dist,\n        'time_source': 'baseline',\n        'peak_frame': peak_frame,\n        'fps': fps,\n    }\n\nprint('[STATUS] improved temporal/spatial v7 ready (gated YOLO)')\n","metadata":{"execution":{"iopub.status.busy":"2026-07-15T04:15:33.212518Z","iopub.execute_input":"2026-07-15T04:15:33.212814Z","iopub.status.idle":"2026-07-15T04:15:33.227117Z","shell.execute_reply.started":"2026-07-15T04:15:33.212787Z","shell.execute_reply":"2026-07-15T04:15:33.226235Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"1dfd9b43","cell_type":"code","source":"# [IMPROVED v7] Synthetic-supervised CLIP classifier (biggest lever for H)\n# Paper allows synthetic labels; real CCTV must stay unlabeled.\n# Train a linear head on CLIP ViT-B/32 embeddings from frames near GT accident time.\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.preprocessing import LabelEncoder\n\nENSEMBLE_BASE_PROMPTS = {\n    'rear-end': [\n        'a car colliding into the back of another car',\n        'rear-end collision between two vehicles on a road',\n        'vehicle hitting the rear bumper of the car ahead',\n        'one car rear-ending another car in a traffic queue',\n        'a following vehicle crashing into a lead vehicle from behind',\n        'same-direction crash where the rear car hits the front car',\n    ],\n    't-bone': [\n        'a car hitting the side of another car at an intersection',\n        't-bone collision at a crossroads between two vehicles',\n        'perpendicular side-impact crash between two cars',\n        'a vehicle striking the passenger door of crossing traffic',\n        'right-angle collision at a junction',\n        'side impact where one car T-bones another',\n    ],\n    'head-on': [\n        'two cars colliding head-on from opposite directions',\n        'frontal collision between two vehicles on a road',\n        'head-on crash between cars driving toward each other',\n        'two vehicles smashing front-to-front',\n        'opposite-direction frontal collision on a highway',\n        'cars colliding nose-to-nose after crossing into oncoming traffic',\n    ],\n    'sideswipe': [\n        'two vehicles scraping alongside each other while driving',\n        'sideswipe collision between cars changing lanes',\n        'glancing lateral blow between parallel vehicles',\n        'a car drifting into an adjacent lane and scraping another vehicle',\n        'same-direction side scrape between two cars',\n        'lateral contact between two vehicles traveling parallel',\n    ],\n    'single': [\n        'a single car crashing into a wall or barrier',\n        'one vehicle running off the road and hitting an obstacle',\n        'a car losing control and hitting a pole or guardrail',\n        'a single-vehicle crash with no other vehicle involved',\n        'one car veering off the roadway into a roadside object',\n        'solo vehicle impact against infrastructure',\n    ],\n}\nENSEMBLE_PROMPTS = {\n    k: list(dict.fromkeys(BASELINE_PROMPTS[k] + ENSEMBLE_BASE_PROMPTS[k]))\n    for k in BASELINE_PROMPTS\n}\nENSEMBLE_TEXT_FEATURES = encode_prompt_bank(ENSEMBLE_PROMPTS)\ndisplay(pd.DataFrame([{'type': k, 'n_prompts': len(v)} for k, v in ENSEMBLE_PROMPTS.items()]))\n\n\ndef clip_image_embedding(video_path, peak_time_s, n_context_frames=CLIP_CONTEXT_FRAMES):\n    pil_frames = extract_frames_around_peak(video_path, peak_time_s, n_context_frames)\n    if not pil_frames:\n        return None\n    with torch.no_grad():\n        imgs = torch.stack([clip_preprocess(f) for f in pil_frames]).to(DEVICE, non_blocking=CUDA_OK)\n        feat = clip_model.encode_image(imgs).float()\n        feat = feat / feat.norm(dim=-1, keepdim=True)\n        feat = feat.mean(dim=0)\n    return feat.detach().cpu().numpy().astype(np.float32)\n\n\n# ---- Build train pool: synthetic videos NOT in EVAL subset ----\neval_stems = set(eval_labels['video_stem'].tolist())\ntrain_pool = eval_pool[~eval_pool['video_stem'].isin(eval_stems)].copy()\n\n# Cap training size for Kaggle runtime (stratified)\nTRAIN_N = min(TRAIN_N_CLF, len(train_pool))\ntrain_labels = stratified_sample(train_pool, TRAIN_N)\nprint(f'[STATUS] Classifier train pool={len(train_pool)} | using TRAIN_N={TRAIN_N}')\ndisplay(train_labels['type'].value_counts().rename('n').to_frame())\n\n\ndef build_xy(label_rows, desc='set'):\n    X, y, kept = [], [], 0\n    for i, (_, row) in enumerate(label_rows.iterrows()):\n        stem = row['video_stem']\n        vp = stem_to_video.get(stem)\n        if vp is None:\n            continue\n        emb = clip_image_embedding(vp, float(row['accident_time']))\n        if emb is None:\n            continue\n        X.append(emb)\n        y.append(str(row['type']).strip().lower())\n        kept += 1\n        if (i + 1) % 25 == 0 or (i + 1) == len(label_rows):\n            print(f'[CLIP-emb {desc}] {i+1}/{len(label_rows)}')\n    return np.stack(X), np.array(y), kept\n\nprint('[STATUS] Extracting CLIP embeddings for train...')\nX_train, y_train, n_tr = build_xy(train_labels, 'train')\nprint(f'[SUCCESS] train embeddings: {X_train.shape}')\n\nlabel_encoder = LabelEncoder()\ny_train_id = label_encoder.fit_transform(y_train)\n\nclf = LogisticRegression(\n    max_iter=2000,\n    multi_class='multinomial',\n    class_weight='balanced',\n    C=1.0,\n    solver='lbfgs',\n)\nclf.fit(X_train, y_train_id)\ntrain_acc = float((clf.predict(X_train) == y_train_id).mean())\nprint(f'[SUCCESS] LogisticRegression trained | train_acc={train_acc:.3f} | classes={list(label_encoder.classes_)}')\n\n\ndef classify_v7(video_path, peak_time_s):\n    # 1) Supervised CLIP probe (primary)\n    emb = clip_image_embedding(video_path, peak_time_s)\n    if emb is not None:\n        pred_id = int(clf.predict(emb.reshape(1, -1))[0])\n        proba = clf.predict_proba(emb.reshape(1, -1))[0]\n        conf = float(proba.max())\n        pred = label_encoder.inverse_transform([pred_id])[0]\n        # 2) If probe uncertain, fall back to zero-shot baseline CLIP\n        if conf >= 0.35:\n            return pred, conf, 'probe'\n    zs = clip_classify(video_path, peak_time_s, BASELINE_TEXT_FEATURES, n_context_frames=CLIP_CONTEXT_FRAMES)\n    return zs, 0.0, 'zeroshot'\n\nprint('[STATUS] classify_v7 ready')\n","metadata":{"execution":{"iopub.status.busy":"2026-07-15T04:15:33.228211Z","iopub.execute_input":"2026-07-15T04:15:33.228517Z","iopub.status.idle":"2026-07-15T05:01:11.152007Z","shell.execute_reply.started":"2026-07-15T04:15:33.228489Z","shell.execute_reply":"2026-07-15T05:01:11.150950Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"581263fc","cell_type":"code","source":"# [IMPROVED v7] Full inference\n\ndef improved_run_inference(video_path) -> dict:\n    t, cx, cy, dbg = improved_predict_time_space(\n        video_path,\n        max_pair_dist=YOLO_MAX_PAIR_DIST,\n        blend=YOLO_SPATIAL_BLEND,\n    )\n    k, conf, src = classify_v7(video_path, t)\n    return {\n        'path': str(video_path),\n        'accident_time': t,\n        'center_x': cx,\n        'center_y': cy,\n        'type': k,\n        'method': 'improved_v7',\n        'n_tracks': dbg.get('n_tracks'),\n        'fallback_space': dbg.get('fallback_space'),\n        'used_yolo_spatial': dbg.get('used_yolo_spatial'),\n        'clf_conf': conf,\n        'clf_src': src,\n    }\n\nprint('[STATUS] improved_run_inference v7 ready')\nprint(f'[STATUS] CUDA_OK={CUDA_OK} DEVICE={DEVICE} YOLO_DEVICE={YOLO_DEVICE}')\n","metadata":{"execution":{"iopub.status.busy":"2026-07-15T05:01:11.155137Z","iopub.execute_input":"2026-07-15T05:01:11.156061Z","iopub.status.idle":"2026-07-15T05:01:11.168387Z","shell.execute_reply.started":"2026-07-15T05:01:11.156015Z","shell.execute_reply":"2026-07-15T05:01:11.166619Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"28f44802","cell_type":"markdown","source":"## 5. Baseline vs Improved — Quantitative Comparison\n\nRun both methods on the same stratified synthetic subset and compare mean T, S, C, H.\n","metadata":{}},{"id":"03051dbe","cell_type":"code","source":"# [COMPARE] Baseline vs Improved v7 (evaluation-only)\n\ndef gt_from_label_row(row):\n    return {\n        'accident_time': float(row['accident_time']),\n        'center_x': float(row['center_x']),\n        'center_y': float(row['center_y']),\n        'type': str(row['type']).strip().lower(),\n    }\n\n\ndef run_eval_method(method_name, infer_fn, videos, label_rows):\n    rows = []\n    t0 = time.time()\n    n_fallback = 0\n    n_yolo = 0\n    n_boxes = 0\n    for i, (vp, (_, lab)) in enumerate(zip(videos, label_rows.iterrows())):\n        pred = infer_fn(vp)\n        gt = gt_from_label_row(lab)\n        sc = score_row(pred, gt)\n        if pred.get('fallback_space'):\n            n_fallback += 1\n        if pred.get('used_yolo_spatial'):\n            n_yolo += 1\n        n_boxes += int(pred.get('n_tracks') or 0)\n        rows.append({\n            'video_stem': video_stem(vp),\n            'gt_type': gt['type'],\n            'pred_type': pred['type'],\n            'pred_t': pred['accident_time'],\n            'gt_t': gt['accident_time'],\n            'pred_cx': pred['center_x'], 'pred_cy': pred['center_y'],\n            'gt_cx': gt['center_x'], 'gt_cy': gt['center_y'],\n            'fallback_space': pred.get('fallback_space'),\n            'used_yolo_spatial': pred.get('used_yolo_spatial'),\n            'clf_src': pred.get('clf_src'),\n            'n_tracks': pred.get('n_tracks'),\n            **sc,\n        })\n        if (i + 1) % 5 == 0 or (i + 1) == len(videos):\n            print(f'[{method_name}] {i+1}/{len(videos)} videos')\n    elapsed = time.time() - t0\n    df = pd.DataFrame(rows)\n    df.attrs['elapsed_sec'] = elapsed\n    if 'improved' in method_name:\n        print(f'[DIAG] yolo_spatial={n_yolo}/{len(videos)} | of_fallback={n_fallback}/{len(videos)} | mean_boxes={n_boxes/max(1,len(videos)):.1f}')\n        if 'clf_src' in df.columns:\n            print('[DIAG] clf_src counts:')\n            print(df['clf_src'].value_counts().to_string())\n        sub = df[df['C'] == 1]\n        if len(sub):\n            print(f'[DIAG] among C=1 (n={len(sub)}): mean_S={sub[\"S\"].mean():.4f} mean_H={sub[\"H\"].mean():.4f}')\n    return df\n\n\nif RUN_EVAL:\n    print(f'[STATUS] Starting comparison on {len(eval_videos)} videos...')\n    print('--- BASELINE ---')\n    baseline_eval = run_eval_method('baseline', baseline_run_inference, eval_videos, eval_labels)\n    print('--- IMPROVED v7 ---')\n    improved_eval = run_eval_method('improved', improved_run_inference, eval_videos, eval_labels)\n\n    same_xy = int(((baseline_eval['pred_cx'] == improved_eval['pred_cx']) &\n                   (baseline_eval['pred_cy'] == improved_eval['pred_cy'])).sum())\n    same_type = int((baseline_eval['pred_type'] == improved_eval['pred_type']).sum())\n    print(f'[DIAG] identical spatial vs baseline: {same_xy}/{len(baseline_eval)}')\n    print(f'[DIAG] identical type vs baseline: {same_type}/{len(baseline_eval)}')\n    print('[SUCCESS] Comparison runs complete')\nelse:\n    print('[STATUS] RUN_EVAL=False — skipped 60-video comparison for full submission run.')\n","metadata":{"execution":{"iopub.status.busy":"2026-07-15T10:09:04.837553Z","iopub.execute_input":"2026-07-15T10:09:04.838357Z","iopub.status.idle":"2026-07-15T10:32:46.687719Z","shell.execute_reply.started":"2026-07-15T10:09:04.838319Z","shell.execute_reply":"2026-07-15T10:32:46.685889Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"38ff4924","cell_type":"code","source":"# [COMPARE] Aggregate score table (evaluation-only)\n\nif RUN_EVAL:\n    def summarize(df, name):\n        return {\n            'method': name,\n            'N': len(df),\n            'mean_T': round(df['T'].mean(), 4),\n            'mean_S': round(df['S'].mean(), 4),\n            'mean_C': round(df['C'].mean(), 4),\n            'mean_H': round(df['H'].mean(), 4),\n            'C_accuracy_%': round(100 * df['C'].mean(), 2),\n            'elapsed_sec': round(df.attrs.get('elapsed_sec', float('nan')), 1),\n        }\n\n    comparison_summary = pd.DataFrame([\n        summarize(baseline_eval, 'Baseline (paper)'),\n        summarize(improved_eval, 'Improved v7 (gated-YOLO + CLIP-probe)'),\n    ])\n\n    delta = comparison_summary.iloc[1][['mean_T', 'mean_S', 'mean_C', 'mean_H']].astype(float) \\\n          - comparison_summary.iloc[0][['mean_T', 'mean_S', 'mean_C', 'mean_H']].astype(float)\n    delta_row = {\n        'method': 'Delta (improved - baseline)',\n        'N': comparison_summary.iloc[0]['N'],\n        'mean_T': round(float(delta['mean_T']), 4),\n        'mean_S': round(float(delta['mean_S']), 4),\n        'mean_C': round(float(delta['mean_C']), 4),\n        'mean_H': round(float(delta['mean_H']), 4),\n        'C_accuracy_%': round(100 * float(delta['mean_C']), 2),\n        'elapsed_sec': None,\n    }\n    comparison_summary = pd.concat([comparison_summary, pd.DataFrame([delta_row])], ignore_index=True)\n\n    print('========== BASELINE vs IMPROVED ==========')\n    display(comparison_summary)\n    comparison_summary.to_csv(OUTPUT_DIR / 'comparison_summary.csv', index=False)\n    print('[SUCCESS] Saved /kaggle/working/comparison_summary.csv')\nelse:\n    print('[STATUS] RUN_EVAL=False — skipped comparison summary.')\n","metadata":{"execution":{"iopub.status.busy":"2026-07-15T10:32:46.776507Z","iopub.execute_input":"2026-07-15T10:32:46.776869Z","iopub.status.idle":"2026-07-15T10:32:46.808090Z","shell.execute_reply.started":"2026-07-15T10:32:46.776837Z","shell.execute_reply":"2026-07-15T10:32:46.807276Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"428300a2","cell_type":"code","source":"# [COMPARE] Per-class accuracy and plots (evaluation-only)\n\nif RUN_EVAL:\n    def per_class_acc(df):\n        return df.groupby('gt_type')['C'].mean().rename('accuracy')\n\n    pc = pd.DataFrame({\n        'baseline_C': per_class_acc(baseline_eval),\n        'improved_C': per_class_acc(improved_eval),\n    }).fillna(0.0)\n    pc['Delta'] = pc['improved_C'] - pc['baseline_C']\n    display(pc.round(4))\n\n    fig, axes = plt.subplots(1, 2, figsize=(12, 4))\n    metrics = ['mean_T', 'mean_S', 'mean_C', 'mean_H']\n    x = np.arange(len(metrics))\n    b = comparison_summary.iloc[0][metrics].astype(float).values\n    im = comparison_summary.iloc[1][metrics].astype(float).values\n    w = 0.35\n    axes[0].bar(x - w/2, b, w, label='Baseline (paper)')\n    axes[0].bar(x + w/2, im, w, label='Improved')\n    axes[0].set_xticks(x)\n    axes[0].set_xticklabels(['T', 'S', 'C', 'H'])\n    axes[0].set_ylim(0, 1)\n    axes[0].set_title('Component scores (synthetic eval)')\n    axes[0].legend()\n\n    vc = improved_eval['pred_type'].value_counts().reindex(COLLISION_TYPE_DIRS, fill_value=0)\n    vc.plot(kind='bar', ax=axes[1], color='steelblue')\n    axes[1].set_title('Improved: predicted type counts')\n    axes[1].set_xlabel('')\n    plt.tight_layout()\n    plt.savefig(OUTPUT_DIR / 'comparison_plots.png', dpi=140, bbox_inches='tight')\n    plt.show()\n    print('[SUCCESS] Saved comparison_plots.png')\nelse:\n    print('[STATUS] RUN_EVAL=False — skipped evaluation plots.')\n","metadata":{"execution":{"iopub.status.busy":"2026-07-15T10:32:47.689142Z","iopub.execute_input":"2026-07-15T10:32:47.689939Z","iopub.status.idle":"2026-07-15T10:32:48.297276Z","shell.execute_reply.started":"2026-07-15T10:32:47.689904Z","shell.execute_reply":"2026-07-15T10:32:48.296413Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"3419eea9","cell_type":"code","source":"# [COMPARE] Confusion matrices (evaluation-only)\nfrom sklearn.metrics import confusion_matrix\n\nif RUN_EVAL:\n    fig, axes = plt.subplots(1, 2, figsize=(12, 5))\n    for ax, df, title in [\n        (axes[0], baseline_eval, 'Baseline confusion'),\n        (axes[1], improved_eval, 'Improved confusion'),\n    ]:\n        labels = COLLISION_TYPE_DIRS\n        cm = confusion_matrix(df['gt_type'], df['pred_type'], labels=labels)\n        sns.heatmap(cm, annot=True, fmt='d', cmap='Blues',\n                    xticklabels=labels, yticklabels=labels, ax=ax)\n        ax.set_xlabel('Predicted')\n        ax.set_ylabel('Ground truth')\n        ax.set_title(title)\n    plt.tight_layout()\n    plt.savefig(OUTPUT_DIR / 'confusion_matrices.png', dpi=140, bbox_inches='tight')\n    plt.show()\n\n    baseline_eval.to_csv(OUTPUT_DIR / 'baseline_eval_details.csv', index=False)\n    improved_eval.to_csv(OUTPUT_DIR / 'improved_eval_details.csv', index=False)\n    print('[SUCCESS] Saved confusion + detail CSVs')\nelse:\n    print('[STATUS] RUN_EVAL=False — skipped confusion matrices.')\n","metadata":{"execution":{"iopub.status.busy":"2026-07-15T10:32:48.299289Z","iopub.execute_input":"2026-07-15T10:32:48.299707Z","iopub.status.idle":"2026-07-15T10:32:49.246777Z","shell.execute_reply.started":"2026-07-15T10:32:48.299674Z","shell.execute_reply":"2026-07-15T10:32:49.245882Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"9fb69b2c","cell_type":"markdown","source":"## 6. Qualitative Example\n\nOne evaluation video: GT vs Baseline vs Improved peak frames and impact markers.\n","metadata":{}},{"id":"c2401012","cell_type":"code","source":"# [VIZ] Qualitative example (evaluation-only)\nif RUN_EVAL:\n    sample_idx = 0\n    vp = eval_videos[sample_idx]\n    gt = gt_from_label_row(eval_labels.iloc[sample_idx])\n    bp = baseline_run_inference(vp)\n    ip = improved_run_inference(vp)\n\n    cap = cv2.VideoCapture(str(vp))\n    fps = cap.get(cv2.CAP_PROP_FPS) or 20.0\n    total = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    W = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH))\n    H = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))\n\n    def grab(t):\n        f = min(total - 1, max(0, int(t * fps)))\n        cap.set(cv2.CAP_PROP_POS_FRAMES, f)\n        ok, fr = cap.read()\n        return f, cv2.cvtColor(fr, cv2.COLOR_BGR2RGB) if ok else None\n\n    _, img_gt = grab(gt['accident_time'])\n    _, img_b = grab(bp['accident_time'])\n    _, img_i = grab(ip['accident_time'])\n    cap.release()\n\n    fig, axes = plt.subplots(1, 3, figsize=(15, 4))\n    for ax, img, title, pred in [\n        (axes[0], img_gt, f'GT t={gt[\"accident_time\"]:.2f}s type={gt[\"type\"]}', gt),\n        (axes[1], img_b, f'Baseline t={bp[\"accident_time\"]:.2f}s type={bp[\"type\"]}', bp),\n        (axes[2], img_i, f'Improved t={ip[\"accident_time\"]:.2f}s type={ip[\"type\"]}', ip),\n    ]:\n        if img is not None:\n            ax.imshow(img)\n            ax.scatter([pred['center_x'] * W], [pred['center_y'] * H], c='red', s=80, marker='x')\n        ax.set_title(title, fontsize=10)\n        ax.axis('off')\n    plt.suptitle(pathlib.Path(vp).name)\n    plt.tight_layout()\n    plt.savefig(OUTPUT_DIR / 'qualitative_sample.png', dpi=140, bbox_inches='tight')\n    plt.show()\nelse:\n    print('[STATUS] RUN_EVAL=False — skipped qualitative example.')\n","metadata":{"execution":{"iopub.status.busy":"2026-07-15T05:01:11.270420Z","iopub.execute_input":"2026-07-15T05:01:11.270989Z","iopub.status.idle":"2026-07-15T05:01:11.287107Z","shell.execute_reply.started":"2026-07-15T05:01:11.270958Z","shell.execute_reply":"2026-07-15T05:01:11.286084Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"45f2efad","cell_type":"markdown","source":"## 7. Full Test Inference + Submission (Improved Only)\n\nSet `RUN_FULL_TEST = True` in the config cell, then Restart & Run All.\n","metadata":{}},{"id":"39029261","cell_type":"code","source":"# [SUBMIT] Resumable full-test inference with atomic checkpoints\nimport os\n\nCHECKPOINT_PATH = OUTPUT_DIR / 'test_predictions_checkpoint.csv'\nCHECKPOINT_TMP_PATH = OUTPUT_DIR / 'test_predictions_checkpoint.tmp.csv'\nSUBMISSION_PATH = OUTPUT_DIR / 'submission.csv'\nPREDICTION_COLUMNS = ['path', 'accident_time', 'center_x', 'center_y', 'type']\nVALID_TYPES = set(COLLISION_TYPE_DIRS)\n\n\ndef load_checkpoint(path, expected_paths):\n    if not path.exists():\n        return pd.DataFrame(columns=PREDICTION_COLUMNS)\n    df = pd.read_csv(path)\n    missing_columns = [c for c in PREDICTION_COLUMNS if c not in df.columns]\n    if missing_columns:\n        raise ValueError(f'Checkpoint missing columns: {missing_columns}')\n    df = df[df['path'].isin(expected_paths)].copy()\n    df = df.drop_duplicates(subset=['path'], keep='last').reset_index(drop=True)\n    print(f'[RESUME] Loaded {len(df)} completed predictions from {path}')\n    return df\n\n\ndef save_checkpoint_atomic(df, path=CHECKPOINT_PATH, tmp_path=CHECKPOINT_TMP_PATH):\n    clean = df.drop_duplicates(subset=['path'], keep='last').reset_index(drop=True)\n    clean.to_csv(tmp_path, index=False)\n    os.replace(tmp_path, path)\n    return clean\n\n\ndef validate_submission(df, expected_paths):\n    errors = []\n    if list(df.columns) != PREDICTION_COLUMNS:\n        errors.append(f'columns={list(df.columns)}')\n    if len(df) != len(expected_paths):\n        errors.append(f'rows={len(df)}, expected={len(expected_paths)}')\n    if df['path'].duplicated().any():\n        errors.append('duplicate paths')\n    if set(df['path']) != set(expected_paths):\n        errors.append('path set mismatch')\n\n    numeric = df[['accident_time', 'center_x', 'center_y']].apply(pd.to_numeric, errors='coerce')\n    if not np.isfinite(numeric.to_numpy()).all():\n        errors.append('non-finite numeric predictions')\n    if (numeric['accident_time'] < 0).any():\n        errors.append('negative accident_time')\n    if not numeric['center_x'].between(0, 1).all():\n        errors.append('center_x outside [0,1]')\n    if not numeric['center_y'].between(0, 1).all():\n        errors.append('center_y outside [0,1]')\n    invalid_types = sorted(set(df['type'].astype(str)) - VALID_TYPES)\n    if invalid_types:\n        errors.append(f'invalid types={invalid_types}')\n    if errors:\n        raise ValueError('Invalid submission: ' + '; '.join(errors))\n    return True\n\n\nif RUN_FULL_TEST:\n    expected_paths = ['videos/' + vp.name for vp in real_videos]\n    checkpoint_df = load_checkpoint(CHECKPOINT_PATH, set(expected_paths))\n    completed_paths = set(checkpoint_df['path'].astype(str))\n    remaining_videos = [vp for vp in real_videos if 'videos/' + vp.name not in completed_paths]\n\n    print(f'[STATUS] Full test videos: {len(real_videos)}')\n    print(f'[STATUS] Already complete: {len(completed_paths)} | Remaining: {len(remaining_videos)}')\n    print(f'[STATUS] Checkpoint: {CHECKPOINT_PATH}')\n\n    new_results = []\n    t0 = time.time()\n    for i, vp in enumerate(remaining_videos, start=1):\n        result = improved_run_inference(vp)\n        result['path'] = 'videos/' + vp.name\n        new_results.append(result)\n\n        should_checkpoint = (i % CHECKPOINT_EVERY == 0) or (i == len(remaining_videos))\n        if should_checkpoint:\n            batch_df = pd.DataFrame(new_results)\n            checkpoint_df = pd.concat([checkpoint_df, batch_df], ignore_index=True)\n            checkpoint_df = save_checkpoint_atomic(checkpoint_df)\n            new_results.clear()\n\n            elapsed = time.time() - t0\n            rate = i / max(elapsed, 1e-9)\n            remaining_count = len(remaining_videos) - i\n            eta_min = remaining_count / max(rate, 1e-9) / 60.0\n            total_done = len(completed_paths) + i\n            print(\n                f'[CHECKPOINT] {total_done}/{len(real_videos)} complete | '\n                f'run={elapsed/60:.1f} min | rate={rate:.3f} video/s | ETA={eta_min:.1f} min'\n            )\n\n    checkpoint_df = load_checkpoint(CHECKPOINT_PATH, set(expected_paths))\n    missing_paths = sorted(set(expected_paths) - set(checkpoint_df['path'].astype(str)))\n\n    if missing_paths:\n        print(f'[INCOMPLETE] Missing {len(missing_paths)} videos; checkpoint retained for resume.')\n        print('[INCOMPLETE] First missing paths:', missing_paths[:10])\n    else:\n        ordered_paths = sample_sub[['path']] if not sample_sub.empty else pd.DataFrame({'path': expected_paths})\n        submission_df = ordered_paths.merge(\n            checkpoint_df[PREDICTION_COLUMNS], on='path', how='left', validate='one_to_one'\n        )[PREDICTION_COLUMNS]\n        validate_submission(submission_df, expected_paths)\n        submission_df.to_csv(SUBMISSION_PATH, index=False)\n        print(f'[SUCCESS] Validated and wrote {SUBMISSION_PATH} | rows={len(submission_df)}')\n        display(submission_df.head())\n        display(submission_df['type'].value_counts().rename('count').to_frame())\nelse:\n    print('[STATUS] RUN_FULL_TEST=False — skipped full test inference.')\n","metadata":{"execution":{"iopub.status.busy":"2026-07-15T05:01:11.288367Z","iopub.execute_input":"2026-07-15T05:01:11.289501Z","iopub.status.idle":"2026-07-15T09:48:04.860784Z","shell.execute_reply.started":"2026-07-15T05:01:11.289450Z","shell.execute_reply":"2026-07-15T09:48:04.858720Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"99c9fe80","cell_type":"markdown","source":"## 8. Conclusion (v7)\n\nEvidence chain from your runs:\n1. GPU works (Tesla T4)\n2. YOLO spatial alone: +S but **-H** (gains land on wrong-class videos)\n3. Zero-shot CLIP C≈23%, **rear-end=0%** → harmonic mean ceiling is tiny\n\nv7 bets on **synthetic-supervised CLIP linear probe** (challenge-legal) + **gated YOLO spatial**.\n\nExpected: C and H jump; S ≥ baseline.\n","metadata":{}}]}