{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":127283,"databundleVersionId":15634477}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install ultralytics","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T18:36:45.599829Z","iopub.execute_input":"2026-04-08T18:36:45.600201Z","iopub.status.idle":"2026-04-08T18:36:52.916993Z","shell.execute_reply.started":"2026-04-08T18:36:45.600167Z","shell.execute_reply":"2026-04-08T18:36:52.916132Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nfrom ultralytics import YOLO\nfrom itertools import combinations\nimport torch # Import torch to check device\n\nclass KFHEngine:\n    def __init__(self, yolo_model_path='yolov8n.pt', flow_threshold=5.0, proximity_threshold=50, frame_stride=3):\n        self.model = YOLO(yolo_model_path)\n        self.flow_threshold = flow_threshold         \n        self.proximity_threshold = proximity_threshold \n        self.vehicle_classes = [2, 3, 5, 7] \n        \n        # NEW: Frame Stride (Process 1 frame every X frames)\n        self.frame_stride = frame_stride\n\n    def _get_bbox_center(self, box):\n        x1, y1, x2, y2 = box\n        return int((x1 + x2) / 2), int((y1 + y2) / 2)\n\n    def _calculate_distance(self, center1, center2):\n        return np.linalg.norm(np.array(center1) - np.array(center2))\n\n    def _get_union_bbox(self, box1, box2, frame_shape):\n        x1 = max(0, min(box1[0], box2[0]))\n        y1 = max(0, min(box1[1], box2[1]))\n        x2 = min(frame_shape[1], max(box1[2], box2[2]))\n        y2 = min(frame_shape[0], max(box1[3], box2[3]))\n        return int(x1), int(y1), int(x2), int(y2)\n\n    def process_video(self, video_path):\n        cap = cv2.VideoCapture(video_path)\n        \n        # --- FIX: Grab all video properties immediately at the top ---\n        fps = cap.get(cv2.CAP_PROP_FPS)\n        total_frames = cap.get(cv2.CAP_PROP_FRAME_COUNT) # 🚀 Safely locked in memory here!\n        frame_width = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH))\n        frame_height = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))\n        \n        prev_gray_crops = {} \n        highest_flow_spike = 0\n        predicted_frame = -1\n        predicted_center = (0.5, 0.5) \n\n        frame_idx = 0\n        \n        results = self.model.track(\n            video_path, \n            classes=self.vehicle_classes, \n            persist=True, \n            tracker=\"botsort.yaml\", \n            stream=True, \n            verbose=False,\n            device='cuda:0',  \n            vid_stride=self.frame_stride \n        )\n        \n        for result in results:\n            frame_idx += self.frame_stride \n            \n            if result.boxes is None or result.boxes.id is None:\n                continue \n                \n            current_frame = result.orig_img\n            gray_frame = cv2.cvtColor(current_frame, cv2.COLOR_BGR2GRAY)\n            \n            boxes = result.boxes.xyxy.cpu().numpy()\n            track_ids = result.boxes.id.int().cpu().tolist()\n            \n            active_vehicles = list(zip(track_ids, boxes))\n            \n            for (id1, box1), (id2, box2) in combinations(active_vehicles, 2):\n                center1 = self._get_bbox_center(box1)\n                center2 = self._get_bbox_center(box2)\n                \n                if self._calculate_distance(center1, center2) < self.proximity_threshold:\n                    ux1, uy1, ux2, uy2 = self._get_union_bbox(box1, box2, gray_frame.shape)\n                    pair_id = tuple(sorted([id1, id2]))\n                    current_crop = gray_frame[uy1:uy2, ux1:ux2]\n                    \n                    if pair_id in prev_gray_crops:\n                        prev_crop = prev_gray_crops[pair_id]\n                        if prev_crop.shape == current_crop.shape and prev_crop.size > 0:\n                            flow = cv2.calcOpticalFlowFarneback(prev_crop, current_crop, None, \n                                                                0.5, 3, 15, 3, 5, 1.2, 0)\n                            mag, _ = cv2.cartToPolar(flow[..., 0], flow[..., 1])\n                            mean_flow_magnitude = np.mean(mag)\n                            \n                            if mean_flow_magnitude > highest_flow_spike:\n                                highest_flow_spike = mean_flow_magnitude\n                                predicted_frame = frame_idx\n                                \n                                center_x = ((ux1 + ux2) / 2) / frame_width\n                                center_y = ((uy1 + uy2) / 2) / frame_height\n                                predicted_center = (center_x, center_y)\n                                \n                    prev_gray_crops[pair_id] = current_crop\n\n        cap.release()\n        \n        # --- THE SMART FALLBACK FIX ---\n        if predicted_frame == -1:\n            predicted_frame = total_frames / 2\n            predicted_center = (0.5, 0.5) \n            \n        predicted_time = predicted_frame / fps if fps > 0 else 0.0\n        predicted_time = max(0.0, predicted_time) \n        \n        return predicted_time, predicted_center[0], predicted_center[1], highest_flow_spike","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-08T18:58:47.467500Z","iopub.execute_input":"2026-04-08T18:58:47.468384Z","iopub.status.idle":"2026-04-08T18:58:47.485447Z","shell.execute_reply.started":"2026-04-08T18:58:47.468349Z","shell.execute_reply":"2026-04-08T18:58:47.484613Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport math\n\nclass KinematicClassifier:\n    \"\"\"\n    Zero-Shot Accident Classifier based on bounding box trajectory vectors.\n    \"\"\"\n    def __init__(self, vector_frame_window=10):\n        # How many frames back we look to calculate the trajectory vector\n        self.window = vector_frame_window \n        \n    def _calculate_vector(self, track_history):\n        \"\"\"\n        Calculates the movement vector (dx, dy) of a single bounding box \n        over the given frame window.\n        \"\"\"\n        if len(track_history) < 2:\n            return np.array([0.0, 0.0])\n            \n        # Get the oldest available center point within the window\n        start_idx = max(0, len(track_history) - self.window)\n        start_point = track_history[start_idx]\n        \n        # Get the impact point (the most recent point)\n        end_point = track_history[-1]\n        \n        # Vector = End - Start\n        return np.array([end_point[0] - start_point[0], end_point[1] - start_point[1]])\n\n    def _angle_between_vectors(self, v1, v2):\n        \"\"\"Calculates the angle in degrees between two 2D vectors.\"\"\"\n        unit_v1 = v1 / (np.linalg.norm(v1) + 1e-8) # Add epsilon to avoid division by zero\n        unit_v2 = v2 / (np.linalg.norm(v2) + 1e-8)\n        \n        dot_product = np.dot(unit_v1, unit_v2)\n        # Clip to [-1, 1] to avoid NaN errors from floating point imprecision\n        angle_rad = np.arccos(np.clip(dot_product, -1.0, 1.0))\n        return np.degrees(angle_rad)\n\n    def classify_collision(self, track_history_A, track_history_B):\n        \"\"\"\n        Classifies the accident based on the trajectory histories of the two vehicles.\n        track_history: List of (x, y) center coordinates leading up to the crash.\n        \"\"\"\n        vec_A = self._calculate_vector(track_history_A)\n        vec_B = self._calculate_vector(track_history_B)\n        \n        # If one vehicle was stationary (vector magnitude near 0), \n        # we check where it was hit relative to its bounding box aspect ratio (advanced logic).\n        # For this baseline, we focus on moving vehicles.\n        if np.linalg.norm(vec_A) < 1.0 or np.linalg.norm(vec_B) < 1.0:\n            return \"side-swipe\" # Fallback/default for stationary hits\n            \n        angle = self._angle_between_vectors(vec_A, vec_B)\n        \n        # Classify based on the intersection angle\n        if angle <= 45 or angle >= 315:\n            return \"rear-end\"\n        elif 45 < angle < 135 or 225 < angle < 315:\n            return \"t-bone\"\n        elif 135 <= angle <= 225:\n            return \"head-on\"\n        else:\n            return \"unknown\"\n\n# --- USAGE EXAMPLE ---\n# Assume these lists are populated by our Phase 1 tracking engine leading up to the crash\n# track_A = [(100, 500), (105, 500), (110, 500)] # Moving right\n# track_B = [(500, 500), (490, 500), (480, 500)] # Moving left\n# \n# classifier = KinematicClassifier()\n# accident_type = classifier.classify_collision(track_A, track_B)\n# print(f\"Predicted Collision Type: {accident_type}\") # Output: head-on","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T18:37:37.048108Z","iopub.execute_input":"2026-04-08T18:37:37.048518Z","iopub.status.idle":"2026-04-08T18:37:37.058308Z","shell.execute_reply.started":"2026-04-08T18:37:37.048490Z","shell.execute_reply":"2026-04-08T18:37:37.057611Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport itertools\n\nclass ACCIDENTScorer:\n    \"\"\"\n    Evaluates predictions against ground truth and calculates the Harmonic Mean metric.\n    \"\"\"\n    def __init__(self, sigma_t=1.0, sigma_s=0.1):\n        # We assume some variance (sigma) for the Gaussian falloff. \n        # In the actual competition, this might be fixed, but we define it here for completeness.\n        self.sigma_t = sigma_t\n        self.sigma_s = sigma_s\n\n    def calculate_T(self, pred_time, true_time):\n        \"\"\"Gaussian similarity for temporal localization.\"\"\"\n        error_sq = (pred_time - true_time) ** 2\n        return np.exp(-error_sq / (2 * self.sigma_t ** 2))\n\n    def calculate_S(self, pred_x, pred_y, true_x, true_y):\n        \"\"\"Gaussian similarity for spatial localization.\"\"\"\n        error_sq = (pred_x - true_x) ** 2 + (pred_y - true_y) ** 2\n        return np.exp(-error_sq / (2 * self.sigma_s ** 2))\n\n    def calculate_C(self, pred_type, true_type):\n        \"\"\"Exact match for classification.\"\"\"\n        return 1.0 if pred_type == true_type else 0.0\n\n    def get_harmonic_mean(self, T, S, C):\n        \"\"\"Calculates the competition final score.\"\"\"\n        # Add a tiny epsilon to prevent division by zero if a score is completely 0\n        eps = 1e-6\n        T, S, C = max(T, eps), max(S, eps), max(C, eps)\n        return 3 / ((1/T) + (1/S) + (1/C))\n\ndef optimize_hyperparameters(synthetic_labels_path, video_dir):\n    \"\"\"\n    Runs a grid search over our KFH Engine thresholds using the synthetic data.\n    \"\"\"\n    print(\"Loading synthetic ground truth...\")\n    df = pd.read_csv(synthetic_labels_path)\n    \n    # Define the grid of parameters we want to test\n    flow_thresholds = [2.0, 3.5, 5.0]\n    proximity_thresholds = [40, 60, 80]\n    \n    best_score = 0\n    best_params = {}\n    scorer = ACCIDENTScorer()\n\n    # Loop through all combinations of our hyperparameters\n    for flow_thresh, prox_thresh in itertools.product(flow_thresholds, proximity_thresholds):\n        print(f\"\\n--- Testing Flow: {flow_thresh}, Proximity: {prox_thresh} ---\")\n        \n        # Initialize our engines with the current test parameters\n        # engine = KFHEngine(flow_threshold=flow_thresh, proximity_threshold=prox_thresh)\n        # classifier = KinematicClassifier()\n        \n        total_score = 0\n        valid_videos = 0\n        \n        # In a real run, you'd loop over `df` rows here. \n        # For demonstration, we simulate processing a batch of videos:\n        for index, row in df.head(5).iterrows(): # Just testing on first 5 for speed\n            \n            # --- SIMULATED PIPELINE EXECUTION ---\n            # 1. Run Phase 1\n            # pred_time, pred_x, pred_y, _ = engine.process_video(video_dir + row['rgb_path'])\n            \n            # 2. Run Phase 2 (Assuming track_histories were saved by Phase 1)\n            # pred_type = classifier.classify_collision(track_A, track_B)\n            \n            # (Mock predictions for the sake of the script running)\n            pred_time = row['accident_time'] + np.random.uniform(-0.5, 0.5)\n            pred_x = row['center_x'] + np.random.uniform(-0.05, 0.05)\n            pred_y = row['center_y'] + np.random.uniform(-0.05, 0.05)\n            pred_type = row['type'] if np.random.rand() > 0.2 else \"unknown\" # 80% accuracy\n            \n            # 3. Calculate Component Scores\n            T = scorer.calculate_T(pred_time, row['accident_time'])\n            S = scorer.calculate_S(pred_x, pred_y, row['center_x'], row['center_y'])\n            C = scorer.calculate_C(pred_type, row['type'])\n            \n            # 4. Calculate Final Video Score\n            video_score = scorer.get_harmonic_mean(T, S, C)\n            total_score += video_score\n            valid_videos += 1\n            \n        avg_score = total_score / valid_videos\n        print(f\"Average Harmonic Mean Score: {avg_score:.4f}\")\n        \n        if avg_score > best_score:\n            best_score = avg_score\n            best_params = {'flow': flow_thresh, 'proximity': prox_thresh}\n\n    print(f\"\\nOptimization Complete!\")\n    print(f\"Best Score: {best_score:.4f} with params: {best_params}\")\n    return best_params\n\n# --- USAGE EXAMPLE ---\n# best_hyperparameters = optimize_hyperparameters('labels.csv', 'videos/')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T18:59:06.951763Z","iopub.execute_input":"2026-04-08T18:59:06.952359Z","iopub.status.idle":"2026-04-08T18:59:06.966323Z","shell.execute_reply.started":"2026-04-08T18:59:06.952325Z","shell.execute_reply":"2026-04-08T18:59:06.965436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\nfrom tqdm import tqdm # For a nice progress bar in the notebook\n\ndef generate_submission(test_metadata_path, video_dir, best_params):\n    \"\"\"\n    Runs the optimized KFH Engine over the real CCTV test set and formats the output.\n    \"\"\"\n    print(\"Initializing Models with Optimized Parameters...\")\n    engine = KFHEngine(\n        yolo_model_path='yolov8n.pt', \n        flow_threshold=best_params.get('flow', 3.5), \n        proximity_threshold=best_params.get('proximity', 60)\n    )\n    classifier = KinematicClassifier()\n    \n    # Load test metadata\n    test_df = pd.read_csv(test_metadata_path)\n    submission_data = []\n    \n    print(\"Processing Test Videos...\")\n    for index, row in tqdm(test_df.iterrows(), total=test_df.shape[0]):\n        video_path = os.path.join(video_dir, row['path']) # e.g., videos/iKpoAkiKqjw_00.mp4\n        \n        # 1. Spatio-Temporal Prediction\n        try:\n            # Note: In a full implementation, engine.process_video would return the track histories too\n            pred_time, pred_x, pred_y, max_flow = engine.process_video(video_path)\n            \n            # 2. Collision Type Prediction\n            # (Assuming track_A and track_B were returned by the engine at the time of max_flow)\n            # pred_type = classifier.classify_collision(track_A, track_B)\n            \n            # Placeholder for the classification call (since we mocked the return in Phase 1 for brevity)\n            pred_type = \"rear-end\" # Default fallback\n            if max_flow > best_params.get('flow', 3.5):\n                 # If we actually caught a crash, run the classifier\n                 # pred_type = classifier.classify_collision(actual_tracks...)\n                 pass \n                 \n        except Exception as e:\n            # KAGGLE SURVIVAL RULE: Never let one bad video crash your submission loop!\n            print(f\"Error processing {video_path}: {e}\")\n            pred_time, pred_x, pred_y, pred_type = 0.0, 0.5, 0.5, \"unknown\"\n            \n        # Append to submission list in the EXACT format requested\n        submission_data.append({\n            'path': row['path'],\n            'accident_time': round(pred_time, 2),\n            'center_x': round(pred_x, 3),\n            'center_y': round(pred_y, 3),\n            'type': pred_type\n        })\n        \n    # Create final DataFrame and save to CSV\n    sub_df = pd.DataFrame(submission_data)\n    sub_df.to_csv('submission.csv', index=False)\n    print(\"Submission saved to submission.csv successfully!\")\n    return sub_df\n\n# --- USAGE EXAMPLE ---\n# Assume best_params was returned from Phase 3\n# final_sub = generate_submission('test_metadata.csv', 'videos/', best_params={'flow': 3.5, 'proximity': 60})\n# display(final_sub.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T18:59:11.080522Z","iopub.execute_input":"2026-04-08T18:59:11.081123Z","iopub.status.idle":"2026-04-08T18:59:11.089624Z","shell.execute_reply.started":"2026-04-08T18:59:11.081088Z","shell.execute_reply":"2026-04-08T18:59:11.088932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ndef visualize_prediction(video_path, pred_time, pred_x, pred_y, pred_type):\n    \"\"\"\n    Extracts the exact frame of the predicted accident and plots a target on it.\n    This creates a great thumbnail for your public notebook.\n    \"\"\"\n    cap = cv2.VideoCapture(video_path)\n    fps = cap.get(cv2.CAP_PROP_FPS)\n    target_frame = int(pred_time * fps)\n    \n    cap.set(cv2.CAP_PROP_POS_FRAMES, target_frame)\n    ret, frame = cap.read()\n    \n    if ret:\n        frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n        height, width, _ = frame.shape\n        \n        # Convert normalized coordinates back to pixels\n        px_x = int(pred_x * width)\n        px_y = int(pred_y * height)\n        \n        plt.figure(figsize=(12, 8))\n        plt.imshow(frame)\n        \n        # Draw a high-visibility target at the crash location\n        plt.plot(px_x, px_y, marker='X', color='red', markersize=20, markeredgecolor='white', markeredgewidth=2)\n        \n        plt.title(f\"Predicted Accident at {pred_time:.2f}s | Type: {pred_type.upper()}\", fontsize=16, color='red', weight='bold')\n        plt.axis('off')\n        plt.show()\n    \n    cap.release()\n\n# --- USAGE EXAMPLE ---\n# visualize_prediction('path/to/synthetic/video.mp4', 8.25, 0.633, 0.125, 't-bone')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T18:59:16.098426Z","iopub.execute_input":"2026-04-08T18:59:16.099130Z","iopub.status.idle":"2026-04-08T18:59:16.105402Z","shell.execute_reply.started":"2026-04-08T18:59:16.099099Z","shell.execute_reply":"2026-04-08T18:59:16.104591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# MASTER EXECUTION BLOCK\n# Run this cell to start the actual processing\n# ==========================================\nimport os\nfrom pathlib import Path\n\n# 1. SET YOUR CORRECTED KAGGLE PATHS\n# Based on our diagnostic script, this is the true root of the dataset\nKAGGLE_BASE_DIR = Path('/kaggle/input/competitions/accident')\n\n# Notice the 'sim_dataset' folder catching the labels!\nSYNTHETIC_LABELS_PATH = KAGGLE_BASE_DIR / 'sim_dataset' / 'labels.csv'\nTEST_METADATA_PATH = KAGGLE_BASE_DIR / 'test_metadata.csv'\n\n# We pass the base directory as a string with a trailing slash \n# so our functions can cleanly append the relative paths from the CSVs (e.g., 'videos/clip.mp4')\nBASE_DIR_STR = str(KAGGLE_BASE_DIR) + '/'\n\nprint(\"🚀 Starting KFH Engine Pipeline...\")\n\n# 2. RUN PHASE 3: OPTIMIZATION (Find the best thresholds)\nprint(\"\\n--- Running Hyperparameter Optimization on Synthetic Data ---\")\ntry:\n    best_hyperparameters = optimize_hyperparameters(SYNTHETIC_LABELS_PATH, BASE_DIR_STR)\n    print(f\"✅ Optimization Complete. Locked in parameters: {best_hyperparameters}\")\nexcept FileNotFoundError as e:\n    print(f\"⚠️ Warning: Could not find synthetic labels. Error: {e}\\nUsing default parameters.\")\n    best_hyperparameters = {'flow': 3.5, 'proximity': 60}\n\n# 3. RUN PHASE 4: INFERENCE & SUBMISSION\nprint(\"\\n--- Generating Final Predictions on Real Test Set ---\")\ntry:\n    final_submission_df = generate_submission(TEST_METADATA_PATH, BASE_DIR_STR, best_hyperparameters)\n    \n    print(\"\\n✅ Submission generated successfully! Preview:\")\n    display(final_submission_df.head())\nexcept Exception as e:\n    print(f\"❌ Error during submission generation: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T19:03:41.174169Z","iopub.execute_input":"2026-04-08T19:03:41.174464Z","iopub.status.idle":"2026-04-08T19:03:41.194740Z","shell.execute_reply.started":"2026-04-08T19:03:41.174436Z","shell.execute_reply":"2026-04-08T19:03:41.193638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(\"🔍 Scanning Kaggle Input Directory...\")\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        # We only want to print the paths if they contain our target files\n        if filename in ['test_metadata.csv', 'labels.csv']:\n            print(f\"FOUND: {os.path.join(dirname, filename)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T18:48:06.213453Z","iopub.execute_input":"2026-04-08T18:48:06.213798Z","iopub.status.idle":"2026-04-08T18:48:50.949678Z","shell.execute_reply.started":"2026-04-08T18:48:06.213764Z","shell.execute_reply":"2026-04-08T18:48:50.948957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}