{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":92399,"databundleVersionId":11038207,"sourceType":"competition"}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import libraries\nimport cv2\nimport numpy as np\nimport os\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-25T11:33:27.690496Z","iopub.execute_input":"2025-03-25T11:33:27.690774Z","iopub.status.idle":"2025-03-25T11:33:34.585669Z","shell.execute_reply.started":"2025-03-25T11:33:27.690743Z","shell.execute_reply":"2025-03-25T11:33:34.584991Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Define a Deep Custom 3D CNN Model\n\nThis model consists of three convolutional blocks, each with two 3D convolution layers, batch normalization, ReLU activation, and pooling. A dropout layer is applied before the final fully-connected layer to output a crash probability.\n","metadata":{}},{"cell_type":"code","source":"# Define the deep custom 3D CNN model\nclass DeepCrashNN(nn.Module):\n    def __init__(self):\n        super(DeepCrashNN, self).__init__()\n        # Block 1: 3D Conv + BatchNorm + ReLU, spatial pooling only (temporal resolution remains)\n        self.block1 = nn.Sequential(\n            nn.Conv3d(in_channels=3, out_channels=16, kernel_size=3, padding=1),\n            nn.BatchNorm3d(16),\n            nn.ReLU(),\n            nn.Conv3d(in_channels=16, out_channels=16, kernel_size=3, padding=1),\n            nn.BatchNorm3d(16),\n            nn.ReLU(),\n            nn.MaxPool3d(kernel_size=(1, 2, 2))  # Only spatial pooling: reduces H & W by 2\n        )\n        # Block 2: Increase channels to 32\n        self.block2 = nn.Sequential(\n            nn.Conv3d(in_channels=16, out_channels=32, kernel_size=3, padding=1),\n            nn.BatchNorm3d(32),\n            nn.ReLU(),\n            nn.Conv3d(in_channels=32, out_channels=32, kernel_size=3, padding=1),\n            nn.BatchNorm3d(32),\n            nn.ReLU(),\n            nn.MaxPool3d(kernel_size=(1, 2, 2))\n        )\n        # Block 3: Increase channels to 64 with global pooling at the end\n        self.block3 = nn.Sequential(\n            nn.Conv3d(in_channels=32, out_channels=64, kernel_size=3, padding=1),\n            nn.BatchNorm3d(64),\n            nn.ReLU(),\n            nn.Conv3d(in_channels=64, out_channels=64, kernel_size=3, padding=1),\n            nn.BatchNorm3d(64),\n            nn.ReLU(),\n            nn.AdaptiveAvgPool3d(1)  # Global average pooling over (T, H, W)\n        )\n        self.dropout = nn.Dropout(p=0.5)\n        self.fc = nn.Linear(64, 1)\n    \n    def forward(self, x):\n        x = self.block1(x)  # (B, 16, T, H/2, W/2)\n        x = self.block2(x)  # (B, 32, T, H/4, W/4)\n        x = self.block3(x)  # (B, 64, 1, 1, 1)\n        x = x.view(x.size(0), -1)  # Flatten to (B, 64)\n        x = self.dropout(x)\n        x = self.fc(x)             # (B, 1)\n        return torch.sigmoid(x)    # Probability in [0,1]\n\n# Instantiate the model and set it to evaluation mode\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = DeepCrashNN().to(device)\nmodel.eval()\nprint(\"Deep custom 3D CNN model initialized and set to eval mode.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T11:34:42.888385Z","iopub.execute_input":"2025-03-25T11:34:42.888707Z","iopub.status.idle":"2025-03-25T11:34:43.296209Z","shell.execute_reply.started":"2025-03-25T11:34:42.888676Z","shell.execute_reply":"2025-03-25T11:34:43.295324Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load training and test CSV files\ntrain_df = pd.read_csv('/kaggle/input/nexar-collision-prediction/train.csv')\ntest_df = pd.read_csv('/kaggle/input/nexar-collision-prediction/test.csv')\nprint(\"Training data loaded. Number of training videos:\", len(train_df))\nprint(\"Test data loaded. Number of test videos:\", len(test_df))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T11:35:34.299497Z","iopub.execute_input":"2025-03-25T11:35:34.299844Z","iopub.status.idle":"2025-03-25T11:35:34.327055Z","shell.execute_reply.started":"2025-03-25T11:35:34.299818Z","shell.execute_reply":"2025-03-25T11:35:34.326471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define function to extract randomly sampled frames from a video\ndef extract_random_frames(video_path, num_frames=40, resize=(224, 224)):\n    cap = cv2.VideoCapture(video_path)\n    if not cap.isOpened():\n        print(f\"Error opening video file: {video_path}\")\n        return None\n\n    total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    if total_frames <= 0:\n        cap.release()\n        return None\n\n    # Randomly sample 'num_frames' unique indices and sort them to preserve temporal order\n    frame_indices = sorted(np.random.choice(total_frames, num_frames, replace=False))\n    \n    frames = []\n    current_frame = 0\n    next_idx = 0\n    ret = True\n    while ret and next_idx < len(frame_indices):\n        ret, frame = cap.read()\n        if not ret:\n            break\n        if current_frame == frame_indices[next_idx]:\n            frame = cv2.resize(frame, resize)\n            frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n            frame = frame.astype(np.float32) / 255.0  # Normalize to [0,1]\n            frames.append(frame)\n            next_idx += 1\n        current_frame += 1\n    cap.release()\n    \n    if len(frames) < num_frames:\n        while len(frames) < num_frames:\n            frames.append(frames[-1])\n    \n    # Convert frames to tensor with shape (B, C, T, H, W)\n    frames = np.stack(frames, axis=0)           # (T, H, W, C)\n    frames = np.transpose(frames, (3, 0, 1, 2))   # (C, T, H, W)\n    frames_tensor = torch.from_numpy(frames).unsqueeze(0)  # Add batch dimension\n    return frames_tensor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T11:38:14.456358Z","iopub.execute_input":"2025-03-25T11:38:14.456670Z","iopub.status.idle":"2025-03-25T11:38:14.464004Z","shell.execute_reply.started":"2025-03-25T11:38:14.456648Z","shell.execute_reply":"2025-03-25T11:38:14.462839Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Define the Inference Function\n\nThis function processes a video file by extracting frames, running the model, and returning a crash probability.","metadata":{}},{"cell_type":"code","source":"# Define prediction function\ndef predict_video(video_path):\n    frames_tensor = extract_random_frames(video_path, num_frames=16, resize=(224,224))\n    if frames_tensor is None:\n        return 0.0  # Default probability if video cannot be processed\n    frames_tensor = frames_tensor.to(device)\n    with torch.no_grad():\n        output = model(frames_tensor)  # Output shape: (B, 1)\n        prob = output.item()\n    return prob","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T11:38:40.511587Z","iopub.execute_input":"2025-03-25T11:38:40.511873Z","iopub.status.idle":"2025-03-25T11:38:40.516187Z","shell.execute_reply.started":"2025-03-25T11:38:40.511852Z","shell.execute_reply":"2025-03-25T11:38:40.515373Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Run Inference on Training Videos\n\nWe run inference on the training videos to validate our pipeline.","metadata":{}},{"cell_type":"code","source":"# Generate predictions for training videos\ntrain_predictions = []\n\nfor idx, row in train_df.iterrows():\n    # Convert video ID to integer and format with leading zeros (5 digits)\n    video_id = int(float(row['id']))\n    video_filename = f\"{video_id:05d}.mp4\"  # e.g., 01924.mp4\n    video_path = os.path.join(\"/kaggle/input/nexar-collision-prediction/train\", video_filename)\n    prob = predict_video(video_path)\n    train_predictions.append(prob)\n    if idx % 50 == 0:\n        print(f\"Processed {idx} training videos...\")\n\ntrain_df['predicted_score'] = train_predictions\nprint(\"Training predictions generated.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T11:39:38.535355Z","iopub.execute_input":"2025-03-25T11:39:38.535683Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Run Inference on Test Videos\n\nWe now run the inference function on the test videos (assumed to be in the \"test/\" folder) to generate final predictions.","metadata":{}},{"cell_type":"code","source":"# Generate predictions for test videos\ntest_predictions = []\n\nfor idx, row in test_df.iterrows():\n    video_id = int(float(row['id']))\n    video_filename = f\"{video_id:05d}.mp4\"\n    video_path = os.path.join(\"/kaggle/input/nexar-collision-prediction/test\", video_filename)\n    prob = predict_video(video_path)\n    test_predictions.append(prob)\n    if idx % 50 == 0:\n        print(f\"Processed {idx} test videos...\")\n\ntest_df['score'] = test_predictions\nprint(\"Test predictions generated.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save submission file\nsubmission = test_df[['id', 'score']]\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Submission file 'submission.csv' created successfully.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}