{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":92399,"databundleVersionId":11038207,"sourceType":"competition"}],"dockerImageVersionId":31011,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-18T21:15:49.920879Z","iopub.execute_input":"2025-04-18T21:15:49.921263Z","iopub.status.idle":"2025-04-18T21:15:49.926341Z","shell.execute_reply.started":"2025-04-18T21:15:49.921237Z","shell.execute_reply":"2025-04-18T21:15:49.925274Z"}},"outputs":[],"execution_count":3},{"cell_type":"markdown","source":"## RESNET ","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.models as models\nimport torchvision.transforms as transforms\nfrom PIL import Image\nimport cv2\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import precision_score, recall_score, f1_score\nfrom tqdm import tqdm\n\n# Constants\nFRAME_RATE = 30  # 30 fps\nWINDOW_SIZE = 5  # Number of frames to use for prediction\nPOSITIVE_WINDOW = 1.5  # Seconds before the event to consider as positive\n\n# Load the training data\ntrain_df = pd.read_csv('/kaggle/input/nexar-collision-prediction/train.csv')\nprint(\"Training data loaded:\", train_df.shape)\n\n# Fix the video path issue by ensuring proper formatting of video IDs\ndef get_video_path(video_id, is_train=True):\n    formatted_id = f\"{int(video_id):05d}\"\n    base_dir = '/kaggle/input/nexar-collision-prediction'\n    subfolder = 'train' if is_train else 'test'\n    return os.path.join(base_dir, subfolder, f\"{formatted_id}.mp4\")\n\n# Data preprocessing function to extract frames\ndef extract_frames(video_id, df_row, output_dir, is_train=True):\n    formatted_id = f\"{int(video_id):05d}\"\n    video_path = get_video_path(video_id, is_train)\n    target = df_row['target'] if 'target' in df_row else None\n    \n    os.makedirs(output_dir, exist_ok=True)\n    \n    cap = cv2.VideoCapture(video_path)\n    if not cap.isOpened():\n        print(f\"Error opening video: {video_path}\")\n        return []\n    \n    frame_count = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    fps = cap.get(cv2.CAP_PROP_FPS)\n    \n    event_frame = None\n    alert_frame = None\n    \n    if is_train and target == 1:\n        event_time = float(df_row['time_of_event'])\n        alert_time = float(df_row['time_of_alert'])\n        event_frame = int(event_time * fps)\n        alert_frame = int(alert_time * fps)\n    \n    frame_data = []\n    \n    for frame_idx in range(frame_count):\n        ret, frame = cap.read()\n        if not ret:\n            break\n        \n        label = 0\n        if is_train and target == 1:\n            if alert_frame <= frame_idx <= event_frame:\n                label = 1\n            elif alert_frame - int(POSITIVE_WINDOW * fps) <= frame_idx < alert_frame:\n                label = 1\n        \n        frame_path = os.path.join(output_dir, f\"{formatted_id}_{frame_idx:05d}.jpg\")\n        cv2.imwrite(frame_path, frame)\n        \n        frame_data.append({\n            'frame_path': frame_path,\n            'video_id': formatted_id,\n            'frame_idx': frame_idx,\n            'label': label if is_train else None\n        })\n    \n    cap.release()\n    return frame_data\n\n# Process all videos and create a dataframe of frames\ndef process_videos(df, output_dir, is_train=True):\n    all_frames = []\n    \n    for _, row in tqdm(df.iterrows(), total=len(df)):\n        video_id = row['id']\n        frame_data = extract_frames(video_id, row, output_dir, is_train)\n        all_frames.extend(frame_data)\n    \n    return pd.DataFrame(all_frames)\n\n# Process a subset of videos for faster execution\nsample_train_df = train_df.sample(10, random_state=42) if len(train_df) > 10 else train_df\ntrain_frames_df = process_videos(sample_train_df, '/kaggle/working/train_frames', is_train=True)\n\n# Check if we have any frames before proceeding\nif len(train_frames_df) == 0:\n    print(\"No frames were extracted. Creating dummy dataset...\")\n    dummy_data = []\n    for i in range(100):\n        dummy_data.append({\n            'frame_path': f'/kaggle/working/dummy_{i:05d}.jpg',\n            'video_id': f\"{i % 5:05d}\",\n            'frame_idx': i,\n            'label': i % 2\n        })\n    train_frames_df = pd.DataFrame(dummy_data)\n    os.makedirs('/kaggle/working/dummy_images', exist_ok=True)\n    for i in range(100):\n        img = np.ones((224, 224, 3), dtype=np.uint8) * 128\n        cv2.imwrite(f'/kaggle/working/dummy_{i:05d}.jpg', img)\n\ntrain_frames_df.to_csv('/kaggle/working/train_frames.csv', index=False)\nprint(f\"Processed {len(train_frames_df)} frames from {len(sample_train_df)} videos\")\n\n# Custom Dataset class\nclass NexarDataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df = df\n        self.transform = transform\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img_path = row['frame_path']\n        label = row['label']\n        \n        try:\n            image = Image.open(img_path).convert('RGB')\n            if self.transform:\n                image = self.transform(image)\n            return image, torch.tensor(label, dtype=torch.float32)\n        except Exception as e:\n            print(f\"Error loading image {img_path}: {e}\")\n            placeholder = torch.zeros((3, 224, 224))\n            return placeholder, torch.tensor(0, dtype=torch.float32)\n\n# Define transforms\ntrain_transform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomRotation(10),\n    transforms.ColorJitter(brightness=0.2, contrast=0.2),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\nval_transform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\n# Split data into train, validation, and test sets\ntrain_val_frames, test_frames = train_test_split(train_frames_df, test_size=0.1, random_state=42)\ntrain_frames, val_frames = train_test_split(train_val_frames, test_size=0.2, random_state=42)\n\n# Create datasets\ntrain_dataset = NexarDataset(train_frames, transform=train_transform)\nval_dataset = NexarDataset(val_frames, transform=val_transform)\ntest_dataset = NexarDataset(test_frames, transform=val_transform)\n\n# Create data loaders\ntrain_loader = DataLoader(train_dataset, batch_size=16, shuffle=True, num_workers=2)\nval_loader = DataLoader(val_dataset, batch_size=16, shuffle=False, num_workers=2)\ntest_loader = DataLoader(test_dataset, batch_size=16, shuffle=False, num_workers=2)\n\n# Initialize ResNet-50 model\ndef get_model():\n    model = models.resnet50(pretrained=True)\n    num_ftrs = model.fc.in_features\n    model.fc = nn.Sequential(\n        nn.Linear(num_ftrs, 256),\n        nn.ReLU(),\n        nn.Dropout(0.3),\n        nn.Linear(256, 1),\n        nn.Sigmoid()\n    )\n    return model\n\nmodel = get_model()\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\nmodel = model.to(device)\n\n# Define loss function and optimizer\ncriterion = nn.BCELoss()\noptimizer = optim.Adam(model.parameters(), lr=0.0001)\nscheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=3, verbose=True)\n\n# Training function with additional metrics\ndef train_model(model, train_loader, val_loader, criterion, optimizer, scheduler, num_epochs=3):\n    best_val_loss = float('inf')\n    \n    for epoch in range(num_epochs):\n        # Training phase\n        model.train()\n        running_loss = 0.0\n        train_preds = []\n        train_labels = []\n        \n        for inputs, labels in tqdm(train_loader, desc=f'Epoch {epoch+1}/{num_epochs} - Training'):\n            inputs, labels = inputs.to(device), labels.to(device)\n            \n            optimizer.zero_grad()\n            outputs = model(inputs).squeeze()\n            \n            if outputs.ndim == 0:\n                outputs = outputs.unsqueeze(0)\n                \n            loss = criterion(outputs, labels)\n            loss.backward()\n            optimizer.step()\n            \n            running_loss += loss.item() * inputs.size(0)\n            predicted = (outputs > 0.5).float()\n            train_preds.extend(predicted.cpu().numpy())\n            train_labels.extend(labels.cpu().numpy())\n        \n        epoch_loss = running_loss / len(train_loader.dataset)\n        epoch_acc = np.mean(np.array(train_preds) == np.array(train_labels))\n        epoch_precision = precision_score(train_labels, train_preds, zero_division=0)\n        epoch_recall = recall_score(train_labels, train_preds, zero_division=0)\n        epoch_f1 = f1_score(train_labels, train_preds, zero_division=0)\n        \n        # Validation phase\n        model.eval()\n        val_loss = 0.0\n        val_preds = []\n        val_labels = []\n        \n        with torch.no_grad():\n            for inputs, labels in tqdm(val_loader, desc=f'Epoch {epoch+1}/{num_epochs} - Validation'):\n                inputs, labels = inputs.to(device), labels.to(device)\n                outputs = model(inputs).squeeze()\n                \n                if outputs.ndim == 0:\n                    outputs = outputs.unsqueeze(0)\n                    \n                loss = criterion(outputs, labels)\n                val_loss += loss.item() * inputs.size(0)\n                predicted = (outputs > 0.5).float()\n                val_preds.extend(predicted.cpu().numpy())\n                val_labels.extend(labels.cpu().numpy())\n        \n        val_epoch_loss = val_loss / len(val_loader.dataset)\n        val_epoch_acc = np.mean(np.array(val_preds) == np.array(val_labels))\n        val_epoch_precision = precision_score(val_labels, val_preds, zero_division=0)\n        val_epoch_recall = recall_score(val_labels, val_preds, zero_division=0)\n        val_epoch_f1 = f1_score(val_labels, val_preds, zero_division=0)\n        \n        scheduler.step(val_epoch_loss)\n        \n        print(f'Epoch {epoch+1}/{num_epochs}:')\n        print(f'Train Loss: {epoch_loss:.4f}, Acc: {epoch_acc:.4f}, Precision: {epoch_precision:.4f}, Recall: {epoch_recall:.4f}, F1: {epoch_f1:.4f}')\n        print(f'Val Loss: {val_epoch_loss:.4f}, Acc: {val_epoch_acc:.4f}, Precision: {val_epoch_precision:.4f}, Recall: {val_epoch_recall:.4f}, F1: {val_epoch_f1:.4f}')\n        \n        if val_epoch_loss < best_val_loss:\n            best_val_loss = val_epoch_loss\n            torch.save(model.state_dict(), '/kaggle/working/best_model.pth')\n            print(\"Saved best model!\")\n    \n    return model\n\n# Evaluate model on test set\ndef evaluate_model(model, test_loader, criterion):\n    model.eval()\n    test_loss = 0.0\n    test_preds = []\n    test_labels = []\n    \n    with torch.no_grad():\n        for inputs, labels in tqdm(test_loader, desc='Evaluating on Test Set'):\n            inputs, labels = inputs.to(device), labels.to(device)\n            outputs = model(inputs).squeeze()\n            \n            if outputs.ndim == 0:\n                outputs = outputs.unsqueeze(0)\n                \n            loss = criterion(outputs, labels)\n            test_loss += loss.item() * inputs.size(0)\n            predicted = (outputs > 0.5).float()\n            test_preds.extend(predicted.cpu().numpy())\n            test_labels.extend(labels.cpu().numpy())\n    \n    test_loss = test_loss / len(test_loader.dataset)\n    test_acc = np.mean(np.array(test_preds) == np.array(test_labels))\n    test_precision = precision_score(test_labels, test_preds, zero_division=0)\n    test_recall = recall_score(test_labels, test_preds, zero_division=0)\n    test_f1 = f1_score(test_labels, test_preds, zero_division=0)\n    \n    print(f'Test Loss: {test_loss:.4f}, Acc: {test_acc:.4f}, Precision: {test_precision:.4f}, Recall: {test_recall:.4f}, F1: {test_f1:.4f}')\n\n# Train the model\ntrained_model = train_model(model, train_loader, val_loader, criterion, optimizer, scheduler, num_epochs=3)\n\n# Evaluate on test set\nevaluate_model(trained_model, test_loader, criterion)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-18T21:59:13.361226Z","iopub.execute_input":"2025-04-18T21:59:13.361806Z","iopub.status.idle":"2025-04-18T22:04:57.742862Z","shell.execute_reply.started":"2025-04-18T21:59:13.36178Z","shell.execute_reply":"2025-04-18T22:04:57.741868Z"}},"outputs":[{"name":"stdout","text":"Training data loaded: (1500, 4)\n","output_type":"stream"},{"name":"stderr","text":"100%|██████████| 10/10 [00:55<00:00,  5.54s/it]\n/usr/local/lib/python3.11/dist-packages/torchvision/models/_utils.py:208: UserWarning: The parameter 'pretrained' is deprecated since 0.13 and may be removed in the future, please use 'weights' instead.\n  warnings.warn(\n/usr/local/lib/python3.11/dist-packages/torchvision/models/_utils.py:223: UserWarning: Arguments other than a weight enum or `None` for 'weights' are deprecated since 0.13 and may be removed in the future. The current behavior is equivalent to passing `weights=ResNet50_Weights.IMAGENET1K_V1`. You can also use `weights=ResNet50_Weights.DEFAULT` to get the most up-to-date weights.\n  warnings.warn(msg)\n","output_type":"stream"},{"name":"stdout","text":"Processed 10534 frames from 10 videos\n","output_type":"stream"},{"name":"stderr","text":"/usr/local/lib/python3.11/dist-packages/torch/optim/lr_scheduler.py:62: UserWarning: The verbose parameter is deprecated. Please use get_last_lr() to access the learning rate.\n  warnings.warn(\n","output_type":"stream"},{"name":"stdout","text":"Using device: cuda\n","output_type":"stream"},{"name":"stderr","text":"Epoch 1/3 - Training: 100%|██████████| 474/474 [01:21<00:00,  5.79it/s]\nEpoch 1/3 - Validation: 100%|██████████| 119/119 [00:11<00:00, 10.01it/s]\n","output_type":"stream"},{"name":"stdout","text":"Epoch 1/3:\nTrain Loss: 0.0696, Acc: 0.9744, Precision: 0.8201, Recall: 0.4921, F1: 0.6151\nVal Loss: 0.0173, Acc: 0.9931, Precision: 0.9726, Recall: 0.8659, F1: 0.9161\nSaved best model!\n","output_type":"stream"},{"name":"stderr","text":"Epoch 2/3 - Training: 100%|██████████| 474/474 [01:21<00:00,  5.79it/s]\nEpoch 2/3 - Validation: 100%|██████████| 119/119 [00:11<00:00, 10.00it/s]\n","output_type":"stream"},{"name":"stdout","text":"Epoch 2/3:\nTrain Loss: 0.0287, Acc: 0.9898, Precision: 0.8650, Recall: 0.8952, F1: 0.8799\nVal Loss: 0.0222, Acc: 0.9942, Precision: 1.0000, Recall: 0.8659, F1: 0.9281\n","output_type":"stream"},{"name":"stderr","text":"Epoch 3/3 - Training: 100%|██████████| 474/474 [01:21<00:00,  5.80it/s]\nEpoch 3/3 - Validation: 100%|██████████| 119/119 [00:11<00:00, 10.18it/s]\n","output_type":"stream"},{"name":"stdout","text":"Epoch 3/3:\nTrain Loss: 0.0247, Acc: 0.9924, Precision: 0.9186, Recall: 0.8952, F1: 0.9068\nVal Loss: 0.0268, Acc: 0.9916, Precision: 0.8837, Recall: 0.9268, F1: 0.9048\n","output_type":"stream"},{"name":"stderr","text":"Evaluating on Test Set: 100%|██████████| 66/66 [00:07<00:00,  9.39it/s]","output_type":"stream"},{"name":"stdout","text":"Test Loss: 0.0309, Acc: 0.9915, Precision: 0.8333, Recall: 0.9756, F1: 0.8989\n","output_type":"stream"},{"name":"stderr","text":"\n","output_type":"stream"}],"execution_count":11},{"cell_type":"code","source":"\nimport os\nimport pandas as pd\nimport numpy as np\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\nfrom PIL import Image\nfrom tqdm import tqdm\nimport cv2\nimport shutil\n\n# Modified NexarDataset class with robust error handling\nclass NexarDataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df = df\n        self.transform = transform\n        # Filter out rows with invalid frame paths\n        self.df = self.df[self.df['frame_path'].apply(self._is_valid_file)]\n        print(f\"Filtered dataset to {len(self.df)} valid frames\")\n        \n    def _is_valid_file(self, path):\n        return os.path.isfile(path) and os.access(path, os.R_OK)\n    \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img_path = row['frame_path']\n        label = row['label'] if row['label'] is not None else 0.0\n        \n        try:\n            if not self._is_valid_file(img_path):\n                raise FileNotFoundError(f\"Image file not found or inaccessible: {img_path}\")\n                \n            image = Image.open(img_path).convert('RGB')\n            \n            if self.transform:\n                image = self.transform(image)\n                \n            return image, torch.tensor(label, dtype=torch.float32)\n        except Exception as e:\n            print(f\"Error loading image {img_path}: {e}\")\n            # Return a placeholder image and label\n            placeholder = torch.zeros((3, 224, 224))\n            return placeholder, torch.tensor(label, dtype=torch.float32)\n\n# Modified test data processing\n# Load test data\ntest_df = pd.read_csv('/kaggle/input/nexar-collision-prediction/test.csv')\nprint(\"Test data loaded:\", test_df.shape)\n\n# Process a subset of test videos\nsample_test_df = test_df.sample(2, random_state=42) if len(test_df) > 2 else test_df\ntest_frames_dir = '/kaggle/working/test_frames'\n\n# Clear the test frames directory to avoid conflicts\nif os.path.exists(test_frames_dir):\n    shutil.rmtree(test_frames_dir)\nos.makedirs(test_frames_dir, exist_ok=True)\n\n# Process videos with additional debugging\ndef process_videos(df, output_dir, is_train=False):\n    all_frames = []\n    \n    for _, row in tqdm(df.iterrows(), total=len(df), desc=\"Processing test videos\"):\n        video_id = row['id']\n        frame_data = extract_frames(video_id, row, output_dir, is_train)\n        # Validate saved frames\n        valid_frames = []\n        for frame in frame_data:\n            if os.path.isfile(frame['frame_path']) and os.access(frame['frame_path'], os.R_OK):\n                valid_frames.append(frame)\n            else:\n                print(f\"Invalid frame file: {frame['frame_path']}\")\n        all_frames.extend(valid_frames)\n    \n    return pd.DataFrame(all_frames)\n\ntest_frames_df = process_videos(sample_test_df, test_frames_dir, is_train=False)\n\n# Check if we have any test frames\nif len(test_frames_df) == 0:\n    print(\"No test frames were extracted. Creating dummy test data...\")\n    dummy_test_data = []\n    os.makedirs('/kaggle/working/dummy_test_images', exist_ok=True)\n    for i in range(50):\n        frame_path = f'/kaggle/working/dummy_test_images/dummy_test_{i:05d}.jpg'\n        img = np.ones((224, 224, 3), dtype=np.uint8) * 128\n        cv2.imwrite(frame_path, img)\n        dummy_test_data.append({\n            'frame_path': frame_path,\n            'video_id': f\"{i % 5:05d}\",\n            'frame_idx': i,\n            'label': None\n        })\n    test_frames_df = pd.DataFrame(dummy_test_data)\n\ntest_frames_df.to_csv('/kaggle/working/test_frames.csv', index=False)\nprint(f\"Processed {len(test_frames_df)} test frames from {len(sample_test_df)} videos\")\n\n# Create test dataset and loader\ntest_dataset = NexarDataset(test_frames_df, transform=val_transform)\nif len(test_dataset) == 0:\n    raise ValueError(\"Test dataset is empty after filtering invalid frames\")\n\ntest_loader = DataLoader(test_dataset, batch_size=16, shuffle=False, num_workers=0)  # Set num_workers to 0\n\n# Load best model for inference\nmodel.load_state_dict(torch.load('/kaggle/working/best_model.pth', weights_only=True))  # Set weights_only=True\nmodel.eval()\n\n# Make predictions on test set with error handling\npredictions = []\nvideo_ids = []\n\nwith torch.no_grad():\n    for i, (inputs, _) in enumerate(tqdm(test_loader, desc='Making predictions')):\n        try:\n            inputs = inputs.to(device)\n            outputs = model(inputs).squeeze().cpu().numpy()\n            \n            batch_indices = list(range(i*test_loader.batch_size, \n                                     min((i+1)*test_loader.batch_size, len(test_dataset))))\n            \n            actual_batch_size = min(test_loader.batch_size, len(test_dataset) - i*test_loader.batch_size)\n            batch_indices = batch_indices[:actual_batch_size]\n            \n            batch_video_ids = [test_frames_df.iloc[idx]['video_id'] for idx in batch_indices]\n            \n            if isinstance(outputs, np.float32):\n                outputs = np.array([outputs])\n                \n            if len(outputs) > len(batch_indices):\n                outputs = outputs[:len(batch_indices)]\n                \n            predictions.extend(outputs)\n            video_ids.extend(batch_video_ids)\n        except Exception as e:\n            print(f\"Error processing batch {i}: {e}\")\n            continue\n\nif not predictions:\n    raise ValueError(\"No predictions were generated. Check test data and model inference.\")\n\n# Create a dataframe with predictions for each frame\nprediction_df = pd.DataFrame({\n    'video_id': video_ids,\n    'prediction': predictions\n})\n\n# Aggregate predictions by video\nvideo_predictions = prediction_df.groupby('video_id')['prediction'].max().reset_index()\n\n# Create submission file\nsubmission_df = pd.DataFrame({\n    'id': video_predictions['video_id'],\n    'target': video_predictions['prediction']\n})\n\nsubmission_df.to_csv('/kaggle/working/submission.csv', index=False)\nprint(\"Submission file created!\")\nprint(submission_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-18T22:12:20.305023Z","iopub.execute_input":"2025-04-18T22:12:20.305554Z","iopub.status.idle":"2025-04-18T22:12:31.630859Z","shell.execute_reply.started":"2025-04-18T22:12:20.30553Z","shell.execute_reply":"2025-04-18T22:12:31.630087Z"}},"outputs":[{"name":"stdout","text":"Test data loaded: (1344, 1)\n","output_type":"stream"},{"name":"stderr","text":"Processing test videos: 100%|██████████| 2/2 [00:02<00:00,  1.46s/it]\n","output_type":"stream"},{"name":"stdout","text":"Processed 596 test frames from 2 videos\nFiltered dataset to 596 valid frames\n","output_type":"stream"},{"name":"stderr","text":"Making predictions: 100%|██████████| 38/38 [00:08<00:00,  4.63it/s]","output_type":"stream"},{"name":"stdout","text":"Submission file created!\n      id    target\n0  00531  0.012314\n1  01246  0.511701\n","output_type":"stream"},{"name":"stderr","text":"\n","output_type":"stream"}],"execution_count":13},{"cell_type":"markdown","source":"## ViT B16","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as transforms\nfrom PIL import Image\nimport cv2\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import precision_score, recall_score, f1_score\nfrom tqdm import tqdm\nfrom torchvision.models import vit_b_16, ViT_B_16_Weights\n\n# Constants\nFRAME_RATE = 30  # 30 fps\nWINDOW_SIZE = 5  # Number of frames to use for prediction\nPOSITIVE_WINDOW = 1.5  # Seconds before the event to consider as positive\n\n# Load the training data\ntrain_df = pd.read_csv('/kaggle/input/nexar-collision-prediction/train.csv')\nprint(\"Training data loaded:\", train_df.shape)\n\n# Fix the video path issue by ensuring proper formatting of video IDs\ndef get_video_path(video_id, is_train=True):\n    formatted_id = f\"{int(video_id):05d}\"\n    base_dir = '/kaggle/input/nexar-collision-prediction'\n    subfolder = 'train' if is_train else 'test'\n    return os.path.join(base_dir, subfolder, f\"{formatted_id}.mp4\")\n\n# Data preprocessing function to extract frames\ndef extract_frames(video_id, df_row, output_dir, is_train=True):\n    formatted_id = f\"{int(video_id):05d}\"\n    video_path = get_video_path(video_id, is_train)\n    target = df_row['target'] if 'target' in df_row else None\n    \n    os.makedirs(output_dir, exist_ok=True)\n    \n    cap = cv2.VideoCapture(video_path)\n    if not cap.isOpened():\n        print(f\"Error opening video: {video_path}\")\n        return []\n    \n    frame_count = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    fps = cap.get(cv2.CAP_PROP_FPS)\n    \n    event_frame = None\n    alert_frame = None\n    \n    if is_train and target == 1:\n        event_time = float(df_row['time_of_event'])\n        alert_time = float(df_row['time_of_alert'])\n        event_frame = int(event_time * fps)\n        alert_frame = int(alert_time * fps)\n    \n    frame_data = []\n    \n    for frame_idx in range(frame_count):\n        ret, frame = cap.read()\n        if not ret:\n            break\n        \n        label = 0\n        if is_train and target == 1:\n            if alert_frame <= frame_idx <= event_frame:\n                label = 1\n            elif alert_frame - int(POSITIVE_WINDOW * fps) <= frame_idx < alert_frame:\n                label = 1\n        \n        frame_path = os.path.join(output_dir, f\"{formatted_id}_{frame_idx:05d}.jpg\")\n        cv2.imwrite(frame_path, frame)\n        \n        frame_data.append({\n            'frame_path': frame_path,\n            'video_id': formatted_id,\n            'frame_idx': frame_idx,\n            'label': label if is_train else None\n        })\n    \n    cap.release()\n    return frame_data\n\n# Process all videos and create a dataframe of frames\ndef process_videos(df, output_dir, is_train=True):\n    all_frames = []\n    \n    for _, row in tqdm(df.iterrows(), total=len(df)):\n        video_id = row['id']\n        frame_data = extract_frames(video_id, row, output_dir, is_train)\n        all_frames.extend(frame_data)\n    \n    return pd.DataFrame(all_frames)\n\n# Process a subset of videos for faster execution\nsample_train_df = train_df.sample(10, random_state=42) if len(train_df) > 10 else train_df\ntrain_frames_df = process_videos(sample_train_df, '/kaggle/working/train_frames', is_train=True)\n\n# Check if we have any frames before proceeding\nif len(train_frames_df) == 0:\n    print(\"No frames were extracted. Creating dummy dataset...\")\n    dummy_data = []\n    for i in range(100):\n        dummy_data.append({\n            'frame_path': f'/kaggle/working/dummy_{i:05d}.jpg',\n            'video_id': f\"{i % 5:05d}\",\n            'frame_idx': i,\n            'label': i % 2\n        })\n    train_frames_df = pd.DataFrame(dummy_data)\n    os.makedirs('/kaggle/working/dummy_images', exist_ok=True)\n    for i in range(100):\n        img = np.ones((224, 224, 3), dtype=np.uint8) * 128\n        cv2.imwrite(f'/kaggle/working/dummy_{i:05d}.jpg', img)\n\ntrain_frames_df.to_csv('/kaggle/working/train_frames.csv', index=False)\nprint(f\"Processed {len(train_frames_df)} frames from {len(sample_train_df)} videos\")\n\n# Custom Dataset class\nclass NexarDataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df = df\n        self.transform = transform\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img_path = row['frame_path']\n        label = row['label']\n        \n        try:\n            image = Image.open(img_path).convert('RGB')\n            if self.transform:\n                image = self.transform(image)\n            return image, torch.tensor(label, dtype=torch.float32)\n        except Exception as e:\n            print(f\"Error loading image {img_path}: {e}\")\n            placeholder = torch.zeros((3, 224, 224))\n            return placeholder, torch.tensor(0, dtype=torch.float32)\n\n# Define transforms for ViT - Note that ViT requires a specific input size (224x224 for ViT-B/16)\ntrain_transform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomRotation(10),\n    transforms.ColorJitter(brightness=0.2, contrast=0.2),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\nval_transform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\n# Split data into train, validation, and test sets\ntrain_val_frames, test_frames = train_test_split(train_frames_df, test_size=0.1, random_state=42)\ntrain_frames, val_frames = train_test_split(train_val_frames, test_size=0.2, random_state=42)\n\n# Create datasets\ntrain_dataset = NexarDataset(train_frames, transform=train_transform)\nval_dataset = NexarDataset(val_frames, transform=val_transform)\ntest_dataset = NexarDataset(test_frames, transform=val_transform)\n\n# Create data loaders\ntrain_loader = DataLoader(train_dataset, batch_size=16, shuffle=True, num_workers=2)\nval_loader = DataLoader(val_dataset, batch_size=16, shuffle=False, num_workers=2)\ntest_loader = DataLoader(test_dataset, batch_size=16, shuffle=False, num_workers=2)\n\n# Initialize ViT model\ndef get_vit_model():\n    # Load pre-trained ViT model\n    model = vit_b_16(weights=ViT_B_16_Weights.IMAGENET1K_V1)\n    \n    # Modify the head for binary classification\n    num_features = model.heads.head.in_features\n    model.heads.head = nn.Sequential(\n        nn.Linear(num_features, 256),\n        nn.ReLU(),\n        nn.Dropout(0.3),\n        nn.Linear(256, 1),\n        nn.Sigmoid()\n    )\n    \n    return model\n\nmodel = get_vit_model()\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\nmodel = model.to(device)\n\n# Define loss function and optimizer\ncriterion = nn.BCELoss()\n# For ViT, we often use a lower learning rate\noptimizer = optim.AdamW(model.parameters(), lr=5e-5, weight_decay=0.01)\nscheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=3, verbose=True)\n\n# Training function with additional metrics\ndef train_model(model, train_loader, val_loader, criterion, optimizer, scheduler, num_epochs=3):\n    best_val_loss = float('inf')\n    \n    for epoch in range(num_epochs):\n        # Training phase\n        model.train()\n        running_loss = 0.0\n        train_preds = []\n        train_labels = []\n        \n        for inputs, labels in tqdm(train_loader, desc=f'Epoch {epoch+1}/{num_epochs} - Training'):\n            inputs, labels = inputs.to(device), labels.to(device)\n            \n            optimizer.zero_grad()\n            outputs = model(inputs).squeeze()\n            \n            if outputs.ndim == 0:\n                outputs = outputs.unsqueeze(0)\n                \n            loss = criterion(outputs, labels)\n            loss.backward()\n            optimizer.step()\n            \n            running_loss += loss.item() * inputs.size(0)\n            predicted = (outputs > 0.5).float()\n            train_preds.extend(predicted.cpu().numpy())\n            train_labels.extend(labels.cpu().numpy())\n        \n        epoch_loss = running_loss / len(train_loader.dataset)\n        epoch_acc = np.mean(np.array(train_preds) == np.array(train_labels))\n        epoch_precision = precision_score(train_labels, train_preds, zero_division=0)\n        epoch_recall = recall_score(train_labels, train_preds, zero_division=0)\n        epoch_f1 = f1_score(train_labels, train_preds, zero_division=0)\n        \n        # Validation phase\n        model.eval()\n        val_loss = 0.0\n        val_preds = []\n        val_labels = []\n        \n        with torch.no_grad():\n            for inputs, labels in tqdm(val_loader, desc=f'Epoch {epoch+1}/{num_epochs} - Validation'):\n                inputs, labels = inputs.to(device), labels.to(device)\n                outputs = model(inputs).squeeze()\n                \n                if outputs.ndim == 0:\n                    outputs = outputs.unsqueeze(0)\n                    \n                loss = criterion(outputs, labels)\n                val_loss += loss.item() * inputs.size(0)\n                predicted = (outputs > 0.5).float()\n                val_preds.extend(predicted.cpu().numpy())\n                val_labels.extend(labels.cpu().numpy())\n        \n        val_epoch_loss = val_loss / len(val_loader.dataset)\n        val_epoch_acc = np.mean(np.array(val_preds) == np.array(val_labels))\n        val_epoch_precision = precision_score(val_labels, val_preds, zero_division=0)\n        val_epoch_recall = recall_score(val_labels, val_preds, zero_division=0)\n        val_epoch_f1 = f1_score(val_labels, val_preds, zero_division=0)\n        \n        scheduler.step(val_epoch_loss)\n        \n        print(f'Epoch {epoch+1}/{num_epochs}:')\n        print(f'Train Loss: {epoch_loss:.4f}, Acc: {epoch_acc:.4f}, Precision: {epoch_precision:.4f}, Recall: {epoch_recall:.4f}, F1: {epoch_f1:.4f}')\n        print(f'Val Loss: {val_epoch_loss:.4f}, Acc: {val_epoch_acc:.4f}, Precision: {val_epoch_precision:.4f}, Recall: {val_epoch_recall:.4f}, F1: {val_epoch_f1:.4f}')\n        \n        if val_epoch_loss < best_val_loss:\n            best_val_loss = val_epoch_loss\n            torch.save(model.state_dict(), '/kaggle/working/best_vit_model.pth')\n            print(\"Saved best model!\")\n    \n    return model\n\n# Evaluate model on test set\ndef evaluate_model(model, test_loader, criterion):\n    model.eval()\n    test_loss = 0.0\n    test_preds = []\n    test_labels = []\n    \n    with torch.no_grad():\n        for inputs, labels in tqdm(test_loader, desc='Evaluating on Test Set'):\n            inputs, labels = inputs.to(device), labels.to(device)\n            outputs = model(inputs).squeeze()\n            \n            if outputs.ndim == 0:\n                outputs = outputs.unsqueeze(0)\n                \n            loss = criterion(outputs, labels)\n            test_loss += loss.item() * inputs.size(0)\n            predicted = (outputs > 0.5).float()\n            test_preds.extend(predicted.cpu().numpy())\n            test_labels.extend(labels.cpu().numpy())\n    \n    test_loss = test_loss / len(test_loader.dataset)\n    test_acc = np.mean(np.array(test_preds) == np.array(test_labels))\n    test_precision = precision_score(test_labels, test_preds, zero_division=0)\n    test_recall = recall_score(test_labels, test_preds, zero_division=0)\n    test_f1 = f1_score(test_labels, test_preds, zero_division=0)\n    \n    print(f'Test Loss: {test_loss:.4f}, Acc: {test_acc:.4f}, Precision: {test_precision:.4f}, Recall: {test_recall:.4f}, F1: {test_f1:.4f}')\n\n# Train the model\ntrained_model = train_model(model, train_loader, val_loader, criterion, optimizer, scheduler, num_epochs=3)\n\n# Evaluate on test set\nevaluate_model(trained_model, test_loader, criterion)\n\n# Modified test data processing for inference\nimport shutil\n\n# Modified NexarDataset class with robust error handling\nclass NexarDataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df = df\n        self.transform = transform\n        # Filter out rows with invalid frame paths\n        self.df = self.df[self.df['frame_path'].apply(self._is_valid_file)]\n        print(f\"Filtered dataset to {len(self.df)} valid frames\")\n        \n    def _is_valid_file(self, path):\n        return os.path.isfile(path) and os.access(path, os.R_OK)\n    \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img_path = row['frame_path']\n        label = row['label'] if row['label'] is not None else 0.0\n        \n        try:\n            if not self._is_valid_file(img_path):\n                raise FileNotFoundError(f\"Image file not found or inaccessible: {img_path}\")\n                \n            image = Image.open(img_path).convert('RGB')\n            \n            if self.transform:\n                image = self.transform(image)\n                \n            return image, torch.tensor(label, dtype=torch.float32)\n        except Exception as e:\n            print(f\"Error loading image {img_path}: {e}\")\n            # Return a placeholder image and label\n            placeholder = torch.zeros((3, 224, 224))\n            return placeholder, torch.tensor(label, dtype=torch.float32)\n\n# Load test data\ntest_df = pd.read_csv('/kaggle/input/nexar-collision-prediction/test.csv')\nprint(\"Test data loaded:\", test_df.shape)\n\n# Process a subset of test videos\nsample_test_df = test_df.sample(2, random_state=42) if len(test_df) > 2 else test_df\ntest_frames_dir = '/kaggle/working/test_frames_vit'\n\n# Clear the test frames directory to avoid conflicts\nif os.path.exists(test_frames_dir):\n    shutil.rmtree(test_frames_dir)\nos.makedirs(test_frames_dir, exist_ok=True)\n\n# Process videos with additional debugging\ndef process_videos(df, output_dir, is_train=False):\n    all_frames = []\n    \n    for _, row in tqdm(df.iterrows(), total=len(df), desc=\"Processing test videos\"):\n        video_id = row['id']\n        frame_data = extract_frames(video_id, row, output_dir, is_train)\n        # Validate saved frames\n        valid_frames = []\n        for frame in frame_data:\n            if os.path.isfile(frame['frame_path']) and os.access(frame['frame_path'], os.R_OK):\n                valid_frames.append(frame)\n            else:\n                print(f\"Invalid frame file: {frame['frame_path']}\")\n        all_frames.extend(valid_frames)\n    \n    return pd.DataFrame(all_frames)\n\ntest_frames_df = process_videos(sample_test_df, test_frames_dir, is_train=False)\n\n# Check if we have any test frames\nif len(test_frames_df) == 0:\n    print(\"No test frames were extracted. Creating dummy test data...\")\n    dummy_test_data = []\n    os.makedirs('/kaggle/working/dummy_test_images_vit', exist_ok=True)\n    for i in range(50):\n        frame_path = f'/kaggle/working/dummy_test_images_vit/dummy_test_{i:05d}.jpg'\n        img = np.ones((224, 224, 3), dtype=np.uint8) * 128\n        cv2.imwrite(frame_path, img)\n        dummy_test_data.append({\n            'frame_path': frame_path,\n            'video_id': f\"{i % 5:05d}\",\n            'frame_idx': i,\n            'label': None\n        })\n    test_frames_df = pd.DataFrame(dummy_test_data)\n\ntest_frames_df.to_csv('/kaggle/working/test_frames_vit.csv', index=False)\nprint(f\"Processed {len(test_frames_df)} test frames from {len(sample_test_df)} videos\")\n\n# Create test dataset and loader\ntest_dataset = NexarDataset(test_frames_df, transform=val_transform)\nif len(test_dataset) == 0:\n    raise ValueError(\"Test dataset is empty after filtering invalid frames\")\n\ntest_loader = DataLoader(test_dataset, batch_size=16, shuffle=False, num_workers=0)  # Set num_workers to 0\n\n# Load best model for inference\nmodel.load_state_dict(torch.load('/kaggle/working/best_vit_model.pth', weights_only=True))  # Set weights_only=True\nmodel.eval()\n\n# Make predictions on test set with error handling\npredictions = []\nvideo_ids = []\n\nwith torch.no_grad():\n    for i, (inputs, _) in enumerate(tqdm(test_loader, desc='Making predictions with ViT')):\n        try:\n            inputs = inputs.to(device)\n            outputs = model(inputs).squeeze().cpu().numpy()\n            \n            batch_indices = list(range(i*test_loader.batch_size, \n                                     min((i+1)*test_loader.batch_size, len(test_dataset))))\n            \n            actual_batch_size = min(test_loader.batch_size, len(test_dataset) - i*test_loader.batch_size)\n            batch_indices = batch_indices[:actual_batch_size]\n            \n            batch_video_ids = [test_frames_df.iloc[idx]['video_id'] for idx in batch_indices]\n            \n            if isinstance(outputs, np.float32):\n                outputs = np.array([outputs])\n                \n            if len(outputs) > len(batch_indices):\n                outputs = outputs[:len(batch_indices)]\n                \n            predictions.extend(outputs)\n            video_ids.extend(batch_video_ids)\n        except Exception as e:\n            print(f\"Error processing batch {i}: {e}\")\n            continue\n\nif not predictions:\n    raise ValueError(\"No predictions were generated. Check test data and model inference.\")\n\n# Create a dataframe with predictions for each frame\nprediction_df = pd.DataFrame({\n    'video_id': video_ids,\n    'prediction': predictions\n})\n\n# Aggregate predictions by video\nvideo_predictions = prediction_df.groupby('video_id')['prediction'].max().reset_index()\n\n# Create submission file\nsubmission_df = pd.DataFrame({\n    'id': video_predictions['video_id'],\n    'target': video_predictions['prediction']\n})\n\nsubmission_df.to_csv('/kaggle/working/vit_submission.csv', index=False)\nprint(\"ViT submission file created!\")\nprint(submission_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-18T22:19:21.439769Z","iopub.execute_input":"2025-04-18T22:19:21.440276Z","iopub.status.idle":"2025-04-18T22:35:22.700075Z","shell.execute_reply.started":"2025-04-18T22:19:21.440253Z","shell.execute_reply":"2025-04-18T22:35:22.699189Z"}},"outputs":[{"name":"stdout","text":"Training data loaded: (1500, 4)\n","output_type":"stream"},{"name":"stderr","text":"100%|██████████| 10/10 [00:54<00:00,  5.50s/it]\n","output_type":"stream"},{"name":"stdout","text":"Processed 10534 frames from 10 videos\n","output_type":"stream"},{"name":"stderr","text":"Downloading: \"https://download.pytorch.org/models/vit_b_16-c867db91.pth\" to /root/.cache/torch/hub/checkpoints/vit_b_16-c867db91.pth\n100%|██████████| 330M/330M [00:01<00:00, 208MB/s]  \n/usr/local/lib/python3.11/dist-packages/torch/optim/lr_scheduler.py:62: UserWarning: The verbose parameter is deprecated. Please use get_last_lr() to access the learning rate.\n  warnings.warn(\n","output_type":"stream"},{"name":"stdout","text":"Using device: cuda\n","output_type":"stream"},{"name":"stderr","text":"Epoch 1/3 - Training: 100%|██████████| 474/474 [04:30<00:00,  1.75it/s]\nEpoch 1/3 - Validation: 100%|██████████| 119/119 [00:21<00:00,  5.52it/s]\n","output_type":"stream"},{"name":"stdout","text":"Epoch 1/3:\nTrain Loss: 0.1379, Acc: 0.9595, Precision: 0.5541, Recall: 0.1302, F1: 0.2108\nVal Loss: 0.0897, Acc: 0.9652, Precision: 1.0000, Recall: 0.1951, F1: 0.3265\nSaved best model!\n","output_type":"stream"},{"name":"stderr","text":"Epoch 2/3 - Training: 100%|██████████| 474/474 [04:29<00:00,  1.76it/s]\nEpoch 2/3 - Validation: 100%|██████████| 119/119 [00:21<00:00,  5.50it/s]\n","output_type":"stream"},{"name":"stdout","text":"Epoch 2/3:\nTrain Loss: 0.0636, Acc: 0.9761, Precision: 0.7463, Recall: 0.6444, F1: 0.6917\nVal Loss: 0.0328, Acc: 0.9910, Precision: 0.8421, Recall: 0.9756, F1: 0.9040\nSaved best model!\n","output_type":"stream"},{"name":"stderr","text":"Epoch 3/3 - Training: 100%|██████████| 474/474 [04:29<00:00,  1.76it/s]\nEpoch 3/3 - Validation: 100%|██████████| 119/119 [00:21<00:00,  5.48it/s]\n","output_type":"stream"},{"name":"stdout","text":"Epoch 3/3:\nTrain Loss: 0.0356, Acc: 0.9872, Precision: 0.8586, Recall: 0.8286, F1: 0.8433\nVal Loss: 0.0354, Acc: 0.9910, Precision: 0.9851, Recall: 0.8049, F1: 0.8859\n","output_type":"stream"},{"name":"stderr","text":"Evaluating on Test Set: 100%|██████████| 66/66 [00:12<00:00,  5.40it/s]\n","output_type":"stream"},{"name":"stdout","text":"Test Loss: 0.0330, Acc: 0.9924, Precision: 1.0000, Recall: 0.8049, F1: 0.8919\nTest data loaded: (1344, 1)\n","output_type":"stream"},{"name":"stderr","text":"Processing test videos: 100%|██████████| 2/2 [00:03<00:00,  1.50s/it]\n","output_type":"stream"},{"name":"stdout","text":"Processed 596 test frames from 2 videos\nFiltered dataset to 596 valid frames\n","output_type":"stream"},{"name":"stderr","text":"Making predictions with ViT: 100%|██████████| 38/38 [00:11<00:00,  3.23it/s]","output_type":"stream"},{"name":"stdout","text":"ViT submission file created!\n      id    target\n0  00531  0.077021\n1  01246  0.863352\n","output_type":"stream"},{"name":"stderr","text":"\n","output_type":"stream"}],"execution_count":14},{"cell_type":"markdown","source":"## EfficientNet B1","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as transforms\nfrom PIL import Image\nimport cv2\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import precision_score, recall_score, f1_score\nfrom tqdm import tqdm\nfrom torchvision.models import efficientnet_b1, EfficientNet_B1_Weights\n\n# Constants\nFRAME_RATE = 30  # 30 fps\nWINDOW_SIZE = 5  # Number of frames to use for prediction\nPOSITIVE_WINDOW = 1.5  # Seconds before the event to consider as positive\n\n# Load the training data\ntrain_df = pd.read_csv('/kaggle/input/nexar-collision-prediction/train.csv')\nprint(\"Training data loaded:\", train_df.shape)\n\n# Fix the video path issue by ensuring proper formatting of video IDs\ndef get_video_path(video_id, is_train=True):\n    formatted_id = f\"{int(video_id):05d}\"\n    base_dir = '/kaggle/input/nexar-collision-prediction'\n    subfolder = 'train' if is_train else 'test'\n    return os.path.join(base_dir, subfolder, f\"{formatted_id}.mp4\")\n\n# Data preprocessing function to extract frames\ndef extract_frames(video_id, df_row, output_dir, is_train=True):\n    formatted_id = f\"{int(video_id):05d}\"\n    video_path = get_video_path(video_id, is_train)\n    target = df_row['target'] if 'target' in df_row else None\n    \n    os.makedirs(output_dir, exist_ok=True)\n    \n    cap = cv2.VideoCapture(video_path)\n    if not cap.isOpened():\n        print(f\"Error opening video: {video_path}\")\n        return []\n    \n    frame_count = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    fps = cap.get(cv2.CAP_PROP_FPS)\n    \n    event_frame = None\n    alert_frame = None\n    \n    if is_train and target == 1:\n        event_time = float(df_row['time_of_event'])\n        alert_time = float(df_row['time_of_alert'])\n        event_frame = int(event_time * fps)\n        alert_frame = int(alert_time * fps)\n    \n    frame_data = []\n    \n    for frame_idx in range(frame_count):\n        ret, frame = cap.read()\n        if not ret:\n            break\n        \n        label = 0\n        if is_train and target == 1:\n            if alert_frame <= frame_idx <= event_frame:\n                label = 1\n            elif alert_frame - int(POSITIVE_WINDOW * fps) <= frame_idx < alert_frame:\n                label = 1\n        \n        frame_path = os.path.join(output_dir, f\"{formatted_id}_{frame_idx:05d}.jpg\")\n        cv2.imwrite(frame_path, frame)\n        \n        frame_data.append({\n            'frame_path': frame_path,\n            'video_id': formatted_id,\n            'frame_idx': frame_idx,\n            'label': label if is_train else None\n        })\n    \n    cap.release()\n    return frame_data\n\n# Process all videos and create a dataframe of frames\ndef process_videos(df, output_dir, is_train=True):\n    all_frames = []\n    \n    for _, row in tqdm(df.iterrows(), total=len(df)):\n        video_id = row['id']\n        frame_data = extract_frames(video_id, row, output_dir, is_train)\n        all_frames.extend(frame_data)\n    \n    return pd.DataFrame(all_frames)\n\n# Process a subset of videos for faster execution\nsample_train_df = train_df.sample(10, random_state=42) if len(train_df) > 10 else train_df\ntrain_frames_df = process_videos(sample_train_df, '/kaggle/working/train_frames_efficientnet', is_train=True)\n\n# Check if we have any frames before proceeding\nif len(train_frames_df) == 0:\n    print(\"No frames were extracted. Creating dummy dataset...\")\n    dummy_data = []\n    for i in range(100):\n        dummy_data.append({\n            'frame_path': f'/kaggle/working/dummy_efficientnet_{i:05d}.jpg',\n            'video_id': f\"{i % 5:05d}\",\n            'frame_idx': i,\n            'label': i % 2\n        })\n    train_frames_df = pd.DataFrame(dummy_data)\n    os.makedirs('/kaggle/working/dummy_images_efficientnet', exist_ok=True)\n    for i in range(100):\n        img = np.ones((224, 224, 3), dtype=np.uint8) * 128\n        cv2.imwrite(f'/kaggle/working/dummy_efficientnet_{i:05d}.jpg', img)\n\ntrain_frames_df.to_csv('/kaggle/working/train_frames_efficientnet.csv', index=False)\nprint(f\"Processed {len(train_frames_df)} frames from {len(sample_train_df)} videos\")\n\n# Custom Dataset class\nclass NexarDataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df = df\n        self.transform = transform\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img_path = row['frame_path']\n        label = row['label']\n        \n        try:\n            image = Image.open(img_path).convert('RGB')\n            if self.transform:\n                image = self.transform(image)\n            return image, torch.tensor(label, dtype=torch.float32)\n        except Exception as e:\n            print(f\"Error loading image {img_path}: {e}\")\n            placeholder = torch.zeros((3, 224, 224))\n            return placeholder, torch.tensor(0, dtype=torch.float32)\n\n# Define transforms for EfficientNet-B1\n# EfficientNet-B1 expects input images of size 240x240\ntrain_transform = transforms.Compose([\n    transforms.Resize((240, 240)),\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomRotation(10),\n    transforms.ColorJitter(brightness=0.2, contrast=0.2),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\nval_transform = transforms.Compose([\n    transforms.Resize((240, 240)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\n# Split data into train, validation, and test sets\ntrain_val_frames, test_frames = train_test_split(train_frames_df, test_size=0.1, random_state=42)\ntrain_frames, val_frames = train_test_split(train_val_frames, test_size=0.2, random_state=42)\n\n# Create datasets\ntrain_dataset = NexarDataset(train_frames, transform=train_transform)\nval_dataset = NexarDataset(val_frames, transform=val_transform)\ntest_dataset = NexarDataset(test_frames, transform=val_transform)\n\n# Create data loaders\ntrain_loader = DataLoader(train_dataset, batch_size=16, shuffle=True, num_workers=2)\nval_loader = DataLoader(val_dataset, batch_size=16, shuffle=False, num_workers=2)\ntest_loader = DataLoader(test_dataset, batch_size=16, shuffle=False, num_workers=2)\n\n# Initialize EfficientNet-B1 model\ndef get_efficientnet_model():\n    # Load pre-trained EfficientNet-B1 model\n    model = efficientnet_b1(weights=EfficientNet_B1_Weights.IMAGENET1K_V1)\n    \n    # Modify the classifier for binary classification\n    num_features = model.classifier[1].in_features\n    model.classifier = nn.Sequential(\n        nn.Dropout(p=0.2, inplace=True),\n        nn.Linear(num_features, 256),\n        nn.ReLU(),\n        nn.Dropout(0.3),\n        nn.Linear(256, 1),\n        nn.Sigmoid()\n    )\n    \n    return model\n\nmodel = get_efficientnet_model()\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\nmodel = model.to(device)\n\n# Define loss function and optimizer\ncriterion = nn.BCELoss()\n# For EfficientNet, we often use a similar learning rate as with ViT\noptimizer = optim.AdamW(model.parameters(), lr=5e-5, weight_decay=0.01)\nscheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=3, verbose=True)\n\n# Training function with additional metrics\ndef train_model(model, train_loader, val_loader, criterion, optimizer, scheduler, num_epochs=3):\n    best_val_loss = float('inf')\n    \n    for epoch in range(num_epochs):\n        # Training phase\n        model.train()\n        running_loss = 0.0\n        train_preds = []\n        train_labels = []\n        \n        for inputs, labels in tqdm(train_loader, desc=f'Epoch {epoch+1}/{num_epochs} - Training'):\n            inputs, labels = inputs.to(device), labels.to(device)\n            \n            optimizer.zero_grad()\n            outputs = model(inputs).squeeze()\n            \n            if outputs.ndim == 0:\n                outputs = outputs.unsqueeze(0)\n                \n            loss = criterion(outputs, labels)\n            loss.backward()\n            optimizer.step()\n            \n            running_loss += loss.item() * inputs.size(0)\n            predicted = (outputs > 0.5).float()\n            train_preds.extend(predicted.cpu().numpy())\n            train_labels.extend(labels.cpu().numpy())\n        \n        epoch_loss = running_loss / len(train_loader.dataset)\n        epoch_acc = np.mean(np.array(train_preds) == np.array(train_labels))\n        epoch_precision = precision_score(train_labels, train_preds, zero_division=0)\n        epoch_recall = recall_score(train_labels, train_preds, zero_division=0)\n        epoch_f1 = f1_score(train_labels, train_preds, zero_division=0)\n        \n        # Validation phase\n        model.eval()\n        val_loss = 0.0\n        val_preds = []\n        val_labels = []\n        \n        with torch.no_grad():\n            for inputs, labels in tqdm(val_loader, desc=f'Epoch {epoch+1}/{num_epochs} - Validation'):\n                inputs, labels = inputs.to(device), labels.to(device)\n                outputs = model(inputs).squeeze()\n                \n                if outputs.ndim == 0:\n                    outputs = outputs.unsqueeze(0)\n                    \n                loss = criterion(outputs, labels)\n                val_loss += loss.item() * inputs.size(0)\n                predicted = (outputs > 0.5).float()\n                val_preds.extend(predicted.cpu().numpy())\n                val_labels.extend(labels.cpu().numpy())\n        \n        val_epoch_loss = val_loss / len(val_loader.dataset)\n        val_epoch_acc = np.mean(np.array(val_preds) == np.array(val_labels))\n        val_epoch_precision = precision_score(val_labels, val_preds, zero_division=0)\n        val_epoch_recall = recall_score(val_labels, val_preds, zero_division=0)\n        val_epoch_f1 = f1_score(val_labels, val_preds, zero_division=0)\n        \n        scheduler.step(val_epoch_loss)\n        \n        print(f'Epoch {epoch+1}/{num_epochs}:')\n        print(f'Train Loss: {epoch_loss:.4f}, Acc: {epoch_acc:.4f}, Precision: {epoch_precision:.4f}, Recall: {epoch_recall:.4f}, F1: {epoch_f1:.4f}')\n        print(f'Val Loss: {val_epoch_loss:.4f}, Acc: {val_epoch_acc:.4f}, Precision: {val_epoch_precision:.4f}, Recall: {val_epoch_recall:.4f}, F1: {val_epoch_f1:.4f}')\n        \n        if val_epoch_loss < best_val_loss:\n            best_val_loss = val_epoch_loss\n            torch.save(model.state_dict(), '/kaggle/working/best_efficientnet_model.pth')\n            print(\"Saved best model!\")\n    \n    return model\n\n# Evaluate model on test set\ndef evaluate_model(model, test_loader, criterion):\n    model.eval()\n    test_loss = 0.0\n    test_preds = []\n    test_labels = []\n    \n    with torch.no_grad():\n        for inputs, labels in tqdm(test_loader, desc='Evaluating on Test Set'):\n            inputs, labels = inputs.to(device), labels.to(device)\n            outputs = model(inputs).squeeze()\n            \n            if outputs.ndim == 0:\n                outputs = outputs.unsqueeze(0)\n                \n            loss = criterion(outputs, labels)\n            test_loss += loss.item() * inputs.size(0)\n            predicted = (outputs > 0.5).float()\n            test_preds.extend(predicted.cpu().numpy())\n            test_labels.extend(labels.cpu().numpy())\n    \n    test_loss = test_loss / len(test_loader.dataset)\n    test_acc = np.mean(np.array(test_preds) == np.array(test_labels))\n    test_precision = precision_score(test_labels, test_preds, zero_division=0)\n    test_recall = recall_score(test_labels, test_preds, zero_division=0)\n    test_f1 = f1_score(test_labels, test_preds, zero_division=0)\n    \n    print(f'Test Loss: {test_loss:.4f}, Acc: {test_acc:.4f}, Precision: {test_precision:.4f}, Recall: {test_recall:.4f}, F1: {test_f1:.4f}')\n\n# Train the model\ntrained_model = train_model(model, train_loader, val_loader, criterion, optimizer, scheduler, num_epochs=3)\n\n# Evaluate on test set\nevaluate_model(trained_model, test_loader, criterion)\n\n# Modified test data processing for inference\nimport shutil\n\n# Modified NexarDataset class with robust error handling\nclass NexarDataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df = df\n        self.transform = transform\n        # Filter out rows with invalid frame paths\n        self.df = self.df[self.df['frame_path'].apply(self._is_valid_file)]\n        print(f\"Filtered dataset to {len(self.df)} valid frames\")\n        \n    def _is_valid_file(self, path):\n        return os.path.isfile(path) and os.access(path, os.R_OK)\n    \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img_path = row['frame_path']\n        label = row['label'] if row['label'] is not None else 0.0\n        \n        try:\n            if not self._is_valid_file(img_path):\n                raise FileNotFoundError(f\"Image file not found or inaccessible: {img_path}\")\n                \n            image = Image.open(img_path).convert('RGB')\n            \n            if self.transform:\n                image = self.transform(image)\n                \n            return image, torch.tensor(label, dtype=torch.float32)\n        except Exception as e:\n            print(f\"Error loading image {img_path}: {e}\")\n            # Return a placeholder image and label\n            placeholder = torch.zeros((3, 240, 240))  # Updated size for EfficientNet-B1\n            return placeholder, torch.tensor(label, dtype=torch.float32)\n\n# Load test data\ntest_df = pd.read_csv('/kaggle/input/nexar-collision-prediction/test.csv')\nprint(\"Test data loaded:\", test_df.shape)\n\n# Process a subset of test videos\nsample_test_df = test_df.sample(5, random_state=42) if len(test_df) > 5 else test_df\ntest_frames_dir = '/kaggle/working/test_frames_efficientnet'\n\n# Clear the test frames directory to avoid conflicts\nif os.path.exists(test_frames_dir):\n    shutil.rmtree(test_frames_dir)\nos.makedirs(test_frames_dir, exist_ok=True)\n\n# Process videos with additional debugging\ndef process_videos(df, output_dir, is_train=False):\n    all_frames = []\n    \n    for _, row in tqdm(df.iterrows(), total=len(df), desc=\"Processing test videos\"):\n        video_id = row['id']\n        frame_data = extract_frames(video_id, row, output_dir, is_train)\n        # Validate saved frames\n        valid_frames = []\n        for frame in frame_data:\n            if os.path.isfile(frame['frame_path']) and os.access(frame['frame_path'], os.R_OK):\n                valid_frames.append(frame)\n            else:\n                print(f\"Invalid frame file: {frame['frame_path']}\")\n        all_frames.extend(valid_frames)\n    \n    return pd.DataFrame(all_frames)\n\ntest_frames_df = process_videos(sample_test_df, test_frames_dir, is_train=False)\n\n# Check if we have any test frames\nif len(test_frames_df) == 0:\n    print(\"No test frames were extracted. Creating dummy test data...\")\n    dummy_test_data = []\n    os.makedirs('/kaggle/working/dummy_test_images_efficientnet', exist_ok=True)\n    for i in range(50):\n        frame_path = f'/kaggle/working/dummy_test_images_efficientnet/dummy_test_{i:05d}.jpg'\n        img = np.ones((240, 240, 3), dtype=np.uint8) * 128  # Updated size for EfficientNet-B1\n        cv2.imwrite(frame_path, img)\n        dummy_test_data.append({\n            'frame_path': frame_path,\n            'video_id': f\"{i % 5:05d}\",\n            'frame_idx': i,\n            'label': None\n        })\n    test_frames_df = pd.DataFrame(dummy_test_data)\n\ntest_frames_df.to_csv('/kaggle/working/test_frames_efficientnet.csv', index=False)\nprint(f\"Processed {len(test_frames_df)} test frames from {len(sample_test_df)} videos\")\n\n# Create test dataset and loader\ntest_dataset = NexarDataset(test_frames_df, transform=val_transform)\nif len(test_dataset) == 0:\n    raise ValueError(\"Test dataset is empty after filtering invalid frames\")\n\ntest_loader = DataLoader(test_dataset, batch_size=16, shuffle=False, num_workers=0)  # Set num_workers to 0\n\n# Load best model for inference\nmodel.load_state_dict(torch.load('/kaggle/working/best_efficientnet_model.pth', weights_only=True))  # Set weights_only=True\nmodel.eval()\n\n# Make predictions on test set with error handling\npredictions = []\nvideo_ids = []\n\nwith torch.no_grad():\n    for i, (inputs, _) in enumerate(tqdm(test_loader, desc='Making predictions with EfficientNet-B1')):\n        try:\n            inputs = inputs.to(device)\n            outputs = model(inputs).squeeze().cpu().numpy()\n            \n            batch_indices = list(range(i*test_loader.batch_size, \n                                     min((i+1)*test_loader.batch_size, len(test_dataset))))\n            \n            actual_batch_size = min(test_loader.batch_size, len(test_dataset) - i*test_loader.batch_size)\n            batch_indices = batch_indices[:actual_batch_size]\n            \n            batch_video_ids = [test_frames_df.iloc[idx]['video_id'] for idx in batch_indices]\n            \n            if isinstance(outputs, np.float32):\n                outputs = np.array([outputs])\n                \n            if len(outputs) > len(batch_indices):\n                outputs = outputs[:len(batch_indices)]\n                \n            predictions.extend(outputs)\n            video_ids.extend(batch_video_ids)\n        except Exception as e:\n            print(f\"Error processing batch {i}: {e}\")\n            continue\n\nif not predictions:\n    raise ValueError(\"No predictions were generated. Check test data and model inference.\")\n\n# Create a dataframe with predictions for each frame\nprediction_df = pd.DataFrame({\n    'video_id': video_ids,\n    'prediction': predictions\n})\n\n# Aggregate predictions by video\nvideo_predictions = prediction_df.groupby('video_id')['prediction'].max().reset_index()\n\n# Create submission file\nsubmission_df = pd.DataFrame({\n    'id': video_predictions['video_id'],\n    'target': video_predictions['prediction']\n})\n\nsubmission_df.to_csv('/kaggle/working/efficientnet_submission.csv', index=False)\nprint(\"EfficientNet-B1 submission file created!\")\nprint(submission_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-18T22:46:56.710057Z","iopub.execute_input":"2025-04-18T22:46:56.710791Z","iopub.status.idle":"2025-04-18T22:52:20.291712Z","shell.execute_reply.started":"2025-04-18T22:46:56.710759Z","shell.execute_reply":"2025-04-18T22:52:20.290784Z"}},"outputs":[{"name":"stdout","text":"Training data loaded: (1500, 4)\n","output_type":"stream"},{"name":"stderr","text":"100%|██████████| 10/10 [00:53<00:00,  5.39s/it]\n","output_type":"stream"},{"name":"stdout","text":"Processed 10534 frames from 10 videos\n","output_type":"stream"},{"name":"stderr","text":"/usr/local/lib/python3.11/dist-packages/torch/optim/lr_scheduler.py:62: UserWarning: The verbose parameter is deprecated. Please use get_last_lr() to access the learning rate.\n  warnings.warn(\n","output_type":"stream"},{"name":"stdout","text":"Using device: cuda\n","output_type":"stream"},{"name":"stderr","text":"Epoch 1/3 - Training: 100%|██████████| 474/474 [01:07<00:00,  7.07it/s]\nEpoch 1/3 - Validation: 100%|██████████| 119/119 [00:11<00:00, 10.06it/s]\n","output_type":"stream"},{"name":"stdout","text":"Epoch 1/3:\nTrain Loss: 0.0980, Acc: 0.9767, Precision: 0.9207, Recall: 0.4794, F1: 0.6305\nVal Loss: 0.0226, Acc: 0.9979, Precision: 0.9756, Recall: 0.9756, F1: 0.9756\nSaved best model!\n","output_type":"stream"},{"name":"stderr","text":"Epoch 2/3 - Training: 100%|██████████| 474/474 [01:06<00:00,  7.08it/s]\nEpoch 2/3 - Validation: 100%|██████████| 119/119 [00:11<00:00,  9.92it/s]\n","output_type":"stream"},{"name":"stdout","text":"Epoch 2/3:\nTrain Loss: 0.0200, Acc: 0.9938, Precision: 0.9323, Recall: 0.9175, F1: 0.9248\nVal Loss: 0.0111, Acc: 0.9968, Precision: 0.9419, Recall: 0.9878, F1: 0.9643\nSaved best model!\n","output_type":"stream"},{"name":"stderr","text":"Epoch 3/3 - Training: 100%|██████████| 474/474 [01:06<00:00,  7.09it/s]\nEpoch 3/3 - Validation: 100%|██████████| 119/119 [00:11<00:00, 10.48it/s]\n","output_type":"stream"},{"name":"stdout","text":"Epoch 3/3:\nTrain Loss: 0.0121, Acc: 0.9962, Precision: 0.9414, Recall: 0.9683, F1: 0.9546\nVal Loss: 0.0067, Acc: 0.9979, Precision: 0.9643, Recall: 0.9878, F1: 0.9759\nSaved best model!\n","output_type":"stream"},{"name":"stderr","text":"Evaluating on Test Set: 100%|██████████| 66/66 [00:06<00:00, 10.21it/s]\n","output_type":"stream"},{"name":"stdout","text":"Test Loss: 0.0047, Acc: 0.9991, Precision: 0.9762, Recall: 1.0000, F1: 0.9880\nTest data loaded: (1344, 1)\n","output_type":"stream"},{"name":"stderr","text":"Processing test videos: 100%|██████████| 5/5 [00:07<00:00,  1.51s/it]\n","output_type":"stream"},{"name":"stdout","text":"Processed 1459 test frames from 5 videos\nFiltered dataset to 1459 valid frames\n","output_type":"stream"},{"name":"stderr","text":"Making predictions with EfficientNet-B1: 100%|██████████| 92/92 [00:18<00:00,  4.93it/s]","output_type":"stream"},{"name":"stdout","text":"EfficientNet-B1 submission file created!\n      id    target\n0  00390  0.017101\n1  00531  0.011080\n2  01113  0.826879\n3  01246  0.901982\n4  02561  0.574777\n","output_type":"stream"},{"name":"stderr","text":"\n","output_type":"stream"}],"execution_count":17},{"cell_type":"markdown","source":"## MobileNet V2","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as transforms\nfrom PIL import Image\nimport cv2\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import precision_score, recall_score, f1_score\nfrom tqdm import tqdm\nfrom torchvision.models import mobilenet_v2, MobileNet_V2_Weights\n\n# Constants\nFRAME_RATE = 30  # 30 fps\nWINDOW_SIZE = 5  # Number of frames to use for prediction\nPOSITIVE_WINDOW = 1.5  # Seconds before the event to consider as positive\n\n# Load the training data\ntrain_df = pd.read_csv('/kaggle/input/nexar-collision-prediction/train.csv')\nprint(\"Training data loaded:\", train_df.shape)\n\n# Fix the video path issue by ensuring proper formatting of video IDs\ndef get_video_path(video_id, is_train=True):\n    formatted_id = f\"{int(video_id):05d}\"\n    base_dir = '/kaggle/input/nexar-collision-prediction'\n    subfolder = 'train' if is_train else 'test'\n    return os.path.join(base_dir, subfolder, f\"{formatted_id}.mp4\")\n\n# Data preprocessing function to extract frames\ndef extract_frames(video_id, df_row, output_dir, is_train=True):\n    formatted_id = f\"{int(video_id):05d}\"\n    video_path = get_video_path(video_id, is_train)\n    target = df_row['target'] if 'target' in df_row else None\n    \n    os.makedirs(output_dir, exist_ok=True)\n    \n    cap = cv2.VideoCapture(video_path)\n    if not cap.isOpened():\n        print(f\"Error opening video: {video_path}\")\n        return []\n    \n    frame_count = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    fps = cap.get(cv2.CAP_PROP_FPS)\n    \n    event_frame = None\n    alert_frame = None\n    \n    if is_train and target == 1:\n        event_time = float(df_row['time_of_event'])\n        alert_time = float(df_row['time_of_alert'])\n        event_frame = int(event_time * fps)\n        alert_frame = int(alert_time * fps)\n    \n    frame_data = []\n    \n    for frame_idx in range(frame_count):\n        ret, frame = cap.read()\n        if not ret:\n            break\n        \n        label = 0\n        if is_train and target == 1:\n            if alert_frame <= frame_idx <= event_frame:\n                label = 1\n            elif alert_frame - int(POSITIVE_WINDOW * fps) <= frame_idx < alert_frame:\n                label = 1\n        \n        frame_path = os.path.join(output_dir, f\"{formatted_id}_{frame_idx:05d}.jpg\")\n        cv2.imwrite(frame_path, frame)\n        \n        frame_data.append({\n            'frame_path': frame_path,\n            'video_id': formatted_id,\n            'frame_idx': frame_idx,\n            'label': label if is_train else None\n        })\n    \n    cap.release()\n    return frame_data\n\n# Process all videos and create a dataframe of frames\ndef process_videos(df, output_dir, is_train=True):\n    all_frames = []\n    \n    for _, row in tqdm(df.iterrows(), total=len(df)):\n        video_id = row['id']\n        frame_data = extract_frames(video_id, row, output_dir, is_train)\n        all_frames.extend(frame_data)\n    \n    return pd.DataFrame(all_frames)\n\n# Process a subset of videos for faster execution\nsample_train_df = train_df.sample(10, random_state=42) if len(train_df) > 10 else train_df\ntrain_frames_df = process_videos(sample_train_df, '/kaggle/working/train_frames_mobilenet', is_train=True)\n\n# Check if we have any frames before proceeding\nif len(train_frames_df) == 0:\n    print(\"No frames were extracted. Creating dummy dataset...\")\n    dummy_data = []\n    for i in range(100):\n        dummy_data.append({\n            'frame_path': f'/kaggle/working/dummy_{i:05d}.jpg',\n            'video_id': f\"{i % 5:05d}\",\n            'frame_idx': i,\n            'label': i % 2\n        })\n    train_frames_df = pd.DataFrame(dummy_data)\n    os.makedirs('/kaggle/working/dummy_images', exist_ok=True)\n    for i in range(100):\n        img = np.ones((224, 224, 3), dtype=np.uint8) * 128\n        cv2.imwrite(f'/kaggle/working/dummy_{i:05d}.jpg', img)\n\ntrain_frames_df.to_csv('/kaggle/working/train_frames_mobilenet.csv', index=False)\nprint(f\"Processed {len(train_frames_df)} frames from {len(sample_train_df)} videos\")\n\n# Custom Dataset class\nclass NexarDataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df = df\n        self.transform = transform\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img_path = row['frame_path']\n        label = row['label']\n        \n        try:\n            image = Image.open(img_path).convert('RGB')\n            if self.transform:\n                image = self.transform(image)\n            return image, torch.tensor(label, dtype=torch.float32)\n        except Exception as e:\n            print(f\"Error loading image {img_path}: {e}\")\n            placeholder = torch.zeros((3, 224, 224))\n            return placeholder, torch.tensor(0, dtype=torch.float32)\n\n# Define transforms for MobileNetV2\ntrain_transform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomRotation(10),\n    transforms.ColorJitter(brightness=0.2, contrast=0.2),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\nval_transform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\n# Split data into train, validation, and test sets\ntrain_val_frames, test_frames = train_test_split(train_frames_df, test_size=0.1, random_state=42)\ntrain_frames, val_frames = train_test_split(train_val_frames, test_size=0.2, random_state=42)\n\n# Create datasets\ntrain_dataset = NexarDataset(train_frames, transform=train_transform)\nval_dataset = NexarDataset(val_frames, transform=val_transform)\ntest_dataset = NexarDataset(test_frames, transform=val_transform)\n\n# Create data loaders\ntrain_loader = DataLoader(train_dataset, batch_size=32, shuffle=True, num_workers=2)\nval_loader = DataLoader(val_dataset, batch_size=32, shuffle=False, num_workers=2)\ntest_loader = DataLoader(test_dataset, batch_size=32, shuffle=False, num_workers=2)\n\n# Initialize MobileNetV2 model\ndef get_mobilenet_model():\n    # Load pre-trained MobileNetV2 model\n    model = mobilenet_v2(weights=MobileNet_V2_Weights.IMAGENET1K_V1)\n    \n    # Modify the classifier for binary classification\n    in_features = model.classifier[1].in_features\n    model.classifier = nn.Sequential(\n        nn.Dropout(0.2),\n        nn.Linear(in_features, 256),\n        nn.ReLU(),\n        nn.Dropout(0.2),\n        nn.Linear(256, 1),\n        nn.Sigmoid()\n    )\n    \n    return model\n\nmodel = get_mobilenet_model()\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\nmodel = model.to(device)\n\n# Define loss function and optimizer\ncriterion = nn.BCELoss()\n# For MobileNetV2, we can use a slightly higher learning rate than ViT\noptimizer = optim.AdamW(model.parameters(), lr=1e-4, weight_decay=0.01)\nscheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=3, verbose=True)\n\n# Training function with additional metrics\ndef train_model(model, train_loader, val_loader, criterion, optimizer, scheduler, num_epochs=5):\n    best_val_loss = float('inf')\n    \n    for epoch in range(num_epochs):\n        # Training phase\n        model.train()\n        running_loss = 0.0\n        train_preds = []\n        train_labels = []\n        \n        for inputs, labels in tqdm(train_loader, desc=f'Epoch {epoch+1}/{num_epochs} - Training'):\n            inputs, labels = inputs.to(device), labels.to(device)\n            \n            optimizer.zero_grad()\n            outputs = model(inputs).squeeze()\n            \n            if outputs.ndim == 0:\n                outputs = outputs.unsqueeze(0)\n                \n            loss = criterion(outputs, labels)\n            loss.backward()\n            optimizer.step()\n            \n            running_loss += loss.item() * inputs.size(0)\n            predicted = (outputs > 0.5).float()\n            train_preds.extend(predicted.cpu().numpy())\n            train_labels.extend(labels.cpu().numpy())\n        \n        epoch_loss = running_loss / len(train_loader.dataset)\n        epoch_acc = np.mean(np.array(train_preds) == np.array(train_labels))\n        epoch_precision = precision_score(train_labels, train_preds, zero_division=0)\n        epoch_recall = recall_score(train_labels, train_preds, zero_division=0)\n        epoch_f1 = f1_score(train_labels, train_preds, zero_division=0)\n        \n        # Validation phase\n        model.eval()\n        val_loss = 0.0\n        val_preds = []\n        val_labels = []\n        \n        with torch.no_grad():\n            for inputs, labels in tqdm(val_loader, desc=f'Epoch {epoch+1}/{num_epochs} - Validation'):\n                inputs, labels = inputs.to(device), labels.to(device)\n                outputs = model(inputs).squeeze()\n                \n                if outputs.ndim == 0:\n                    outputs = outputs.unsqueeze(0)\n                    \n                loss = criterion(outputs, labels)\n                val_loss += loss.item() * inputs.size(0)\n                predicted = (outputs > 0.5).float()\n                val_preds.extend(predicted.cpu().numpy())\n                val_labels.extend(labels.cpu().numpy())\n        \n        val_epoch_loss = val_loss / len(val_loader.dataset)\n        val_epoch_acc = np.mean(np.array(val_preds) == np.array(val_labels))\n        val_epoch_precision = precision_score(val_labels, val_preds, zero_division=0)\n        val_epoch_recall = recall_score(val_labels, val_preds, zero_division=0)\n        val_epoch_f1 = f1_score(val_labels, val_preds, zero_division=0)\n        \n        scheduler.step(val_epoch_loss)\n        \n        print(f'Epoch {epoch+1}/{num_epochs}:')\n        print(f'Train Loss: {epoch_loss:.4f}, Acc: {epoch_acc:.4f}, Precision: {epoch_precision:.4f}, Recall: {epoch_recall:.4f}, F1: {epoch_f1:.4f}')\n        print(f'Val Loss: {val_epoch_loss:.4f}, Acc: {val_epoch_acc:.4f}, Precision: {val_epoch_precision:.4f}, Recall: {val_epoch_recall:.4f}, F1: {val_epoch_f1:.4f}')\n        \n        if val_epoch_loss < best_val_loss:\n            best_val_loss = val_epoch_loss\n            torch.save(model.state_dict(), '/kaggle/working/best_mobilenet_model.pth')\n            print(\"Saved best model!\")\n    \n    return model\n\n# Evaluate model on test set\ndef evaluate_model(model, test_loader, criterion):\n    model.eval()\n    test_loss = 0.0\n    test_preds = []\n    test_labels = []\n    \n    with torch.no_grad():\n        for inputs, labels in tqdm(test_loader, desc='Evaluating on Test Set'):\n            inputs, labels = inputs.to(device), labels.to(device)\n            outputs = model(inputs).squeeze()\n            \n            if outputs.ndim == 0:\n                outputs = outputs.unsqueeze(0)\n                \n            loss = criterion(outputs, labels)\n            test_loss += loss.item() * inputs.size(0)\n            predicted = (outputs > 0.5).float()\n            test_preds.extend(predicted.cpu().numpy())\n            test_labels.extend(labels.cpu().numpy())\n    \n    test_loss = test_loss / len(test_loader.dataset)\n    test_acc = np.mean(np.array(test_preds) == np.array(test_labels))\n    test_precision = precision_score(test_labels, test_preds, zero_division=0)\n    test_recall = recall_score(test_labels, test_preds, zero_division=0)\n    test_f1 = f1_score(test_labels, test_preds, zero_division=0)\n    \n    print(f'Test Loss: {test_loss:.4f}, Acc: {test_acc:.4f}, Precision: {test_precision:.4f}, Recall: {test_recall:.4f}, F1: {test_f1:.4f}')\n\n# Train the model\ntrained_model = train_model(model, train_loader, val_loader, criterion, optimizer, scheduler, num_epochs=5)\n\n# Evaluate on test set\nevaluate_model(trained_model, test_loader, criterion)\n\n# Modified test data processing for inference\nimport shutil\n\n# Modified NexarDataset class with robust error handling\nclass NexarDataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df = df\n        self.transform = transform\n        # Filter out rows with invalid frame paths\n        self.df = self.df[self.df['frame_path'].apply(self._is_valid_file)]\n        print(f\"Filtered dataset to {len(self.df)} valid frames\")\n        \n    def _is_valid_file(self, path):\n        return os.path.isfile(path) and os.access(path, os.R_OK)\n    \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img_path = row['frame_path']\n        label = row['label'] if row['label'] is not None else 0.0\n        \n        try:\n            if not self._is_valid_file(img_path):\n                raise FileNotFoundError(f\"Image file not found or inaccessible: {img_path}\")\n                \n            image = Image.open(img_path).convert('RGB')\n            \n            if self.transform:\n                image = self.transform(image)\n                \n            return image, torch.tensor(label, dtype=torch.float32)\n        except Exception as e:\n            print(f\"Error loading image {img_path}: {e}\")\n            # Return a placeholder image and label\n            placeholder = torch.zeros((3, 224, 224))\n            return placeholder, torch.tensor(label, dtype=torch.float32)\n\n# Load test data\ntest_df = pd.read_csv('/kaggle/input/nexar-collision-prediction/test.csv')\nprint(\"Test data loaded:\", test_df.shape)\n\n# Process a subset of test videos\nsample_test_df = test_df.sample(2, random_state=42) if len(test_df) > 2 else test_df\ntest_frames_dir = '/kaggle/working/test_frames_mobilenet'\n\n# Clear the test frames directory to avoid conflicts\nif os.path.exists(test_frames_dir):\n    shutil.rmtree(test_frames_dir)\nos.makedirs(test_frames_dir, exist_ok=True)\n\n# Process videos with additional debugging\ndef process_videos(df, output_dir, is_train=False):\n    all_frames = []\n    \n    for _, row in tqdm(df.iterrows(), total=len(df), desc=\"Processing test videos\"):\n        video_id = row['id']\n        frame_data = extract_frames(video_id, row, output_dir, is_train)\n        # Validate saved frames\n        valid_frames = []\n        for frame in frame_data:\n            if os.path.isfile(frame['frame_path']) and os.access(frame['frame_path'], os.R_OK):\n                valid_frames.append(frame)\n            else:\n                print(f\"Invalid frame file: {frame['frame_path']}\")\n        all_frames.extend(valid_frames)\n    \n    return pd.DataFrame(all_frames)\n\ntest_frames_df = process_videos(sample_test_df, test_frames_dir, is_train=False)\n\n# Check if we have any test frames\nif len(test_frames_df) == 0:\n    print(\"No test frames were extracted. Creating dummy test data...\")\n    dummy_test_data = []\n    os.makedirs('/kaggle/working/dummy_test_images_mobilenet', exist_ok=True)\n    for i in range(50):\n        frame_path = f'/kaggle/working/dummy_test_images_mobilenet/dummy_test_{i:05d}.jpg'\n        img = np.ones((224, 224, 3), dtype=np.uint8) * 128\n        cv2.imwrite(frame_path, img)\n        dummy_test_data.append({\n            'frame_path': frame_path,\n            'video_id': f\"{i % 5:05d}\",\n            'frame_idx': i,\n            'label': None\n        })\n    test_frames_df = pd.DataFrame(dummy_test_data)\n\ntest_frames_df.to_csv('/kaggle/working/test_frames_mobilenet.csv', index=False)\nprint(f\"Processed {len(test_frames_df)} test frames from {len(sample_test_df)} videos\")\n\n# Create test dataset and loader\ntest_dataset = NexarDataset(test_frames_df, transform=val_transform)\nif len(test_dataset) == 0:\n    raise ValueError(\"Test dataset is empty after filtering invalid frames\")\n\ntest_loader = DataLoader(test_dataset, batch_size=32, shuffle=False, num_workers=0)  # Set num_workers to 0\n\n# Load best model for inference\nmodel.load_state_dict(torch.load('/kaggle/working/best_mobilenet_model.pth', weights_only=True))  # Set weights_only=True\nmodel.eval()\n\n# Make predictions on test set with error handling\npredictions = []\nvideo_ids = []\n\nwith torch.no_grad():\n    for i, (inputs, _) in enumerate(tqdm(test_loader, desc='Making predictions with MobileNetV2')):\n        try:\n            inputs = inputs.to(device)\n            outputs = model(inputs).squeeze().cpu().numpy()\n            \n            batch_indices = list(range(i*test_loader.batch_size, \n                                     min((i+1)*test_loader.batch_size, len(test_dataset))))\n            \n            actual_batch_size = min(test_loader.batch_size, len(test_dataset) - i*test_loader.batch_size)\n            batch_indices = batch_indices[:actual_batch_size]\n            \n            batch_video_ids = [test_frames_df.iloc[idx]['video_id'] for idx in batch_indices]\n            \n            if isinstance(outputs, np.float32):\n                outputs = np.array([outputs])\n                \n            if len(outputs) > len(batch_indices):\n                outputs = outputs[:len(batch_indices)]\n                \n            predictions.extend(outputs)\n            video_ids.extend(batch_video_ids)\n        except Exception as e:\n            print(f\"Error processing batch {i}: {e}\")\n            continue\n\nif not predictions:\n    raise ValueError(\"No predictions were generated. Check test data and model inference.\")\n\n# Create a dataframe with predictions for each frame\nprediction_df = pd.DataFrame({\n    'video_id': video_ids,\n    'prediction': predictions\n})\n\n# Aggregate predictions by video\nvideo_predictions = prediction_df.groupby('video_id')['prediction'].max().reset_index()\n\n# Create submission file\nsubmission_df = pd.DataFrame({\n    'id': video_predictions['video_id'],\n    'target': video_predictions['prediction']\n})\n\nsubmission_df.to_csv('/kaggle/working/mobilenet_submission.csv', index=False)\nprint(\"MobileNetV2 submission file created!\")\nprint(submission_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-18T22:57:31.664334Z","iopub.execute_input":"2025-04-18T22:57:31.664987Z","iopub.status.idle":"2025-04-18T23:04:09.458527Z","shell.execute_reply.started":"2025-04-18T22:57:31.664963Z","shell.execute_reply":"2025-04-18T23:04:09.457553Z"}},"outputs":[{"name":"stdout","text":"Training data loaded: (1500, 4)\n","output_type":"stream"},{"name":"stderr","text":"100%|██████████| 10/10 [00:55<00:00,  5.54s/it]\nDownloading: \"https://download.pytorch.org/models/mobilenet_v2-b0353104.pth\" to /root/.cache/torch/hub/checkpoints/mobilenet_v2-b0353104.pth\n","output_type":"stream"},{"name":"stdout","text":"Processed 10534 frames from 10 videos\n","output_type":"stream"},{"name":"stderr","text":"100%|██████████| 13.6M/13.6M [00:00<00:00, 104MB/s] \n/usr/local/lib/python3.11/dist-packages/torch/optim/lr_scheduler.py:62: UserWarning: The verbose parameter is deprecated. Please use get_last_lr() to access the learning rate.\n  warnings.warn(\n","output_type":"stream"},{"name":"stdout","text":"Using device: cuda\n","output_type":"stream"},{"name":"stderr","text":"Epoch 1/5 - Training: 100%|██████████| 237/237 [00:54<00:00,  4.33it/s]\nEpoch 1/5 - Validation: 100%|██████████| 60/60 [00:10<00:00,  5.52it/s]\n","output_type":"stream"},{"name":"stdout","text":"Epoch 1/5:\nTrain Loss: 0.0657, Acc: 0.9780, Precision: 0.9022, Recall: 0.5270, F1: 0.6653\nVal Loss: 0.0165, Acc: 0.9947, Precision: 0.9390, Recall: 0.9390, F1: 0.9390\nSaved best model!\n","output_type":"stream"},{"name":"stderr","text":"Epoch 2/5 - Training: 100%|██████████| 237/237 [00:54<00:00,  4.37it/s]\nEpoch 2/5 - Validation: 100%|██████████| 60/60 [00:10<00:00,  5.60it/s]\n","output_type":"stream"},{"name":"stdout","text":"Epoch 2/5:\nTrain Loss: 0.0214, Acc: 0.9942, Precision: 0.9357, Recall: 0.9238, F1: 0.9297\nVal Loss: 0.0069, Acc: 0.9984, Precision: 0.9759, Recall: 0.9878, F1: 0.9818\nSaved best model!\n","output_type":"stream"},{"name":"stderr","text":"Epoch 3/5 - Training: 100%|██████████| 237/237 [00:53<00:00,  4.43it/s]\nEpoch 3/5 - Validation: 100%|██████████| 60/60 [00:11<00:00,  5.29it/s]\n","output_type":"stream"},{"name":"stdout","text":"Epoch 3/5:\nTrain Loss: 0.0113, Acc: 0.9964, Precision: 0.9557, Recall: 0.9587, F1: 0.9572\nVal Loss: 0.0028, Acc: 0.9984, Precision: 0.9759, Recall: 0.9878, F1: 0.9818\nSaved best model!\n","output_type":"stream"},{"name":"stderr","text":"Epoch 4/5 - Training: 100%|██████████| 237/237 [00:53<00:00,  4.42it/s]\nEpoch 4/5 - Validation: 100%|██████████| 60/60 [00:11<00:00,  5.39it/s]\n","output_type":"stream"},{"name":"stdout","text":"Epoch 4/5:\nTrain Loss: 0.0125, Acc: 0.9958, Precision: 0.9436, Recall: 0.9556, F1: 0.9495\nVal Loss: 0.0081, Acc: 0.9958, Precision: 1.0000, Recall: 0.9024, F1: 0.9487\n","output_type":"stream"},{"name":"stderr","text":"Epoch 5/5 - Training: 100%|██████████| 237/237 [00:54<00:00,  4.37it/s]\nEpoch 5/5 - Validation: 100%|██████████| 60/60 [00:11<00:00,  5.37it/s]\n","output_type":"stream"},{"name":"stdout","text":"Epoch 5/5:\nTrain Loss: 0.0071, Acc: 0.9976, Precision: 0.9744, Recall: 0.9683, F1: 0.9713\nVal Loss: 0.0047, Acc: 0.9984, Precision: 1.0000, Recall: 0.9634, F1: 0.9814\n","output_type":"stream"},{"name":"stderr","text":"Evaluating on Test Set: 100%|██████████| 33/33 [00:06<00:00,  5.29it/s]\n","output_type":"stream"},{"name":"stdout","text":"Test Loss: 0.0075, Acc: 0.9972, Precision: 0.9318, Recall: 1.0000, F1: 0.9647\nTest data loaded: (1344, 1)\n","output_type":"stream"},{"name":"stderr","text":"Processing test videos: 100%|██████████| 2/2 [00:03<00:00,  1.52s/it]\n","output_type":"stream"},{"name":"stdout","text":"Processed 596 test frames from 2 videos\nFiltered dataset to 596 valid frames\n","output_type":"stream"},{"name":"stderr","text":"Making predictions with MobileNetV2: 100%|██████████| 19/19 [00:06<00:00,  2.84it/s]","output_type":"stream"},{"name":"stdout","text":"MobileNetV2 submission file created!\n      id    target\n0  00531  0.120665\n1  01246  0.266248\n","output_type":"stream"},{"name":"stderr","text":"\n","output_type":"stream"}],"execution_count":18},{"cell_type":"markdown","source":"## DenseNet121","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as transforms\nfrom PIL import Image\nimport cv2\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import precision_score, recall_score, f1_score\nfrom tqdm import tqdm\nfrom torchvision.models import densenet121, DenseNet121_Weights\n\n# Constants\nFRAME_RATE = 30  # 30 fps\nWINDOW_SIZE = 5  # Number of frames to use for prediction\nPOSITIVE_WINDOW = 1.5  # Seconds before the event to consider as positive\n\n# Load the training data\ntrain_df = pd.read_csv('/kaggle/input/nexar-collision-prediction/train.csv')\nprint(\"Training data loaded:\", train_df.shape)\n\n# Fix the video path issue by ensuring proper formatting of video IDs\ndef get_video_path(video_id, is_train=True):\n    formatted_id = f\"{int(video_id):05d}\"\n    base_dir = '/kaggle/input/nexar-collision-prediction'\n    subfolder = 'train' if is_train else 'test'\n    return os.path.join(base_dir, subfolder, f\"{formatted_id}.mp4\")\n\n# Data preprocessing function to extract frames\ndef extract_frames(video_id, df_row, output_dir, is_train=True):\n    formatted_id = f\"{int(video_id):05d}\"\n    video_path = get_video_path(video_id, is_train)\n    target = df_row['target'] if 'target' in df_row else None\n    \n    os.makedirs(output_dir, exist_ok=True)\n    \n    cap = cv2.VideoCapture(video_path)\n    if not cap.isOpened():\n        print(f\"Error opening video: {video_path}\")\n        return []\n    \n    frame_count = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    fps = cap.get(cv2.CAP_PROP_FPS)\n    \n    event_frame = None\n    alert_frame = None\n    \n    if is_train and target == 1:\n        event_time = float(df_row['time_of_event'])\n        alert_time = float(df_row['time_of_alert'])\n        event_frame = int(event_time * fps)\n        alert_frame = int(alert_time * fps)\n    \n    frame_data = []\n    \n    for frame_idx in range(frame_count):\n        ret, frame = cap.read()\n        if not ret:\n            break\n        \n        label = 0\n        if is_train and target == 1:\n            if alert_frame <= frame_idx <= event_frame:\n                label = 1\n            elif alert_frame - int(POSITIVE_WINDOW * fps) <= frame_idx < alert_frame:\n                label = 1\n        \n        frame_path = os.path.join(output_dir, f\"{formatted_id}_{frame_idx:05d}.jpg\")\n        cv2.imwrite(frame_path, frame)\n        \n        frame_data.append({\n            'frame_path': frame_path,\n            'video_id': formatted_id,\n            'frame_idx': frame_idx,\n            'label': label if is_train else None\n        })\n    \n    cap.release()\n    return frame_data\n\n# Process all videos and create a dataframe of frames\ndef process_videos(df, output_dir, is_train=True):\n    all_frames = []\n    \n    for _, row in tqdm(df.iterrows(), total=len(df)):\n        video_id = row['id']\n        frame_data = extract_frames(video_id, row, output_dir, is_train)\n        all_frames.extend(frame_data)\n    \n    return pd.DataFrame(all_frames)\n\n# Process a subset of videos for faster execution\nsample_train_df = train_df.sample(10, random_state=42) if len(train_df) > 10 else train_df\ntrain_frames_df = process_videos(sample_train_df, '/kaggle/working/train_frames_densenet', is_train=True)\n\n# Check if we have any frames before proceeding\nif len(train_frames_df) == 0:\n    print(\"No frames were extracted. Creating dummy dataset...\")\n    dummy_data = []\n    for i in range(100):\n        dummy_data.append({\n            'frame_path': f'/kaggle/working/dummy_densenet_{i:05d}.jpg',\n            'video_id': f\"{i % 5:05d}\",\n            'frame_idx': i,\n            'label': i % 2\n        })\n    train_frames_df = pd.DataFrame(dummy_data)\n    os.makedirs('/kaggle/working/dummy_images_densenet', exist_ok=True)\n    for i in range(100):\n        img = np.ones((224, 224, 3), dtype=np.uint8) * 128\n        cv2.imwrite(f'/kaggle/working/dummy_densenet_{i:05d}.jpg', img)\n\ntrain_frames_df.to_csv('/kaggle/working/train_frames_densenet.csv', index=False)\nprint(f\"Processed {len(train_frames_df)} frames from {len(sample_train_df)} videos\")\n\n# Custom Dataset class\nclass NexarDataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df = df\n        self.transform = transform\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img_path = row['frame_path']\n        label = row['label']\n        \n        try:\n            image = Image.open(img_path).convert('RGB')\n            if self.transform:\n                image = self.transform(image)\n            return image, torch.tensor(label, dtype=torch.float32)\n        except Exception as e:\n            print(f\"Error loading image {img_path}: {e}\")\n            placeholder = torch.zeros((3, 224, 224))\n            return placeholder, torch.tensor(0, dtype=torch.float32)\n\n# Define transforms for DenseNet\ntrain_transform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomRotation(10),\n    transforms.ColorJitter(brightness=0.2, contrast=0.2),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\nval_transform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\n# Split data into train, validation, and test sets\ntrain_val_frames, test_frames = train_test_split(train_frames_df, test_size=0.1, random_state=42)\ntrain_frames, val_frames = train_test_split(train_val_frames, test_size=0.2, random_state=42)\n\n# Create datasets\ntrain_dataset = NexarDataset(train_frames, transform=train_transform)\nval_dataset = NexarDataset(val_frames, transform=val_transform)\ntest_dataset = NexarDataset(test_frames, transform=val_transform)\n\n# Create data loaders\ntrain_loader = DataLoader(train_dataset, batch_size=16, shuffle=True, num_workers=2)\nval_loader = DataLoader(val_dataset, batch_size=16, shuffle=False, num_workers=2)\ntest_loader = DataLoader(test_dataset, batch_size=16, shuffle=False, num_workers=2)\n\n# Initialize DenseNet-121 model\ndef get_densenet_model():\n    # Load pre-trained DenseNet-121 model\n    model = densenet121(weights=DenseNet121_Weights.IMAGENET1K_V1)\n    \n    # Modify the classifier for binary classification\n    num_features = model.classifier.in_features\n    model.classifier = nn.Sequential(\n        nn.Linear(num_features, 256),\n        nn.ReLU(),\n        nn.Dropout(0.3),\n        nn.Linear(256, 1),\n        nn.Sigmoid()\n    )\n    \n    return model\n\nmodel = get_densenet_model()\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\nmodel = model.to(device)\n\n# Define loss function and optimizer\ncriterion = nn.BCELoss()\n# DenseNet often works well with similar hyperparameters to ViT\noptimizer = optim.AdamW(model.parameters(), lr=5e-5, weight_decay=0.01)\nscheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=3, verbose=True)\n\n# Training function with additional metrics\ndef train_model(model, train_loader, val_loader, criterion, optimizer, scheduler, num_epochs=3):\n    best_val_loss = float('inf')\n    \n    for epoch in range(num_epochs):\n        # Training phase\n        model.train()\n        running_loss = 0.0\n        train_preds = []\n        train_labels = []\n        \n        for inputs, labels in tqdm(train_loader, desc=f'Epoch {epoch+1}/{num_epochs} - Training'):\n            inputs, labels = inputs.to(device), labels.to(device)\n            \n            optimizer.zero_grad()\n            outputs = model(inputs).squeeze()\n            \n            if outputs.ndim == 0:\n                outputs = outputs.unsqueeze(0)\n                \n            loss = criterion(outputs, labels)\n            loss.backward()\n            optimizer.step()\n            \n            running_loss += loss.item() * inputs.size(0)\n            predicted = (outputs > 0.5).float()\n            train_preds.extend(predicted.cpu().numpy())\n            train_labels.extend(labels.cpu().numpy())\n        \n        epoch_loss = running_loss / len(train_loader.dataset)\n        epoch_acc = np.mean(np.array(train_preds) == np.array(train_labels))\n        epoch_precision = precision_score(train_labels, train_preds, zero_division=0)\n        epoch_recall = recall_score(train_labels, train_preds, zero_division=0)\n        epoch_f1 = f1_score(train_labels, train_preds, zero_division=0)\n        \n        # Validation phase\n        model.eval()\n        val_loss = 0.0\n        val_preds = []\n        val_labels = []\n        \n        with torch.no_grad():\n            for inputs, labels in tqdm(val_loader, desc=f'Epoch {epoch+1}/{num_epochs} - Validation'):\n                inputs, labels = inputs.to(device), labels.to(device)\n                outputs = model(inputs).squeeze()\n                \n                if outputs.ndim == 0:\n                    outputs = outputs.unsqueeze(0)\n                    \n                loss = criterion(outputs, labels)\n                val_loss += loss.item() * inputs.size(0)\n                predicted = (outputs > 0.5).float()\n                val_preds.extend(predicted.cpu().numpy())\n                val_labels.extend(labels.cpu().numpy())\n        \n        val_epoch_loss = val_loss / len(val_loader.dataset)\n        val_epoch_acc = np.mean(np.array(val_preds) == np.array(val_labels))\n        val_epoch_precision = precision_score(val_labels, val_preds, zero_division=0)\n        val_epoch_recall = recall_score(val_labels, val_preds, zero_division=0)\n        val_epoch_f1 = f1_score(val_labels, val_preds, zero_division=0)\n        \n        scheduler.step(val_epoch_loss)\n        \n        print(f'Epoch {epoch+1}/{num_epochs}:')\n        print(f'Train Loss: {epoch_loss:.4f}, Acc: {epoch_acc:.4f}, Precision: {epoch_precision:.4f}, Recall: {epoch_recall:.4f}, F1: {epoch_f1:.4f}')\n        print(f'Val Loss: {val_epoch_loss:.4f}, Acc: {val_epoch_acc:.4f}, Precision: {val_epoch_precision:.4f}, Recall: {val_epoch_recall:.4f}, F1: {val_epoch_f1:.4f}')\n        \n        if val_epoch_loss < best_val_loss:\n            best_val_loss = val_epoch_loss\n            torch.save(model.state_dict(), '/kaggle/working/best_densenet_model.pth')\n            print(\"Saved best model!\")\n    \n    return model\n\n# Evaluate model on test set\ndef evaluate_model(model, test_loader, criterion):\n    model.eval()\n    test_loss = 0.0\n    test_preds = []\n    test_labels = []\n    \n    with torch.no_grad():\n        for inputs, labels in tqdm(test_loader, desc='Evaluating on Test Set'):\n            inputs, labels = inputs.to(device), labels.to(device)\n            outputs = model(inputs).squeeze()\n            \n            if outputs.ndim == 0:\n                outputs = outputs.unsqueeze(0)\n                \n            loss = criterion(outputs, labels)\n            test_loss += loss.item() * inputs.size(0)\n            predicted = (outputs > 0.5).float()\n            test_preds.extend(predicted.cpu().numpy())\n            test_labels.extend(labels.cpu().numpy())\n    \n    test_loss = test_loss / len(test_loader.dataset)\n    test_acc = np.mean(np.array(test_preds) == np.array(test_labels))\n    test_precision = precision_score(test_labels, test_preds, zero_division=0)\n    test_recall = recall_score(test_labels, test_preds, zero_division=0)\n    test_f1 = f1_score(test_labels, test_preds, zero_division=0)\n    \n    print(f'Test Loss: {test_loss:.4f}, Acc: {test_acc:.4f}, Precision: {test_precision:.4f}, Recall: {test_recall:.4f}, F1: {test_f1:.4f}')\n\n# Train the model\ntrained_model = train_model(model, train_loader, val_loader, criterion, optimizer, scheduler, num_epochs=3)\n\n# Evaluate on test set\nevaluate_model(trained_model, test_loader, criterion)\n\n# Modified test data processing for inference\nimport shutil\n\n# Modified NexarDataset class with robust error handling\nclass NexarDataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df = df\n        self.transform = transform\n        # Filter out rows with invalid frame paths\n        self.df = self.df[self.df['frame_path'].apply(self._is_valid_file)]\n        print(f\"Filtered dataset to {len(self.df)} valid frames\")\n        \n    def _is_valid_file(self, path):\n        return os.path.isfile(path) and os.access(path, os.R_OK)\n    \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img_path = row['frame_path']\n        label = row['label'] if row['label'] is not None else 0.0\n        \n        try:\n            if not self._is_valid_file(img_path):\n                raise FileNotFoundError(f\"Image file not found or inaccessible: {img_path}\")\n                \n            image = Image.open(img_path).convert('RGB')\n            \n            if self.transform:\n                image = self.transform(image)\n                \n            return image, torch.tensor(label, dtype=torch.float32)\n        except Exception as e:\n            print(f\"Error loading image {img_path}: {e}\")\n            # Return a placeholder image and label\n            placeholder = torch.zeros((3, 224, 224))\n            return placeholder, torch.tensor(label, dtype=torch.float32)\n\n# Load test data\ntest_df = pd.read_csv('/kaggle/input/nexar-collision-prediction/test.csv')\nprint(\"Test data loaded:\", test_df.shape)\n\n# Process a subset of test videos\nsample_test_df = test_df.sample(5, random_state=42) if len(test_df) > 5 else test_df\ntest_frames_dir = '/kaggle/working/test_frames_densenet'\n\n# Clear the test frames directory to avoid conflicts\nif os.path.exists(test_frames_dir):\n    shutil.rmtree(test_frames_dir)\nos.makedirs(test_frames_dir, exist_ok=True)\n\n# Process videos with additional debugging\ntest_frames_df = process_videos(sample_test_df, test_frames_dir, is_train=False)\n\n# Check if we have any test frames\nif len(test_frames_df) == 0:\n    print(\"No test frames were extracted. Creating dummy test data...\")\n    dummy_test_data = []\n    os.makedirs('/kaggle/working/dummy_test_images_densenet', exist_ok=True)\n    for i in range(50):\n        frame_path = f'/kaggle/working/dummy_test_images_densenet/dummy_test_{i:05d}.jpg'\n        img = np.ones((224, 224, 3), dtype=np.uint8) * 128\n        cv2.imwrite(frame_path, img)\n        dummy_test_data.append({\n            'frame_path': frame_path,\n            'video_id': f\"{i % 5:05d}\",\n            'frame_idx': i,\n            'label': None\n        })\n    test_frames_df = pd.DataFrame(dummy_test_data)\n\ntest_frames_df.to_csv('/kaggle/working/test_frames_densenet.csv', index=False)\nprint(f\"Processed {len(test_frames_df)} test frames from {len(sample_test_df)} videos\")\n\n# Create test dataset and loader\ntest_dataset = NexarDataset(test_frames_df, transform=val_transform)\nif len(test_dataset) == 0:\n    raise ValueError(\"Test dataset is empty after filtering invalid frames\")\n\ntest_loader = DataLoader(test_dataset, batch_size=16, shuffle=False, num_workers=0)  # Set num_workers to 0\n\n# Load best model for inference\nmodel.load_state_dict(torch.load('/kaggle/working/best_densenet_model.pth', weights_only=True))\nmodel.eval()\n\n# Make predictions on test set with error handling\npredictions = []\nvideo_ids = []\n\nwith torch.no_grad():\n    for i, (inputs, _) in enumerate(tqdm(test_loader, desc='Making predictions with DenseNet')):\n        try:\n            inputs = inputs.to(device)\n            outputs = model(inputs).squeeze().cpu().numpy()\n            \n            batch_indices = list(range(i*test_loader.batch_size, \n                                     min((i+1)*test_loader.batch_size, len(test_dataset))))\n            \n            actual_batch_size = min(test_loader.batch_size, len(test_dataset) - i*test_loader.batch_size)\n            batch_indices = batch_indices[:actual_batch_size]\n            \n            batch_video_ids = [test_frames_df.iloc[idx]['video_id'] for idx in batch_indices]\n            \n            if isinstance(outputs, np.float32):\n                outputs = np.array([outputs])\n                \n            if len(outputs) > len(batch_indices):\n                outputs = outputs[:len(batch_indices)]\n                \n            predictions.extend(outputs)\n            video_ids.extend(batch_video_ids)\n        except Exception as e:\n            print(f\"Error processing batch {i}: {e}\")\n            continue\n\nif not predictions:\n    raise ValueError(\"No predictions were generated. Check test data and model inference.\")\n\n# Create a dataframe with predictions for each frame\nprediction_df = pd.DataFrame({\n    'video_id': video_ids,\n    'prediction': predictions\n})\n\n# Aggregate predictions by video\nvideo_predictions = prediction_df.groupby('video_id')['prediction'].max().reset_index()\n\n# Create submission file\nsubmission_df = pd.DataFrame({\n    'id': video_predictions['video_id'],\n    'target': video_predictions['prediction']\n})\n\nsubmission_df.to_csv('/kaggle/working/densenet_submission.csv', index=False)\nprint(\"DenseNet submission file created!\")\nprint(submission_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-18T23:26:46.794693Z","iopub.execute_input":"2025-04-18T23:26:46.794986Z","iopub.status.idle":"2025-04-18T23:32:53.857825Z","shell.execute_reply.started":"2025-04-18T23:26:46.794963Z","shell.execute_reply":"2025-04-18T23:32:53.857039Z"}},"outputs":[{"name":"stdout","text":"Training data loaded: (1500, 4)\n","output_type":"stream"},{"name":"stderr","text":"100%|██████████| 10/10 [00:54<00:00,  5.49s/it]\nDownloading: \"https://download.pytorch.org/models/densenet121-a639ec97.pth\" to /root/.cache/torch/hub/checkpoints/densenet121-a639ec97.pth\n","output_type":"stream"},{"name":"stdout","text":"Processed 10534 frames from 10 videos\n","output_type":"stream"},{"name":"stderr","text":"100%|██████████| 30.8M/30.8M [00:00<00:00, 138MB/s] \n/usr/local/lib/python3.11/dist-packages/torch/optim/lr_scheduler.py:62: UserWarning: The verbose parameter is deprecated. Please use get_last_lr() to access the learning rate.\n  warnings.warn(\n","output_type":"stream"},{"name":"stdout","text":"Using device: cuda\n","output_type":"stream"},{"name":"stderr","text":"Epoch 1/3 - Training: 100%|██████████| 474/474 [01:20<00:00,  5.91it/s]\nEpoch 1/3 - Validation: 100%|██████████| 119/119 [00:11<00:00,  9.94it/s]\n","output_type":"stream"},{"name":"stdout","text":"Epoch 1/3:\nTrain Loss: 0.0813, Acc: 0.9740, Precision: 0.8933, Recall: 0.4254, F1: 0.5763\nVal Loss: 0.0230, Acc: 0.9947, Precision: 0.9737, Recall: 0.9024, F1: 0.9367\nSaved best model!\n","output_type":"stream"},{"name":"stderr","text":"Epoch 2/3 - Training: 100%|██████████| 474/474 [01:20<00:00,  5.92it/s]\nEpoch 2/3 - Validation: 100%|██████████| 119/119 [00:11<00:00, 10.18it/s]\n","output_type":"stream"},{"name":"stdout","text":"Epoch 2/3:\nTrain Loss: 0.0205, Acc: 0.9931, Precision: 0.9175, Recall: 0.9175, F1: 0.9175\nVal Loss: 0.0100, Acc: 0.9974, Precision: 0.9529, Recall: 0.9878, F1: 0.9701\nSaved best model!\n","output_type":"stream"},{"name":"stderr","text":"Epoch 3/3 - Training: 100%|██████████| 474/474 [01:20<00:00,  5.92it/s]\nEpoch 3/3 - Validation: 100%|██████████| 119/119 [00:12<00:00,  9.64it/s]\n","output_type":"stream"},{"name":"stdout","text":"Epoch 3/3:\nTrain Loss: 0.0137, Acc: 0.9963, Precision: 0.9527, Recall: 0.9587, F1: 0.9557\nVal Loss: 0.0113, Acc: 0.9963, Precision: 0.9747, Recall: 0.9390, F1: 0.9565\n","output_type":"stream"},{"name":"stderr","text":"Evaluating on Test Set: 100%|██████████| 66/66 [00:06<00:00, 10.15it/s]\n","output_type":"stream"},{"name":"stdout","text":"Test Loss: 0.0093, Acc: 0.9972, Precision: 0.9750, Recall: 0.9512, F1: 0.9630\nTest data loaded: (1344, 1)\n","output_type":"stream"},{"name":"stderr","text":"100%|██████████| 5/5 [00:07<00:00,  1.46s/it]\n","output_type":"stream"},{"name":"stdout","text":"Processed 1459 test frames from 5 videos\nFiltered dataset to 1459 valid frames\n","output_type":"stream"},{"name":"stderr","text":"Making predictions with DenseNet: 100%|██████████| 92/92 [00:20<00:00,  4.43it/s]","output_type":"stream"},{"name":"stdout","text":"DenseNet submission file created!\n      id    target\n0  00390  0.000130\n1  00531  0.004468\n2  01113  0.007501\n3  01246  0.549108\n4  02561  0.032208\n","output_type":"stream"},{"name":"stderr","text":"\n","output_type":"stream"}],"execution_count":19},{"cell_type":"markdown","source":"## VGG16","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as transforms\nfrom PIL import Image\nimport cv2\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import precision_score, recall_score, f1_score\nfrom tqdm import tqdm\nfrom torchvision.models import vgg16, VGG16_Weights\n\n# Constants\nFRAME_RATE = 30  # 30 fps\nWINDOW_SIZE = 5  # Number of frames to use for prediction\nPOSITIVE_WINDOW = 1.5  # Seconds before the event to consider as positive\n\n# Load the training data\ntrain_df = pd.read_csv('/kaggle/input/nexar-collision-prediction/train.csv')\nprint(\"Training data loaded:\", train_df.shape)\n\n# Fix the video path issue by ensuring proper formatting of video IDs\ndef get_video_path(video_id, is_train=True):\n    formatted_id = f\"{int(video_id):05d}\"\n    base_dir = '/kaggle/input/nexar-collision-prediction'\n    subfolder = 'train' if is_train else 'test'\n    return os.path.join(base_dir, subfolder, f\"{formatted_id}.mp4\")\n\n# Data preprocessing function to extract frames\ndef extract_frames(video_id, df_row, output_dir, is_train=True):\n    formatted_id = f\"{int(video_id):05d}\"\n    video_path = get_video_path(video_id, is_train)\n    target = df_row['target'] if 'target' in df_row else None\n    \n    os.makedirs(output_dir, exist_ok=True)\n    \n    cap = cv2.VideoCapture(video_path)\n    if not cap.isOpened():\n        print(f\"Error opening video: {video_path}\")\n        return []\n    \n    frame_count = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    fps = cap.get(cv2.CAP_PROP_FPS)\n    \n    event_frame = None\n    alert_frame = None\n    \n    if is_train and target == 1:\n        event_time = float(df_row['time_of_event'])\n        alert_time = float(df_row['time_of_alert'])\n        event_frame = int(event_time * fps)\n        alert_frame = int(alert_time * fps)\n    \n    frame_data = []\n    \n    for frame_idx in range(frame_count):\n        ret, frame = cap.read()\n        if not ret:\n            break\n        \n        label = 0\n        if is_train and target == 1:\n            if alert_frame <= frame_idx <= event_frame:\n                label = 1\n            elif alert_frame - int(POSITIVE_WINDOW * fps) <= frame_idx < alert_frame:\n                label = 1\n        \n        frame_path = os.path.join(output_dir, f\"{formatted_id}_{frame_idx:05d}.jpg\")\n        cv2.imwrite(frame_path, frame)\n        \n        frame_data.append({\n            'frame_path': frame_path,\n            'video_id': formatted_id,\n            'frame_idx': frame_idx,\n            'label': label if is_train else None\n        })\n    \n    cap.release()\n    return frame_data\n\n# Process all videos and create a dataframe of frames\n# Modified training data processing to focus on positive cases (limited to 200)\ndef process_training_videos_balanced(df, output_dir, is_train=True, max_positive_videos=20):\n    all_frames = []\n    \n    # Get positive videos (limited to max_positive_videos)\n    positive_videos = df[df['target'] == 1]\n    if len(positive_videos) > max_positive_videos:\n        positive_videos = positive_videos.sample(max_positive_videos, random_state=42)\n    \n    \n    print(f\"Processing {len(positive_videos)} negative videos\")\n    \n    # Process all selected negative videos\n    for _, row in tqdm(positive_videos.iterrows(), total=len(positive_videos)):\n        video_id = row['id']\n        frame_data = extract_frames(video_id, row, output_dir, is_train)\n        all_frames.extend(frame_data)\n    \n    frames_df = pd.DataFrame(all_frames)\n    \n    # Print class distribution\n    if is_train:\n        negative_frames = frames_df[frames_df['label'] == 0].shape[0]\n        print(f\"Extracted {negative_frames} negative frames\")\n    \n    return frames_df\n\n# Replace the previous processing call with this balanced version\ntrain_frames_df = process_training_videos_balanced(train_df, '/kaggle/working/train_frames_vgg', \n                                                 is_train=True, max_positive_videos=20)\n\n\n# Check if we have any frames before proceeding\nif len(train_frames_df) == 0:\n    print(\"No frames were extracted. Creating dummy dataset...\")\n    dummy_data = []\n    for i in range(100):\n        dummy_data.append({\n            'frame_path': f'/kaggle/working/dummy_vgg_{i:05d}.jpg',\n            'video_id': f\"{i % 5:05d}\",\n            'frame_idx': i,\n            'label': i % 2\n        })\n    train_frames_df = pd.DataFrame(dummy_data)\n    os.makedirs('/kaggle/working/dummy_images_vgg', exist_ok=True)\n    for i in range(100):\n        img = np.ones((224, 224, 3), dtype=np.uint8) * 128\n        cv2.imwrite(f'/kaggle/working/dummy_vgg_{i:05d}.jpg', img)\n\ntrain_frames_df.to_csv('/kaggle/working/train_frames_vgg.csv', index=False)\n\n# Custom Dataset class\nclass NexarDataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df = df\n        self.transform = transform\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img_path = row['frame_path']\n        label = row['label']\n        \n        try:\n            image = Image.open(img_path).convert('RGB')\n            if self.transform:\n                image = self.transform(image)\n            return image, torch.tensor(label, dtype=torch.float32)\n        except Exception as e:\n            print(f\"Error loading image {img_path}: {e}\")\n            placeholder = torch.zeros((3, 224, 224))\n            return placeholder, torch.tensor(0, dtype=torch.float32)\n\n# Define transforms for VGG-16 - Standard size is 224x224\ntrain_transform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomRotation(10),\n    transforms.ColorJitter(brightness=0.2, contrast=0.2),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\nval_transform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\n# Split data into train, validation, and test sets\ntrain_val_frames, test_frames = train_test_split(train_frames_df, test_size=0.1, random_state=42)\ntrain_frames, val_frames = train_test_split(train_val_frames, test_size=0.2, random_state=42)\n\n# Create datasets\ntrain_dataset = NexarDataset(train_frames, transform=train_transform)\nval_dataset = NexarDataset(val_frames, transform=val_transform)\ntest_dataset = NexarDataset(test_frames, transform=val_transform)\n\n# Create data loaders\ntrain_loader = DataLoader(train_dataset, batch_size=2, shuffle=True, num_workers=2)\nval_loader = DataLoader(val_dataset, batch_size=16, shuffle=False, num_workers=2)\ntest_loader = DataLoader(test_dataset, batch_size=16, shuffle=False, num_workers=2)\n\n# Initialize VGG-16 model\ndef get_vgg16_model():\n    # Load pre-trained VGG16 model\n    model = vgg16(weights=VGG16_Weights.IMAGENET1K_V1)\n    \n    # Modify the classifier for binary classification\n    # VGG16's classifier has 6 layers, we'll replace the last layer\n    model.classifier[6] = nn.Sequential(\n        nn.Linear(4096, 256),\n        nn.ReLU(),\n        nn.Dropout(0.5),\n        nn.Linear(256, 1),\n        nn.Sigmoid()\n    )\n    \n    return model\n\nmodel = get_vgg16_model()\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\nmodel = model.to(device)\n\n# Define loss function and optimizer\ncriterion = nn.BCELoss()\n# For VGG, we'll use a slightly higher learning rate than ViT\noptimizer = optim.Adam(model.parameters(), lr=1e-4, weight_decay=0.01)\nscheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=3, verbose=True)\n\n# Training function with additional metrics\ndef train_model(model, train_loader, val_loader, criterion, optimizer, scheduler, num_epochs=3):\n    best_val_loss = float('inf')\n    \n    for epoch in range(num_epochs):\n        # Training phase\n        model.train()\n        running_loss = 0.0\n        train_preds = []\n        train_labels = []\n        \n        for inputs, labels in tqdm(train_loader, desc=f'Epoch {epoch+1}/{num_epochs} - Training'):\n            inputs, labels = inputs.to(device), labels.to(device)\n            \n            optimizer.zero_grad()\n            outputs = model(inputs).squeeze()\n            \n            if outputs.ndim == 0:\n                outputs = outputs.unsqueeze(0)\n                \n            loss = criterion(outputs, labels)\n            loss.backward()\n            optimizer.step()\n            \n            running_loss += loss.item() * inputs.size(0)\n            predicted = (outputs > 0.5).float()\n            train_preds.extend(predicted.cpu().numpy())\n            train_labels.extend(labels.cpu().numpy())\n        \n        epoch_loss = running_loss / len(train_loader.dataset)\n        epoch_acc = np.mean(np.array(train_preds) == np.array(train_labels))\n        epoch_precision = precision_score(train_labels, train_preds, zero_division=0)\n        epoch_recall = recall_score(train_labels, train_preds, zero_division=0)\n        epoch_f1 = f1_score(train_labels, train_preds, zero_division=0)\n        \n        # Validation phase\n        model.eval()\n        val_loss = 0.0\n        val_preds = []\n        val_labels = []\n        \n        with torch.no_grad():\n            for inputs, labels in tqdm(val_loader, desc=f'Epoch {epoch+1}/{num_epochs} - Validation'):\n                inputs, labels = inputs.to(device), labels.to(device)\n                outputs = model(inputs).squeeze()\n                \n                if outputs.ndim == 0:\n                    outputs = outputs.unsqueeze(0)\n                    \n                loss = criterion(outputs, labels)\n                val_loss += loss.item() * inputs.size(0)\n                predicted = (outputs > 0.5).float()\n                val_preds.extend(predicted.cpu().numpy())\n                val_labels.extend(labels.cpu().numpy())\n        \n        val_epoch_loss = val_loss / len(val_loader.dataset)\n        val_epoch_acc = np.mean(np.array(val_preds) == np.array(val_labels))\n        val_epoch_precision = precision_score(val_labels, val_preds, zero_division=0)\n        val_epoch_recall = recall_score(val_labels, val_preds, zero_division=0)\n        val_epoch_f1 = f1_score(val_labels, val_preds, zero_division=0)\n        \n        scheduler.step(val_epoch_loss)\n        \n        print(f'Epoch {epoch+1}/{num_epochs}:')\n        print(f'Train Loss: {epoch_loss:.4f}, Acc: {epoch_acc:.4f}, Precision: {epoch_precision:.4f}, Recall: {epoch_recall:.4f}, F1: {epoch_f1:.4f}')\n        print(f'Val Loss: {val_epoch_loss:.4f}, Acc: {val_epoch_acc:.4f}, Precision: {val_epoch_precision:.4f}, Recall: {val_epoch_recall:.4f}, F1: {val_epoch_f1:.4f}')\n        \n        if val_epoch_loss < best_val_loss:\n            best_val_loss = val_epoch_loss\n            torch.save(model.state_dict(), '/kaggle/working/best_vgg16_model.pth')\n            print(\"Saved best model!\")\n    \n    return model\n\n# Evaluate model on test set\ndef evaluate_model(model, test_loader, criterion):\n    model.eval()\n    test_loss = 0.0\n    test_preds = []\n    test_labels = []\n    \n    with torch.no_grad():\n        for inputs, labels in tqdm(test_loader, desc='Evaluating on Test Set'):\n            inputs, labels = inputs.to(device), labels.to(device)\n            outputs = model(inputs).squeeze()\n            \n            if outputs.ndim == 0:\n                outputs = outputs.unsqueeze(0)\n                \n            loss = criterion(outputs, labels)\n            test_loss += loss.item() * inputs.size(0)\n            predicted = (outputs > 0.5).float()\n            test_preds.extend(predicted.cpu().numpy())\n            test_labels.extend(labels.cpu().numpy())\n    \n    test_loss = test_loss / len(test_loader.dataset)\n    test_acc = np.mean(np.array(test_preds) == np.array(test_labels))\n    test_precision = precision_score(test_labels, test_preds, zero_division=0)\n    test_recall = recall_score(test_labels, test_preds, zero_division=0)\n    test_f1 = f1_score(test_labels, test_preds, zero_division=0)\n    \n    print(f'Test Loss: {test_loss:.4f}, Acc: {test_acc:.4f}, Precision: {test_precision:.4f}, Recall: {test_recall:.4f}, F1: {test_f1:.4f}')\n\n# Train the model\ntrained_model = train_model(model, train_loader, val_loader, criterion, optimizer, scheduler, num_epochs=3)\n\n# Evaluate on test set\nevaluate_model(trained_model, test_loader, criterion)\n\n# Modified test data processing for inference\nimport shutil\n\n# Modified NexarDataset class with robust error handling\nclass NexarDataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df = df\n        self.transform = transform\n        # Filter out rows with invalid frame paths\n        self.df = self.df[self.df['frame_path'].apply(self._is_valid_file)]\n        print(f\"Filtered dataset to {len(self.df)} valid frames\")\n        \n    def _is_valid_file(self, path):\n        return os.path.isfile(path) and os.access(path, os.R_OK)\n    \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img_path = row['frame_path']\n        label = row['label'] if row['label'] is not None else 0.0\n        \n        try:\n            if not self._is_valid_file(img_path):\n                raise FileNotFoundError(f\"Image file not found or inaccessible: {img_path}\")\n                \n            image = Image.open(img_path).convert('RGB')\n            \n            if self.transform:\n                image = self.transform(image)\n                \n            return image, torch.tensor(label, dtype=torch.float32)\n        except Exception as e:\n            print(f\"Error loading image {img_path}: {e}\")\n            # Return a placeholder image and label\n            placeholder = torch.zeros((3, 224, 224))\n            return placeholder, torch.tensor(label, dtype=torch.float32)\n\n# Load test data\ntest_df = pd.read_csv('/kaggle/input/nexar-collision-prediction/test.csv')\nprint(\"Test data loaded:\", test_df.shape)\n\n# Process a subset of test videos\nsample_test_df = test_df.sample(2, random_state=42) if len(test_df) > 2 else test_df\ntest_frames_dir = '/kaggle/working/test_frames_vgg16'\n\n# Clear the test frames directory to avoid conflicts\nif os.path.exists(test_frames_dir):\n    shutil.rmtree(test_frames_dir)\nos.makedirs(test_frames_dir, exist_ok=True)\n\n# Process videos with additional debugging\ndef process_videos(df, output_dir, is_train=False):\n    all_frames = []\n    \n    for _, row in tqdm(df.iterrows(), total=len(df), desc=\"Processing test videos\"):\n        video_id = row['id']\n        frame_data = extract_frames(video_id, row, output_dir, is_train)\n        # Validate saved frames\n        valid_frames = []\n        for frame in frame_data:\n            if os.path.isfile(frame['frame_path']) and os.access(frame['frame_path'], os.R_OK):\n                valid_frames.append(frame)\n            else:\n                print(f\"Invalid frame file: {frame['frame_path']}\")\n        all_frames.extend(valid_frames)\n    \n    return pd.DataFrame(all_frames)\n\ntest_frames_df = process_videos(sample_test_df, test_frames_dir, is_train=False)\n\n# Check if we have any test frames\nif len(test_frames_df) == 0:\n    print(\"No test frames were extracted. Creating dummy test data...\")\n    dummy_test_data = []\n    os.makedirs('/kaggle/working/dummy_test_images_vgg16', exist_ok=True)\n    for i in range(50):\n        frame_path = f'/kaggle/working/dummy_test_images_vgg16/dummy_test_{i:05d}.jpg'\n        img = np.ones((224, 224, 3), dtype=np.uint8) * 128\n        cv2.imwrite(frame_path, img)\n        dummy_test_data.append({\n            'frame_path': frame_path,\n            'video_id': f\"{i % 5:05d}\",\n            'frame_idx': i,\n            'label': None\n        })\n    test_frames_df = pd.DataFrame(dummy_test_data)\n\ntest_frames_df.to_csv('/kaggle/working/test_frames_vgg16.csv', index=False)\nprint(f\"Processed {len(test_frames_df)} test frames from {len(sample_test_df)} videos\")\n\n# Create test dataset and loader\ntest_dataset = NexarDataset(test_frames_df, transform=val_transform)\nif len(test_dataset) == 0:\n    raise ValueError(\"Test dataset is empty after filtering invalid frames\")\n\ntest_loader = DataLoader(test_dataset, batch_size=16, shuffle=False, num_workers=0)  # Set num_workers to 0\n\n# Load best model for inference\nmodel.load_state_dict(torch.load('/kaggle/working/best_vgg16_model.pth', weights_only=True))  # Set weights_only=True\nmodel.eval()\n\n# Make predictions on test set with error handling\npredictions = []\nvideo_ids = []\n\nwith torch.no_grad():\n    for i, (inputs, _) in enumerate(tqdm(test_loader, desc='Making predictions with VGG-16')):\n        try:\n            inputs = inputs.to(device)\n            outputs = model(inputs).squeeze().cpu().numpy()\n            \n            batch_indices = list(range(i*test_loader.batch_size, \n                                     min((i+1)*test_loader.batch_size, len(test_dataset))))\n            \n            actual_batch_size = min(test_loader.batch_size, len(test_dataset) - i*test_loader.batch_size)\n            batch_indices = batch_indices[:actual_batch_size]\n            \n            batch_video_ids = [test_frames_df.iloc[idx]['video_id'] for idx in batch_indices]\n            \n            if isinstance(outputs, np.float32):\n                outputs = np.array([outputs])\n                \n            if len(outputs) > len(batch_indices):\n                outputs = outputs[:len(batch_indices)]\n                \n            predictions.extend(outputs)\n            video_ids.extend(batch_video_ids)\n        except Exception as e:\n            print(f\"Error processing batch {i}: {e}\")\n            continue\n\nif not predictions:\n    raise ValueError(\"No predictions were generated. Check test data and model inference.\")\n\n# Create a dataframe with predictions for each frame\nprediction_df = pd.DataFrame({\n    'video_id': video_ids,\n    'prediction': predictions\n})\n\n# Aggregate predictions by video\nvideo_predictions = prediction_df.groupby('video_id')['prediction'].max().reset_index()\n\n# Create submission file\nsubmission_df = pd.DataFrame({\n    'id': video_predictions['video_id'],\n    'target': video_predictions['prediction']\n})\n\nsubmission_df.to_csv('/kaggle/working/vgg16_submission.csv', index=False)\nprint(\"VGG-16 submission file created!\")\nprint(submission_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T03:31:48.442851Z","iopub.execute_input":"2025-04-19T03:31:48.443096Z","execution_failed":"2025-04-19T05:59:47.035Z"}},"outputs":[{"name":"stdout","text":"Training data loaded: (1500, 4)\nProcessing 20 negative videos\n","output_type":"stream"},{"name":"stderr","text":"100%|██████████| 20/20 [02:11<00:00,  6.55s/it]\n","output_type":"stream"},{"name":"stdout","text":"Extracted 21890 negative frames\n","output_type":"stream"},{"name":"stderr","text":"Downloading: \"https://download.pytorch.org/models/vgg16-397923af.pth\" to /root/.cache/torch/hub/checkpoints/vgg16-397923af.pth\n100%|██████████| 528M/528M [00:07<00:00, 76.0MB/s] \n","output_type":"stream"},{"name":"stdout","text":"Using device: cuda\n","output_type":"stream"},{"name":"stderr","text":"/usr/local/lib/python3.11/dist-packages/torch/optim/lr_scheduler.py:62: UserWarning: The verbose parameter is deprecated. Please use get_last_lr() to access the learning rate.\n  warnings.warn(\nEpoch 1/3 - Training: 100%|██████████| 8540/8540 [13:16<00:00, 10.72it/s]\nEpoch 1/3 - Validation: 100%|██████████| 267/267 [00:29<00:00,  8.96it/s]\n","output_type":"stream"},{"name":"stdout","text":"Epoch 1/3:\nTrain Loss: 0.3249, Acc: 0.9222, Precision: 0.0000, Recall: 0.0000, F1: 0.0000\nVal Loss: 0.2824, Acc: 0.9244, Precision: 0.0000, Recall: 0.0000, F1: 0.0000\nSaved best model!\n","output_type":"stream"},{"name":"stderr","text":"Epoch 2/3 - Training:  88%|████████▊ | 7525/8540 [11:31<01:32, 10.95it/s]","output_type":"stream"}],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}