{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":92399,"databundleVersionId":11038207,"sourceType":"competition"},{"sourceId":397332,"sourceType":"modelInstanceVersion","modelInstanceId":325814,"modelId":346666},{"sourceId":397334,"sourceType":"modelInstanceVersion","modelInstanceId":325815,"modelId":346667}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install --upgrade pip\n!pip install ultralytics\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T12:16:33.332498Z","iopub.execute_input":"2025-05-15T12:16:33.332778Z","iopub.status.idle":"2025-05-15T12:16:37.449841Z","shell.execute_reply.started":"2025-05-15T12:16:33.332757Z","shell.execute_reply":"2025-05-15T12:16:37.448901Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import math\nimport os\nimport shutil\nfrom collections import deque\n\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torchvision.models as models\nimport torchvision.transforms as T\nfrom pytorch_lightning import LightningDataModule, LightningModule\nfrom sklearn.model_selection import train_test_split\nfrom torch.utils.data import DataLoader, Dataset\nfrom torchmetrics.classification import Accuracy\nfrom tqdm import tqdm\nfrom ultralytics import YOLO\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T12:16:37.451444Z","iopub.execute_input":"2025-05-15T12:16:37.451702Z","iopub.status.idle":"2025-05-15T12:16:43.720962Z","shell.execute_reply.started":"2025-05-15T12:16:37.451677Z","shell.execute_reply":"2025-05-15T12:16:43.720419Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Documentation:**\n- Explanation of solution: https://www.kaggle.com/competitions/nexar-collision-prediction/discussion/578712\n- Repository with training code: https://github.com/maxzw/nexar-collision-prediction\n","metadata":{}},{"cell_type":"markdown","source":"## Process test data","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv(\"/kaggle/input/nexar-collision-prediction/test.csv\")\nprint(f\"Test shape: {test_df.shape}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T12:16:43.721703Z","iopub.execute_input":"2025-05-15T12:16:43.722070Z","iopub.status.idle":"2025-05-15T12:16:43.728801Z","shell.execute_reply.started":"2025-05-15T12:16:43.722045Z","shell.execute_reply":"2025-05-15T12:16:43.728276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device(\n    \"cuda\" if torch.cuda.is_available()\n    else \"mps\" if torch.backends.mps.is_available()\n    else \"cpu\"\n)\nprint(f\"Using device: {device}\")\n\nmodel = YOLO(\"yolov8m-seg.pt\")\nmodel.to(device)\nvehicle_classes = {\"car\", \"truck\", \"bus\", \"motorbike\", \"bicycle\"}\n\n@torch.no_grad()\ndef extract_mask(frame: np.ndarray, model: YOLO) -> np.ndarray | None:\n    \"\"\"Extracts vehicle masks from a given frame using the YOLO model.\n\n    If no vehicle masks are found, returns None.\n    \"\"\"\n    model.eval()\n\n    results = model(frame, verbose=False)[0]\n    masks = results.masks\n    classes = results.boxes.cls\n\n    if masks is None or masks.data is None or len(masks.data) == 0:\n        return None\n\n    H, W = frame.shape[:2]\n    mask_out = np.zeros((H, W), dtype=np.uint8)\n\n    # Convert once\n    masks_np = masks.data.cpu().numpy().astype(np.uint8)\n    classes_np = classes.cpu().numpy()\n\n    for seg, cls_id in zip(masks_np, classes_np):\n        class_name = model.model.names[int(cls_id)]\n        if class_name in vehicle_classes:\n            mask = cv2.resize(seg, (W, H), interpolation=cv2.INTER_NEAREST)\n            mask_out[mask > 0] = 255\n\n    return mask_out[None, ...]  # Shape: (1, H, W)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T12:16:43.730123Z","iopub.execute_input":"2025-05-15T12:16:43.730345Z","iopub.status.idle":"2025-05-15T12:16:44.035736Z","shell.execute_reply.started":"2025-05-15T12:16:43.730329Z","shell.execute_reply":"2025-05-15T12:16:44.035158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def compute_flow_channels(frame1: np.ndarray, frame2: np.ndarray) -> np.ndarray:\n    \"\"\"Compute optical flow channels between two frames.\n\n    Flow is returned as a 2-channel image with shape (2, H, W):\n    - Channel 0: Magnitude of flow\n    - Channel 1: Angle of flow\n    \"\"\"\n    # Convert to grayscale\n    prev_gray = cv2.cvtColor(frame1, cv2.COLOR_RGB2GRAY)\n    next_gray = cv2.cvtColor(frame2, cv2.COLOR_RGB2GRAY)\n\n    # Compute Dense Optical Flow (Farneback)\n    flow = cv2.calcOpticalFlowFarneback(prev_gray, next_gray, None, 0.5, 3, 15, 3, 5, 1.2, 0)\n\n    # Compute magnitude and direction\n    magnitude, angle = cv2.cartToPolar(flow[..., 0], flow[..., 1])\n\n    # Normalize magnitude and angle\n    magnitude = cv2.normalize(magnitude, None, 0, 255, cv2.NORM_MINMAX)\n    angle = (angle * 180 / np.pi / 2).astype(np.uint8)\n\n    # Stack into a single 2-channel image\n    flow_channels = np.dstack((magnitude, angle))\n\n    # Reorder dimensions to (2, H, W)\n    return np.moveaxis(flow_channels, -1, 0)  # Shape: (2, H, W)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T12:16:44.036461Z","iopub.execute_input":"2025-05-15T12:16:44.036648Z","iopub.status.idle":"2025-05-15T12:16:44.042200Z","shell.execute_reply.started":"2025-05-15T12:16:44.036633Z","shell.execute_reply":"2025-05-15T12:16:44.041503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def downsample_image(image: np.ndarray, factor: int) -> np.ndarray:\n    \"\"\"Downsample imput image by a given factor.\n\n    Accepts images in both (C, H, W) and (H, W, C) formats, but returns in (C, H, W) format.\n    \"\"\"\n    if image.shape[0] <= 3:\n        # If the image is in (C, H, W) format, transpose to (H, W, C) for resizing\n        image = np.transpose(image, (1, 2, 0))\n\n    # Downsample the image\n    h, w = image.shape[:2]\n    downsampled_image = cv2.resize(image, (w // factor, h // factor), interpolation=cv2.INTER_LINEAR)\n\n    # If downsampled image now has two dimensions, convert to 1 channels (H, W, 1)\n    if len(downsampled_image.shape) == 2:\n        downsampled_image = np.expand_dims(downsampled_image, axis=-1)\n\n    # Make sure to resize back to (C, H, W) format\n    return np.transpose(downsampled_image, (2, 0, 1))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T12:16:44.042800Z","iopub.execute_input":"2025-05-15T12:16:44.042975Z","iopub.status.idle":"2025-05-15T12:16:44.055214Z","shell.execute_reply.started":"2025-05-15T12:16:44.042954Z","shell.execute_reply":"2025-05-15T12:16:44.054619Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"FRAMES_NEEDED = 6  # 3 for optical flow context + 3 to process\n\ndq = deque(maxlen=3)\nfor row in tqdm(test_df.to_dict(orient=\"records\")):\n    video_path = f\"/kaggle/input/nexar-collision-prediction/test/{str(row['id']).zfill(5)}.mp4\"\n    video = cv2.VideoCapture(video_path)\n\n    # Get total number of frames\n    total_frames = int(video.get(cv2.CAP_PROP_FRAME_COUNT))\n    start_frame = max(0, total_frames - FRAMES_NEEDED)\n\n    # Seek to frame N - 6\n    video.set(cv2.CAP_PROP_POS_FRAMES, start_frame)\n    frame_index = start_frame\n    save_index = 0\n\n    while True:\n        ret, frame = video.read()\n        if not ret:\n            break\n\n        frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n        dq.append(frame)\n\n        # Only save the *last 3 frames* (from N-3 to N-1)\n        if frame_index >= total_frames - 3 and len(dq) == 3:\n            mask = extract_mask(frame, model)           # Shape: (1, H, W)\n            flow = compute_flow_channels(dq[0], frame)  # Shape: (2, H, W)\n\n            flow_downsampled = downsample_image(flow, 3)\n            frame_downsampled = downsample_image(frame, 3)\n\n            save_dir = f\"/kaggle/working/processed/test/{str(row['id']).zfill(5)}\"\n            os.makedirs(save_dir, exist_ok=True)\n\n            flow_path = save_dir + f\"/flows/{str(save_index).zfill(2)}.pt\"\n            os.makedirs(os.path.dirname(flow_path), exist_ok=True)\n            torch.save(torch.tensor(flow_downsampled).to(dtype=torch.float32), flow_path)\n\n            frame_path = save_dir + f\"/frames/{str(save_index).zfill(2)}.pt\"\n            os.makedirs(os.path.dirname(frame_path), exist_ok=True)\n            torch.save(torch.tensor(frame_downsampled).to(dtype=torch.int16), frame_path)\n\n            if mask is not None:\n                mask_downsampled = downsample_image(mask, 3)\n                mask_path = save_dir + f\"/masks/{str(save_index).zfill(2)}.pt\"\n                os.makedirs(os.path.dirname(mask_path), exist_ok=True)\n                torch.save(torch.tensor(mask_downsampled).to(dtype=torch.int16), mask_path)\n\n            save_index += 1\n\n        frame_index += 1\n\n    video.release()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T12:18:44.887385Z","iopub.execute_input":"2025-05-15T12:18:44.888053Z","iopub.status.idle":"2025-05-15T12:18:44.892194Z","shell.execute_reply.started":"2025-05-15T12:18:44.888018Z","shell.execute_reply":"2025-05-15T12:18:44.891335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Add path to features\ntest_df[\"features_path\"] = test_df[\"id\"].apply(lambda x: f\"/kaggle/working/processed/test/{str(x).zfill(5)}\")\n\n# Add number of frames (for sampling)\ntest_df[\"n_frames\"] = test_df[\"features_path\"].apply(lambda x: len(os.listdir(x + \"/frames\")))\n\ntest_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T12:17:19.766273Z","iopub.execute_input":"2025-05-15T12:17:19.766576Z","iopub.status.idle":"2025-05-15T12:17:19.799829Z","shell.execute_reply.started":"2025-05-15T12:17:19.766552Z","shell.execute_reply":"2025-05-15T12:17:19.799190Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Dataset","metadata":{}},{"cell_type":"code","source":"class NexarDataset(Dataset):\n    def __init__(\n        self,\n        df: pd.DataFrame,\n        frame_idx: int | None = None,\n        return_label: bool = True,\n        transform=None,\n    ) -> None:\n        self.df = df\n        self.frame_idx = frame_idx\n        self.return_label = return_label\n        self.transform = transform\n\n        self.features_path = df[\"features_path\"].values\n        self.n_frames = df[\"n_frames\"].values\n        if self.return_label:\n            self.labels = df[\"target\"].values\n\n    def __len__(self) -> int:\n        return len(self.df)\n\n    def __getitem__(self, idx: int):\n        data = {\"idx\": idx}\n\n        # Sample frame\n        frame_idx = (\n            self.frame_idx\n            if self.frame_idx is not None\n            else np.random.randint(0, self.n_frames[idx] - 1)\n        )\n        data[\"frame_idx\"] = frame_idx\n\n        # Load features\n        folder = self.features_path[idx]\n        frame = torch.load(folder + f\"/frames/{str(frame_idx).zfill(2)}.pt\")\n        flow = torch.load(folder + f\"/flows/{str(frame_idx).zfill(2)}.pt\")\n        try:\n            mask = torch.load(folder + f\"/masks/{str(frame_idx).zfill(2)}.pt\")\n        except FileNotFoundError:\n            mask = torch.zeros((1, *flow.shape[1:]))\n\n        # Multiply by mask\n        frame = frame * (mask > 0).float()\n        flow = flow * (mask > 0).float()\n        mask_flow = torch.cat([flow, mask], dim=0)\n\n        # Apply transformations\n        if self.transform:\n            frame = self.apply_transform(frame)\n            mask_flow = self.apply_transform(mask_flow)\n        data[\"frame\"] = frame.to(torch.float32)\n        data[\"mask_flow\"] = mask_flow.to(torch.float32)\n\n        if self.return_label:\n            data[\"label\"] = self.labels[idx]\n\n        return data\n\n    def apply_transform(self, image: torch.Tensor) -> torch.Tensor:\n        if image.dtype != torch.float32:\n            image = image.float()\n        if image.max() > 1.0:\n            image = image / 255.0\n        return self.transform(image)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T12:17:19.801024Z","iopub.execute_input":"2025-05-15T12:17:19.801321Z","iopub.status.idle":"2025-05-15T12:17:19.810453Z","shell.execute_reply.started":"2025-05-15T12:17:19.801301Z","shell.execute_reply":"2025-05-15T12:17:19.809640Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def pad_to_square(image: torch.Tensor):\n    \"\"\"Pad the image to a square shape by adding zeros to all sides.\"\"\"\n    _, h, w = image.shape\n    max_dim = max(w, h)\n    pad_w = (max_dim - w) // 2\n    pad_h = (max_dim - h) // 2\n\n    # Padding format for torch.nn.functional.pad is (left, right, top, bottom)\n    padding = (pad_w, max_dim - w - pad_w, pad_h, max_dim - h - pad_h)\n\n    # Pad expects input as (N, C, H, W) or (C, H, W)\n    return nn.functional.pad(image, padding, mode=\"constant\", value=0)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T12:17:19.811124Z","iopub.execute_input":"2025-05-15T12:17:19.811287Z","iopub.status.idle":"2025-05-15T12:17:19.825429Z","shell.execute_reply.started":"2025-05-15T12:17:19.811274Z","shell.execute_reply":"2025-05-15T12:17:19.824761Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model","metadata":{}},{"cell_type":"code","source":"def build_mlp(\n    n_in: int,\n    n_out: int,\n    hidden_layers: list[int],\n    activation_fn: type[nn.Module] = nn.ReLU,\n    dropout: float | None = None,\n) -> nn.Module:\n    \"\"\"Build a simple MLP.\"\"\"\n    layers = []\n    prev_units = n_in\n\n    for units in hidden_layers:\n        layers.append(nn.Linear(prev_units, units))\n        layers.append(activation_fn())\n        if dropout:\n            layers.append(nn.Dropout(dropout))\n        prev_units = units\n\n    layers.append(nn.Linear(prev_units, n_out))\n\n    return nn.Sequential(*layers)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T12:17:19.827170Z","iopub.execute_input":"2025-05-15T12:17:19.827418Z","iopub.status.idle":"2025-05-15T12:17:19.837912Z","shell.execute_reply.started":"2025-05-15T12:17:19.827400Z","shell.execute_reply":"2025-05-15T12:17:19.837215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class NexarClassifier(LightningModule):\n    def __init__(\n        self,\n        lr: float = 1e-3,\n        hidden_layers: list[int] = [],\n        dropout: float | None = None,\n    ) -> None:\n        super().__init__()\n        self.save_hyperparameters()\n\n        # Image backbone\n        pretrained_weights = models.ResNet34_Weights.DEFAULT\n        self.image_backbone = models.resnet34(weights=pretrained_weights)\n        image_backbone_features = self.image_backbone.fc.in_features\n        self.image_backbone.fc = nn.Linear(\n            in_features=image_backbone_features,\n            out_features=image_backbone_features,\n        )\n        for param in self.image_backbone.parameters():\n            param.requires_grad = False\n        for param in self.image_backbone.fc.parameters():\n            param.requires_grad = True\n\n        # Mask flow backbone\n        self.mask_flow_backbone = models.resnet34(weights=pretrained_weights)\n        mask_flow_backbone_features = self.mask_flow_backbone.fc.in_features\n        self.mask_flow_backbone.fc = nn.Identity()\n\n        # Classifier head\n        self.classifier = build_mlp(\n            n_in=image_backbone_features + mask_flow_backbone_features,\n            n_out=1,\n            hidden_layers=hidden_layers,\n            dropout=dropout,\n        )\n\n        # Loss function and accuracy metric\n        self.loss_fn = nn.BCEWithLogitsLoss()\n        self.train_accuracy = Accuracy(task=\"binary\")\n        self.val_accuracy = Accuracy(task=\"binary\")\n\n    def forward(self, x):\n        image, mask_flow = x[\"frame\"], x[\"mask_flow\"]\n        img_emb = self.image_backbone(image)\n        mf_emb = self.mask_flow_backbone(mask_flow)\n        emb = torch.cat([img_emb, mf_emb], dim=1)\n        return self.classifier(emb)\n\n    def training_step(self, batch, batch_idx):\n        labels = batch[\"label\"]\n        pred = self(batch)\n\n        # Compute training loss\n        loss = self.loss_fn(pred.squeeze(), labels.float())\n        self.log(\"train_loss\", loss, prog_bar=True, on_epoch=True)\n\n        # Compute training accuracy\n        self.train_accuracy(pred.squeeze(), labels)\n        self.log(\"train_acc\", self.train_accuracy, prog_bar=True, on_epoch=True)\n\n        return loss\n\n    def validation_step(self, batch, batch_idx):\n        labels = batch[\"label\"]\n        pred = self(batch)\n\n        # Compute validation loss\n        loss = self.loss_fn(pred.squeeze(), labels.float())\n        self.log(\"val_loss\", loss, prog_bar=True, on_epoch=True)\n\n        # Compute validation accuracy\n        self.val_accuracy(pred.squeeze(), labels)\n        self.log(\"val_acc\", self.val_accuracy, prog_bar=True, on_epoch=True)\n\n        return loss\n\n    def configure_optimizers(self):\n        optimizer = torch.optim.AdamW(\n            [\n                {\"params\": self.image_backbone.parameters(), \"lr\": self.hparams.lr * 0.1},      # Lower LR for backbone\n                {\"params\": self.mask_flow_backbone.parameters(), \"lr\": self.hparams.lr * 0.1},  # Lower LR for backbone\n                {\"params\": self.classifier.parameters(), \"lr\": self.hparams.lr},                # Default LR for classifier\n            ],\n            lr=self.hparams.lr,\n        )  # fmt: skip\n        return {\"optimizer\": optimizer}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-15T12:17:19.838679Z","iopub.execute_input":"2025-05-15T12:17:19.838934Z","iopub.status.idle":"2025-05-15T12:17:19.852307Z","shell.execute_reply.started":"2025-05-15T12:17:19.838912Z","shell.execute_reply":"2025-05-15T12:17:19.851678Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Run inference","metadata":{}},{"cell_type":"code","source":"model_paths = [\n    \"/kaggle/input/nexar_cp21sgvt/pytorch/default/1/epoch13-val_acc0.79.ckpt\",  # Seed: 441490\n    \"/kaggle/input/nexar_ym88wo7m/pytorch/default/1/epoch13-val_acc0.79.ckpt\",  # Seed: 879461\n]\n\npredictions_per_model = []\nfor path in model_paths:\n\n    # Load model\n    model = NexarClassifier.load_from_checkpoint(path)\n    model.to(device)\n    model.eval()\n    \n    # Get predictions for each frame\n    predictions = {}\n    indices = [0, 1, 2]\n    weights = [0.2, 0.3, 0.5]   \n    \n    for frame_idx in indices:\n        test_dataset = NexarDataset(\n            test_df, \n            frame_idx=frame_idx, \n            return_label=False, \n            transform=T.Compose([\n                T.Lambda(pad_to_square),\n                T.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n            ]),\n        )\n        test_dataloader = DataLoader(test_dataset, batch_size=64, shuffle=False, drop_last=False)\n        \n        preds = []\n        for batch in tqdm(test_dataloader):\n            batch = {k: v.to(model.device) if isinstance(v, torch.Tensor) else v for k, v in batch.items()}\n            with torch.no_grad():\n                pred = model(batch)\n            pred = torch.sigmoid(pred).squeeze().detach().tolist()\n            preds.extend(pred)\n        \n        predictions[frame_idx] = preds\n    \n    # Take weighted average of predictions\n    final_predictions = np.zeros(len(test_df))\n    for i, frame_idx in enumerate(indices):\n        final_predictions += np.array(predictions[frame_idx]) * weights[i]\n    predictions_per_model.append(final_predictions / sum(weights))\n\n# Save predictions\nsubmission_df = pd.DataFrame({\n    \"id\": test_df[\"id\"].apply(lambda x: str(x).zfill(5)),\n    \"target\": np.mean(predictions_per_model, axis=0),\n})\nsubmission_df.to_csv(\"submission.csv\", index=False)\nsubmission_df.head()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Remove redundant files\nos.remove(\"/kaggle/working/yolov8m-seg.pt\")\nshutil.rmtree(\"/kaggle/working/processed\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}