{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":13896398,"sourceType":"datasetVersion","datasetId":8853473}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## SETUP AND IMPORTS","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport cv2\nimport pydicom\nfrom pathlib import Path\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.optim import Adam\nimport torchvision.models as models\n\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Kaggle paths\nDATA_DIR = \"/kaggle/input/pneumonia-detection-challenge\"\nTRAIN_IMAGES = os.path.join(DATA_DIR, \"stage_2_train_images\")\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\n\n# Load data\nlabels_df = pd.read_csv(os.path.join(DATA_DIR, \"stage_2_train_labels.csv\"))\nclass_df = pd.read_csv(os.path.join(DATA_DIR, \"stage_2_detailed_class_info.csv\"))\ndf = pd.merge(labels_df, class_df, on='patientId', how='left')\n\nprint(f\"Total samples: {len(df)}\")\nprint(f\"Pneumonia cases: {(df['Target']==1).sum()}\")\nprint(f\"Normal cases: {(df['Target']==0).sum()}\")\n\n# ===== USE ONLY 30% OF DATA FOR FASTER TRAINING =====\ndf_sample = df.sample(frac=0.3, random_state=42)\nprint(f\"\\n✓ Using 30% of data: {len(df_sample)} samples\")\nprint(f\"  Pneumonia cases: {(df_sample['Target']==1).sum()}\")\nprint(f\"  Normal cases: {(df_sample['Target']==0).sum()}\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-27T14:16:04.090605Z","iopub.execute_input":"2025-11-27T14:16:04.091104Z","iopub.status.idle":"2025-11-27T14:16:15.131818Z","shell.execute_reply.started":"2025-11-27T14:16:04.091082Z","shell.execute_reply":"2025-11-27T14:16:15.131105Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## DATA VISUALIZATION","metadata":{}},{"cell_type":"code","source":"def visualize_samples_with_boxes(df, image_dir, num_samples=6):\n\n    positive_samples = df[df['Target'] == 1].sample(n=3, random_state=42)\n    negative_samples = df[df['Target'] == 0].sample(n=3, random_state=42)\n    samples = pd.concat([positive_samples, negative_samples])\n    \n    fig, axes = plt.subplots(2, 3, figsize=(15, 10))\n    axes = axes.ravel()\n    \n    for idx, (_, row) in enumerate(samples.iterrows()):\n        patient_id = row['patientId']\n        dicom_path = os.path.join(image_dir, f\"{patient_id}.dcm\")\n        \n        # Load DICOM image\n        dicom = pydicom.dcmread(dicom_path)\n        image = dicom.pixel_array\n        \n        # Normalize image to 0-255\n        image = (image - image.min()) / (image.max() - image.min() + 1e-7) * 255\n        image = image.astype(np.uint8)\n        \n        # Display image\n        axes[idx].imshow(image, cmap='gray')\n        axes[idx].set_title(f\"Patient: {patient_id}\\nPneumonia: {'Yes' if row['Target'] == 1 else 'No'}\")\n        axes[idx].axis('off')\n        \n        # Draw bounding box if pneumonia exists\n        if row['Target'] == 1 and not pd.isna(row['x']):\n            rect = patches.Rectangle(\n                (row['x'], row['y']),\n                row['width'],\n                row['height'],\n                linewidth=3,\n                edgecolor='red',\n                facecolor='none'\n            )\n            axes[idx].add_patch(rect)\n    \n    plt.tight_layout()\n    plt.savefig('pneumonia_samples.png', dpi=150, bbox_inches='tight')\n    plt.show()\n\n# Visualize samples\nprint(\"\\nVisualizing sample images...\")\nvisualize_samples_with_boxes(df_sample, TRAIN_IMAGES)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-21T13:13:01.279619Z","iopub.execute_input":"2025-11-21T13:13:01.280536Z","iopub.status.idle":"2025-11-21T13:13:04.376368Z","shell.execute_reply.started":"2025-11-21T13:13:01.280506Z","shell.execute_reply":"2025-11-21T13:13:04.375373Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## BUILD VGG16 BACKBONE FROM SCRATCH","metadata":{}},{"cell_type":"code","source":"# TODO: Build the VGG16 architecture\n\nclass VGGBackbone(nn.Module):\n    \"\"\"\n    VGG16 backbone for SSD - extracts features at multiple scales\n    \n    YOUR TASK: Build VGG16 from scratch and extract multi-scale features\n    \n    VGG16 Architecture:\n    - Block 1: [Conv 64, Conv 64, MaxPool]\n    - Block 2: [Conv 128, Conv 128, MaxPool]\n    - Block 3: [Conv 256, Conv 256, Conv 256, MaxPool]\n    - Block 4: [Conv 512, Conv 512, Conv 512, MaxPool] ← Extract here (conv4_3)\n    - Block 5: [Conv 512, Conv 512, Conv 512, MaxPool] ← Extract here (conv5_3)\n    - Extra layers for multi-scale detection\n    \n    Expected Feature Map Sizes (224x224 input):\n    - conv4_3: [batch, 512, 28, 28]\n    - conv5_3: [batch, 512, 14, 14]\n    - conv6:   [batch, 1024, 7, 7]\n    - conv7:   [batch, 512, 3, 3]\n    \"\"\"\n    \n    def __init__(self, pretrained=True):\n        super(VGGBackbone, self).__init__()\n        \n        # TODO 1: Build block1\n        # Block 1: Input(3) -> Conv(64) -> Conv(64) -> MaxPool\n        \n        self.block1 = nn.Sequential(\n            # YOUR CODE HERE: Block 1\n        )\n\n        # TODO 2: Build block2\n        # Block 2: Input(64) -> Conv(128) -> Conv(128) -> MaxPool\n        self.block2 = nn.Sequential(   \n            # YOUR CODE HERE: Block 2\n        )\n        \n        # TODO 3: Build block3 \n        # Block 3: Input(128) -> Conv(256) -> Conv(256) -> Conv(256) -> MaxPool\n        self.block3 = nn.Sequential(\n            # YOUR CODE HERE\n        )\n        \n        # TODO 4: Build block4 \n        # Block 4: Input(256) -> Conv(512) -> Conv(512) -> Conv(512) -> MaxPool\n        self.block4_conv = nn.Sequential(\n            # YOUR CODE HERE\n        )\n        self.block4_pool = # YOUR CODE HERE\n        \n        # TODO 5: Build block5 \n        # Block 5: Input(512) -> Conv(512) -> Conv(512) -> Conv(512) -> MaxPool\n        self.block5_conv = nn.Sequential(\n            # YOUR CODE HERE\n        )\n        self.block5_pool = # YOUR CODE HERE\n        \n        # TODO 6: Build extra convolutional layers for multi-scale detection\n        # Extra conv6: 512 -> 1024 channels\n        # Output: [batch, 1024, 7, 7]\n        self.conv6 = nn.Sequential(\n            # YOUR CODE HERE\n        )\n        \n        # Extra conv7: 1024 -> 512 channels\n        # Output: [batch, 512, 3, 3]\n        self.conv7 = nn.Sequential(\n            # YOUR CODE HERE\n        )\n        \n    \n    \n    def forward(self, x):\n        \"\"\"\n        Extract multi-scale features\n        \n        YOUR TASK: Complete the forward pass\n        \"\"\"\n        features = []\n        \n        # Forward through initial blocks\n        x =   # [batch, 64, 112, 112]\n        x =   # [batch, 128, 56, 56]\n        x =   # [batch, 256, 28, 28]\n        \n        # Block 4: Extract features at 28x28 (before pooling)\n        x =   # [batch, 512, 28, 28]\n        features.append(x)        # Feature map 0: conv4_3\n        x =   # [batch, 512, 14, 14]\n        \n        # Block 5: Extract features at 14x14 (before pooling)\n        x =   # [batch, 512, 14, 14]\n        features.append(x)        # Feature map 1: conv5_3\n        x =    # [batch, 512, 7, 7]\n        \n        # Extra layers\n        x =          # [batch, 1024, 7, 7] then pool to [batch, 1024, 3, 3]\n        features.append(x)        # Feature map 2: conv6\n        \n        x =         # [batch, 512, 3, 3] then pool to [batch, 512, 1, 1]\n        features.append(x)        # Feature map 3: conv7\n        \n        return features\n\n\n# Test your implementation\nprint(\"Building VGG16 from scratch with 224x224 input...\")\n\nbackbone = VGGBackbone(pretrained=False)\ntest_input = torch.randn(1, 3, 224, 224)\nfeature_maps = backbone(test_input)\n\nprint(f\"\\n Number of feature maps: {len(feature_maps)}\")\nfor i, fm in enumerate(feature_maps):\n    print(f\"  Feature map {i}: {fm.shape}\")\n\n# print(\"\\nExpected output:\")\n# print(\"  Feature map 0: torch.Size([1, 512, 28, 28])   - conv4_3\")\n# print(\"  Feature map 1: torch.Size([1, 512, 14, 14])   - conv5_3\")\n# print(\"  Feature map 2: torch.Size([1, 1024, 7, 7])    - extra conv6\")\n# print(\"  Feature map 3: torch.Size([1, 512, 3, 3])     - extra conv7\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  BUILD SSD DETECTION HEAD FROM SCRATCH","metadata":{}},{"cell_type":"code","source":"# TODO: Build the detection head\n\nclass SSDDetectionHead(nn.Module):\n    \"\"\"\n    SSD detection head that predicts classes and bounding boxes\n    \n    YOUR TASK: Build the detection head that predicts:\n    1. Classification scores for each anchor box\n    2. Bounding box coordinates for each anchor box\n    \n    For each feature map, we predict:\n    - num_anchors * num_classes classification scores\n    - num_anchors * 4 bounding box coordinates (x, y, w, h)\n    \"\"\"\n    \n    def __init__(self, num_classes=2):\n        super(SSDDetectionHead, self).__init__()\n        self.num_classes = num_classes\n        \n        # Number of anchor boxes per location for each feature map\n        # Feature maps have different scales, so different number of anchors\n        self.num_anchors = [4, 6, 6, 4]\n        \n        # Input channels from VGG backbone\n        # [conv4_3: 512, conv5_3: 512, conv6: 1024, conv7: 512]\n        in_channels = [512, 512, 1024, 512]\n        \n        # TODO 1: Create classification heads\n        # For each feature map, create a Conv2d layer that predicts class scores\n        # Output channels = num_anchors * num_classes\n        self.classification_heads = nn.ModuleList()\n        \n        for in_ch, num_anc in zip(in_channels, self.num_anchors):\n            # YOUR CODE HERE\n            # Create a Conv2d layer:\n            # - Input channels: in_ch\n            # - Output channels: num_anc * num_classes\n            # - Kernel size: 3\n            # - Padding: 1\n            self.classification_heads.append(\n                # YOUR CODE HERE (1 line)\n            )\n        \n        # TODO 2: Create localization heads\n        # For each feature map, create a Conv2d layer that predicts bounding boxes\n        # Output channels = num_anchors * 4 (x, y, w, h)\n        self.localization_heads = nn.ModuleList()\n        \n        for in_ch, num_anc in zip(in_channels, self.num_anchors):\n            # YOUR CODE HERE\n            # Create a Conv2d layer:\n            # - Input channels: in_ch\n            # - Output channels: num_anc * 4\n            # - Kernel size: 3\n            # - Padding: 1\n            self.localization_heads.append(\n                # YOUR CODE HERE (1 line)\n            )\n    \n    def forward(self, feature_maps):\n        \"\"\"\n        Predict classes and boxes for all feature maps\n        \n        YOUR TASK: Complete the forward pass to:\n        1. Apply classification and localization heads to each feature map\n        2. Reshape outputs to (batch, num_boxes, num_classes) and (batch, num_boxes, 4)\n        3. Concatenate predictions from all feature maps\n        \n        Input:\n            feature_maps: List of 4 tensors with shapes:\n                - [batch, 512, 28, 28]\n                - [batch, 512, 14, 14]\n                - [batch, 1024, 7, 7]\n                - [batch, 512, 3, 3]\n        \n        Output:\n            classifications: [batch, total_anchors, num_classes]\n            localizations: [batch, total_anchors, 4]\n            \n            Where total_anchors = 28*28*4 + 14*14*6 + 7*7*6 + 3*3*4 = 4,642\n        \"\"\"\n        batch_size = feature_maps[0].size(0)\n        classifications = []\n        localizations = []\n        \n        # TODO 3: Apply prediction heads to each feature map\n        for i, fmap in enumerate(feature_maps):\n            \n            # Apply classification head\n            # Input: [batch, in_channels, H, W]\n            # Output: [batch, num_anchors*num_classes, H, W]\n            cls_output = self.classification_heads[i](fmap)\n            \n            # Reshape to [batch, H, W, num_anchors*num_classes]\n            cls_output = cls_output.permute(0, 2, 3, 1).contiguous()\n            \n            # Reshape to [batch, H*W*num_anchors, num_classes]\n            cls_output = cls_output.view(batch_size, -1, self.num_classes)\n            classifications.append(cls_output)\n            \n            # TODO: Apply localization head and reshape\n            # YOUR CODE HERE (similar to classification above)\n            # Step 1: Apply self.localization_heads[i] to fmap\n            loc_output = # YOUR CODE HERE\n            \n            # Step 2: Permute to [batch, H, W, num_anchors*4]\n            loc_output = # YOUR CODE HERE\n            \n            # Step 3: View as [batch, H*W*num_anchors, 4]\n            loc_output = # YOUR CODE HERE\n            \n            localizations.append(loc_output)\n        \n        # TODO 4: Concatenate all predictions along the anchor dimension\n        # classifications: List of [batch, num_anchors_i, num_classes]\n        # -> [batch, total_anchors, num_classes]\n        classifications = # YOUR CODE HERE: torch.cat(...)\n        \n        # localizations: List of [batch, num_anchors_i, 4]\n        # -> [batch, total_anchors, 4]\n        localizations = # YOUR CODE HERE: torch.cat(...)\n        \n        return classifications, localizations\n\n\n# Test your implementation\nprint(\"TESTING SSD DETECTION HEAD\")\n\nhead = SSDDetectionHead(num_classes=2)\nclassifications, localizations = head(feature_maps)\n\nprint(f\"\\n Classifications shape: {classifications.shape}\")\nprint(f\" Localizations shape: {localizations.shape}\")\nprint(f\" Total anchor boxes: {classifications.shape[1]}\")\n\n# print(\"\\nExpected output:\")\n# print(\"  Classifications: torch.Size([1, 4642, 2])\")\n# print(\"  Localizations: torch.Size([1, 4642, 4])\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## IMPLEMENT LOSS FUNCTIONS","metadata":{}},{"cell_type":"code","source":"# TODO: Implement TWO loss functions\n\nclass FocalLoss(nn.Module):\n    \"\"\"\n    Focal Loss for classification (handles class imbalance)\n    \n    Formula: FL = -α(1-pt)^γ * log(pt)\n    \n    Where:\n    - pt = probability of correct class\n    - α = balancing factor (default 0.25)\n    - γ = focusing parameter (default 2.0)\n    \n    YOUR TASK: Implement the forward pass\n    \"\"\"\n    \n    def __init__(self, alpha=0.25, gamma=2.0):\n        super().__init__()\n        self.alpha = alpha\n        self.gamma = gamma\n    \n    def forward(self, predictions, targets):\n        \"\"\"\n        Compute focal loss\n        \n        Args:\n            predictions: (batch, num_boxes, num_classes) - raw logits\n            targets: (batch, num_boxes) - class labels (0 or 1)\n        \n        Returns:\n            loss: scalar tensor\n        \"\"\"\n        \n        # TODO 1: Apply sigmoid to get probabilities\n        probs = # YOUR CODE HERE\n        \n        # TODO 2: Compute binary cross entropy loss\n        # Use F.binary_cross_entropy_with_logits\n        # For class 1 (pneumonia): predictions[:, :, 1]\n        # Target should be float: targets.float()\n        bce_loss = # YOUR CODE HERE\n        \n        p_t = probs[:, :, 1] * targets + (1 - probs[:, :, 1]) * (1 - targets)\n        \n        # TODO 3: Compute focal weight: (1 - pt)^gamma\n        focal_weight = # YOUR CODE HERE\n        \n        focal_loss = self.alpha * focal_weight * bce_loss\n        \n        return focal_loss.mean()\n\n\nclass SmoothL1Loss(nn.Module):\n    \"\"\"\n    Smooth L1 Loss for bounding box regression\n    \n    Formula:\n        loss = 0.5 * x^2           if |x| < 1\n        loss = |x| - 0.5           otherwise\n    \n    Where x = prediction - target\n    \n    YOUR TASK: Implement the forward pass\n    \"\"\"\n    \n    def __init__(self):\n        super().__init__()\n    \n    def forward(self, predictions, targets, mask):\n        \"\"\"\n        Compute smooth L1 loss\n        \n        Args:\n            predictions: (batch, num_boxes, 4) - predicted boxes\n            targets: (batch, num_boxes, 4) - target boxes\n            mask: (batch, num_boxes) - which boxes are positive (has object)\n        \n        Returns:\n            loss: scalar tensor\n        \"\"\"\n        \n        # Only compute loss for positive boxes (where mask == True)\n        if mask.sum() == 0:\n            return torch.tensor(0.0, device=predictions.device)\n        \n        # TODO 1: Get positive predictions and targets using mask\n        pos_predictions = # YOUR CODE HERE\n        pos_targets = # YOUR CODE HERE\n        \n        # TODO 2: Compute absolute difference\n        diff = # YOUR CODE HERE\n        \n        # TODO 3: Apply smooth L1 formula using torch.where\n        # If |x| < 1: use 0.5 * diff^2\n        # If |x| >= 1: use diff - 0.5\n        loss = torch.where(\n            diff < 1.0,\n            # YOUR CODE HERE: quadratic part\n            # YOUR CODE HERE: linear part\n        )\n        \n        return loss.mean()\n\n\nclass CombinedLoss(nn.Module):\n    \"\"\"\n    Combined loss: Classification (Focal Loss) + Localization (Smooth L1 Loss)\n    \n    Total Loss = Classification Loss + α * Localization Loss\n    \"\"\"\n    \n    def __init__(self, alpha=1.0):\n        super().__init__()\n        self.focal_loss = FocalLoss()\n        self.smooth_l1_loss = SmoothL1Loss()\n        self.alpha = alpha\n    \n    def forward(self, predictions, targets):\n        \"\"\"\n        Compute combined loss\n        \n        Args:\n            predictions: tuple of (class_preds, box_preds)\n            targets: tuple of (class_targets, box_targets)\n        \n        Returns:\n            total_loss, cls_loss, loc_loss\n        \"\"\"\n        class_preds, box_preds = predictions\n        class_targets, box_targets = targets\n        \n        # Classification loss (all boxes)\n        cls_loss = self.focal_loss(class_preds, class_targets)\n        \n        # Localization loss (only positive boxes)\n        pos_mask = class_targets > 0\n        loc_loss = self.smooth_l1_loss(box_preds, box_targets, pos_mask)\n        \n        # Combined loss\n        total_loss = cls_loss + self.alpha * loc_loss\n        \n        return total_loss, cls_loss, loc_loss\n\n\n# Test your implementations\ncriterion = CombinedLoss(alpha=1.0)\n\n# Create dummy data\ndummy_cls_pred = torch.randn(2, 100, 2)\ndummy_box_pred = torch.randn(2, 100, 4)\ndummy_cls_target = torch.randint(0, 2, (2, 100))\ndummy_box_target = torch.randn(2, 100, 4)\n\ntotal_loss, cls_loss, loc_loss = criterion(\n    (dummy_cls_pred, dummy_box_pred),\n    (dummy_cls_target, dummy_box_target)\n)\n\nprint(f\"\\n Total loss: {total_loss.item():.4f}\")\nprint(f\" Classification loss: {cls_loss.item():.4f}\")\nprint(f\" Localization loss: {loc_loss.item():.4f}\")\nprint(\"\\n You should see numbers to show that your loss functions work!\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## DATASET","metadata":{}},{"cell_type":"code","source":"class SimpleDataset(Dataset):\n    \"\"\"Simple dataset for pneumonia detection\"\"\"\n    \n    def __init__(self, df, image_dir, img_size=224):\n        self.df = df.reset_index(drop=True)\n        self.image_dir = Path(image_dir)\n        self.img_size = img_size\n    \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        patient_id = row['patientId']\n        \n        # Load DICOM image\n        dicom_path = self.image_dir / f\"{patient_id}.dcm\"\n        dicom = pydicom.dcmread(dicom_path)\n        image = dicom.pixel_array\n        \n        # Normalize to [0, 1]\n        image = (image - image.min()) / (image.max() - image.min() + 1e-7)\n        image = (image * 255).astype(np.uint8)\n        \n        # Resize to 224x224\n        image = cv2.resize(image, (self.img_size, self.img_size))\n        image = cv2.cvtColor(image, cv2.COLOR_GRAY2RGB)\n        \n        # Convert to float32\n        image = image.astype(np.float32) / 255.0\n        \n        # Apply ImageNet normalization\n        mean = np.array([0.485, 0.456, 0.406], dtype=np.float32)\n        std = np.array([0.229, 0.224, 0.225], dtype=np.float32)\n        image = (image - mean) / std\n        \n        # Convert to tensor\n        image = torch.from_numpy(image).permute(2, 0, 1).float()\n        \n        # Labels\n        label = torch.tensor([row['Target']], dtype=torch.long)\n        \n        # Bounding boxes (normalized to [0, 1])\n        boxes = torch.zeros(1, 4, dtype=torch.float32)\n        if row['Target'] == 1 and not pd.isna(row['x']):\n            boxes[0] = torch.tensor([\n                row['x'] / 1024, row['y'] / 1024,\n                row['width'] / 1024, row['height'] / 1024\n            ], dtype=torch.float32)\n        \n        return image, (label, boxes)\n\n\n# Prepare data splits (80% train, 20% validation)\ntrain_df, val_df = train_test_split(\n    df_sample, \n    test_size=0.2, \n    random_state=42, \n    stratify=df_sample['Target']\n)\n\ntrain_dataset = SimpleDataset(train_df, TRAIN_IMAGES, img_size=224)\nval_dataset = SimpleDataset(val_df, TRAIN_IMAGES, img_size=224)\n\ntrain_loader = DataLoader(train_dataset, batch_size=16, shuffle=True, num_workers=2)\nval_loader = DataLoader(val_dataset, batch_size=16, shuffle=False, num_workers=2)\n\nprint(\"DATASET PREPARATION\")\nprint(f\" Training samples: {len(train_dataset)}\")\nprint(f\" Validation samples: {len(val_dataset)}\")\nprint(f\" Image size: 224x224\")\nprint(f\" Batch size: 16\")\nprint(f\" Total batches per epoch: {len(train_loader)} (train), {len(val_loader)} (val)\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## COMPLETE SSD MODEL","metadata":{}},{"cell_type":"code","source":"class SSD_VGG(nn.Module):\n    \"\"\"Complete SSD model with VGG backbone\"\"\"\n    \n    def __init__(self, num_classes=2):\n        super(SSD_VGG, self).__init__()\n        self.num_classes = num_classes\n        self.backbone = VGGBackbone(pretrained=True)\n        self.detection_head = SSDDetectionHead(num_classes=num_classes)\n    \n    def forward(self, x):\n        feature_maps = self.backbone(x)\n        classifications, localizations = self.detection_head(feature_maps)\n        return classifications, localizations\n\n\n# Initialize model\nprint(\"MODEL INITIALIZATION\")\n\nmodel = SSD_VGG(num_classes=2).to(device)\ncriterion = CombinedLoss(alpha=1.0)\noptimizer = torch.optim.AdamW(model.parameters(), lr=1e-4, weight_decay=1e-4)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## TRAINING","metadata":{}},{"cell_type":"code","source":"# TODO: Complete the Train and Validation loop\n\ndef train_one_epoch(model, dataloader, criterion, optimizer, device):\n    \"\"\"\n    Train for one epoch\n    \n    \"\"\"\n    model.train()\n    total_loss = 0\n    total_cls_loss = 0\n    total_loc_loss = 0\n    \n    pbar = tqdm(dataloader, desc=\"Training\")\n    for images, (labels, boxes) in pbar:\n        images = images.to(device)\n        labels = labels.to(device)\n        boxes = boxes.to(device)\n        \n        # Squeeze extra dimensions\n        labels = labels.squeeze(1)  # [batch, 1] -> [batch]\n        boxes = boxes.squeeze(1)    # [batch, 1, 4] -> [batch, 4]\n        \n        # TODO 1: Forward pass\n        # Get predictions from model\n        class_preds, box_preds = # YOUR CODE HERE\n        \n        # Expand labels and boxes to match number of predicted boxes\n        num_boxes = class_preds.size(1)  # 4,642 anchors\n        labels_expanded = labels.unsqueeze(1).expand(-1, num_boxes)\n        boxes_expanded = boxes.unsqueeze(1).expand(-1, num_boxes, -1)\n        \n        # Compute loss\n        loss, cls_loss, loc_loss = criterion(\n            (class_preds, box_preds),\n            (labels_expanded, boxes_expanded)\n        )\n        \n        # TODO 2: Backward pass\n        # YOUR CODE HERE: Zero gradients\n        # YOUR CODE HERE: Compute gradients \n        # YOUR CODE HERE: Update weights \n        \n        total_loss += loss.item()\n        total_cls_loss += cls_loss.item()\n        total_loc_loss += loc_loss.item()\n        \n        # Update progress bar\n        pbar.set_postfix({\n            'loss': f'{loss.item():.4f}',\n            'cls': f'{cls_loss.item():.4f}',\n            'loc': f'{loc_loss.item():.4f}'\n        })\n    \n    return {\n        'total_loss': total_loss / len(dataloader),\n        'cls_loss': total_cls_loss / len(dataloader),\n        'loc_loss': total_loc_loss / len(dataloader)\n    }\n\n\ndef validate_one_epoch(model, dataloader, criterion, device):\n    \"\"\"\n    Validate for one epoch\n    \n    \"\"\"\n    model.eval()\n    total_loss = 0\n    total_cls_loss = 0\n    total_loc_loss = 0\n    \n    with torch.no_grad():\n        for images, (labels, boxes) in tqdm(dataloader, desc=\"Validation\"):\n            images = images.to(device)\n            labels = labels.to(device)\n            boxes = boxes.to(device)\n            \n            # Squeeze extra dimensions\n            labels = labels.squeeze(1)\n            boxes = boxes.squeeze(1)\n            \n            # TODO 3: Forward pass (no gradients needed)\n            # Get predictions from model\n            class_preds, box_preds = # YOUR CODE HERE\n            \n            # Expand labels and boxes\n            num_boxes = class_preds.size(1)\n            labels_expanded = labels.unsqueeze(1).expand(-1, num_boxes)\n            boxes_expanded = boxes.unsqueeze(1).expand(-1, num_boxes, -1)\n            \n            # Compute loss\n            loss, cls_loss, loc_loss = criterion(\n                (class_preds, box_preds),\n                (labels_expanded, boxes_expanded)\n            )\n            \n            total_loss += loss.item()\n            total_cls_loss += cls_loss.item()\n            total_loc_loss += loc_loss.item()\n    \n    return {\n        'total_loss': total_loss / len(dataloader),\n        'cls_loss': total_cls_loss / len(dataloader),\n        'loc_loss': total_loc_loss / len(dataloader)\n    }\n\n\n# Learning rate scheduler\nscheduler = torch.optim.lr_scheduler.CosineAnnealingLR(optimizer, T_max=50, eta_min=1e-6)\n\n# Training history\nhistory = {\n    'train_loss': [],\n    'train_cls_loss': [],\n    'train_loc_loss': [],\n    'val_loss': [],\n    'val_cls_loss': [],\n    'val_loc_loss': [],\n    'lr': []\n}\n\n# Training loop\nEPOCHS = 50\nbest_val_loss = float('inf')\n\nprint(\"STARTING TRAINING FOR 50 EPOCHS \\n\")\n\n\nfor epoch in range(EPOCHS):\n    print(f\"\\nEpoch {epoch+1}/{EPOCHS}\")\n    print(\"-\" * 60)\n    \n    # Train\n    train_metrics = train_one_epoch(model, train_loader, criterion, optimizer, device)\n    \n    # Validate\n    val_metrics = validate_one_epoch(model, val_loader, criterion, device)\n    \n    # Update learning rate\n    scheduler.step()\n    current_lr = optimizer.param_groups[0]['lr']\n    \n    # Save history\n    history['train_loss'].append(train_metrics['total_loss'])\n    history['train_cls_loss'].append(train_metrics['cls_loss'])\n    history['train_loc_loss'].append(train_metrics['loc_loss'])\n    history['val_loss'].append(val_metrics['total_loss'])\n    history['val_cls_loss'].append(val_metrics['cls_loss'])\n    history['val_loc_loss'].append(val_metrics['loc_loss'])\n    history['lr'].append(current_lr)\n    \n    # Print metrics\n    print(f\"\\nTrain Loss: {train_metrics['total_loss']:.4f} \"\n          f\"(Cls: {train_metrics['cls_loss']:.4f}, Loc: {train_metrics['loc_loss']:.4f})\")\n    print(f\"Val Loss:   {val_metrics['total_loss']:.4f} \"\n          f\"(Cls: {val_metrics['cls_loss']:.4f}, Loc: {val_metrics['loc_loss']:.4f})\")\n    print(f\"Learning Rate: {current_lr:.6f}\")\n    \n    # Save best model\n    if val_metrics['total_loss'] < best_val_loss:\n        best_val_loss = val_metrics['total_loss']\n        torch.save({\n            'epoch': epoch,\n            'model_state_dict': model.state_dict(),\n            'optimizer_state_dict': optimizer.state_dict(),\n            'val_loss': best_val_loss,\n            'history': history\n        }, 'best_ssd_vgg_pneumonia.pth')\n        print(f\"✓ Best model saved! (Val Loss: {best_val_loss:.4f})\")\n    \n    # Save checkpoint every 10 epochs\n    if (epoch + 1) % 10 == 0:\n        torch.save({\n            'epoch': epoch,\n            'model_state_dict': model.state_dict(),\n            'optimizer_state_dict': optimizer.state_dict(),\n            'history': history\n        }, f'checkpoint_epoch_{epoch+1}.pth')\n        print(f\"✓ Checkpoint saved at epoch {epoch+1}\")\n\nprint(\"\\n\" + \"=\"*60)\nprint(\"TRAINING COMPLETED!\")\nprint(\"=\"*60)\nprint(f\"Best Validation Loss: {best_val_loss:.4f}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## PLOT TRAINING CURVES ","metadata":{}},{"cell_type":"code","source":"# TODO: Ploting loss curves\n\ndef plot_loss_curves(history):\n    \"\"\"\n    Plot training and validation loss curves\n    \n    YOUR TASK: Complete this function to create 4 subplots:\n    1. Total Loss\n    2. Classification Loss\n    3. Localization Loss\n    4. Learning Rate\n    \"\"\"\n    \n    fig, axes = plt.subplots(2, 2, figsize=(15, 10))\n    \n    # TODO 1: Plot Total Loss\n    axes[0, 0].plot(history['train_loss'], label='Train Loss', linewidth=2, color='blue')\n    axes[0, 0].plot(history['val_loss'], label='Val Loss', linewidth=2, color='orange')\n    axes[0, 0].set_xlabel('Epoch', fontsize=12)\n    axes[0, 0].set_ylabel('Loss', fontsize=12)\n    axes[0, 0].set_title('Total Loss', fontsize=14, fontweight='bold')\n    axes[0, 0].legend(fontsize=10)\n    axes[0, 0].grid(True, alpha=0.3)\n    \n    # TODO 2: Plot Classification Loss\n    axes[0, 1].# YOUR CODE HERE: plot train_cls_loss\n    axes[0, 1].# YOUR CODE HERE: plot val_cls_loss\n    axes[0, 1].# YOUR CODE HERE: set xlabel\n    axes[0, 1].# YOUR CODE HERE: set ylabel\n    axes[0, 1].# YOUR CODE HERE: set title 'Classification Loss'\n    axes[0, 1].# YOUR CODE HERE: legend and grid\n    \n    # TODO 3: Plot Localization Loss\n    axes[1, 0].# YOUR CODE HERE: plot train_loc_loss\n    axes[1, 0].# YOUR CODE HERE: plot val_loc_loss\n    axes[1, 0].# YOUR CODE HERE: set xlabel\n    axes[1, 0].# YOUR CODE HERE: set ylabel\n    axes[1, 0].# YOUR CODE HERE: set title 'Localization Loss'\n    axes[1, 0].# YOUR CODE HERE: legend and grid\n    \n    # TODO 4: Plot Learning Rate\n    axes[1, 1].plot(history['lr'], linewidth=2, color='green')\n    axes[1, 1].set_xlabel('Epoch', fontsize=12)\n    axes[1, 1].set_ylabel('Learning Rate', fontsize=12)\n    axes[1, 1].set_title('Learning Rate Schedule', fontsize=14, fontweight='bold')\n    axes[1, 1].grid(True, alpha=0.3)\n    axes[1, 1].set_yscale('log')\n    \n    plt.tight_layout()\n    plt.savefig('training_curves.png', dpi=150, bbox_inches='tight')\n    plt.show()\n    \n    # TODO 5: Print final metrics\n    print(\"TRAINING SUMMARY\")\n    print(f\"Total Epochs: {len(history['train_loss'])}\")\n    print(f\"Final Train Loss: {history['train_loss'][-1]:.4f}\")\n    print(f\"Final Val Loss:   {history['val_loss'][-1]:.4f}\")\n    print(f\"Best Val Loss:    {min(history['val_loss']):.4f} (Epoch {history['val_loss'].index(min(history['val_loss']))+1})\")\n    print(f\"\\nFinal Classification Loss:\")\n    print(f\"  Train: {history['train_cls_loss'][-1]:.4f}\")\n    print(f\"  Val:   {history['val_cls_loss'][-1]:.4f}\")\n    print(f\"\\nFinal Localization Loss:\")\n    print(f\"  Train: {history['train_loc_loss'][-1]:.4f}\")\n    print(f\"  Val:   {history['val_loc_loss'][-1]:.4f}\")\n    print(\"=\"*60)\n\n# Plot the curves\nplot_loss_curves(history)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## MODEL TESTING","metadata":{}},{"cell_type":"code","source":"# TODO: Complete the testing functions\n\ndef load_best_model(model, checkpoint_path='best_ssd_vgg_pneumonia.pth'):\n    \"\"\"\n    Load the best saved model\n    \n    YOUR TASK: Load the checkpoint and restore model weights\n    \"\"\"\n    # TODO 1: Load checkpoint\n    checkpoint = # YOUR CODE HERE:\n    \n    # TODO 2: Load model state\n    # YOUR CODE HERE\n    \n    print(f\" Loaded best model from epoch {checkpoint['epoch']+1}\")\n    print(f\" Best validation loss: {checkpoint['val_loss']:.4f}\")\n    \n    return model\n\n\ndef predict_single_image(model, image_path, device, confidence_threshold=0.5):\n    \"\"\"\n    Make prediction on a single image\n    \n    YOUR TASK: Complete the prediction function\n    \"\"\"\n    model.eval()\n    \n    # Load and preprocess image\n    dicom = pydicom.dcmread(image_path)\n    image = dicom.pixel_array\n    \n    # Normalize\n    image = (image - image.min()) / (image.max() - image.min() + 1e-7)\n    image = (image * 255).astype(np.uint8)\n    original_image = image.copy()\n    \n    # Convert to RGB and resize to 224x224\n    image = cv2.cvtColor(image, cv2.COLOR_GRAY2RGB)\n    image = cv2.resize(image, (224, 224))\n    \n    # Apply normalization\n    image = image.astype(np.float32) / 255.0\n    mean = np.array([0.485, 0.456, 0.406], dtype=np.float32)\n    std = np.array([0.229, 0.224, 0.225], dtype=np.float32)\n    image = (image - mean) / std\n    \n    # Convert to tensor\n    image_tensor = torch.from_numpy(image).permute(2, 0, 1).unsqueeze(0).float()\n    image_tensor = image_tensor.to(device)\n    \n    # TODO 2: Make prediction\n    with torch.no_grad():\n        # YOUR CODE HERE: Get predictions from model\n        class_preds, box_preds = \n    \n    # Process predictions\n    probs = F.softmax(class_preds, dim=2)\n    pneumonia_probs = probs[0, :, 1]\n    \n    # Filter by confidence\n    confident_boxes = pneumonia_probs > confidence_threshold\n    \n    if confident_boxes.sum() > 0:\n        confident_box_coords = box_preds[0][confident_boxes]\n        confident_scores = pneumonia_probs[confident_boxes]\n        \n        # Get best box\n        best_idx = confident_scores.argmax()\n        best_box = confident_box_coords[best_idx]\n        best_score = confident_scores[best_idx]\n        \n        # Convert to pixel coordinates (original image size)\n        h, w = original_image.shape\n        x = int(best_box[0].item() * w)\n        y = int(best_box[1].item() * h)\n        box_w = int(best_box[2].item() * w)\n        box_h = int(best_box[3].item() * h)\n        \n        return {\n            'has_pneumonia': True,\n            'confidence': best_score.item(),\n            'box': (x, y, box_w, box_h),\n            'image': original_image\n        }\n    else:\n        return {\n            'has_pneumonia': False,\n            'confidence': pneumonia_probs.max().item(),\n            'box': None,\n            'image': original_image\n        }\n\n\ndef visualize_predictions(model, df, image_dir, num_samples=6, confidence_threshold=0.5):\n    \"\"\"\n    Visualize model predictions on test samples\n    \n    YOUR TASK: Complete the visualization\n    \"\"\"\n    \n    # Sample images\n    samples = df.sample(n=num_samples, random_state=42)\n    \n    fig, axes = plt.subplots(2, 3, figsize=(15, 10))\n    axes = axes.ravel()\n    \n    for idx, (_, row) in enumerate(samples.iterrows()):\n        patient_id = row['patientId']\n        image_path = os.path.join(image_dir, f\"{patient_id}.dcm\")\n        \n        # Make prediction\n        result = predict_single_image(model, image_path, device, confidence_threshold)\n        \n        # Display image\n        axes[idx].imshow(result['image'], cmap='gray')\n        \n        # TODO 3: Create title with prediction info\n        # YOUR CODE HERE: Create title showing:\n        # - Patient ID\n        # - Predicted class (Pneumonia/Normal)\n        # - Confidence score\n        # - Actual class\n        title = # YOUR CODE HERE\n        axes[idx].set_title(title, fontsize=10)\n        axes[idx].axis('off')\n        \n        # TODO 4: Draw predicted bounding box (RED)\n        if result['has_pneumonia'] and result['box'] is not None:\n            x, y, w, h = result['box']\n            #  Create Rectangle patch\n            rect = # YOUR CODE HERE\n            axes[idx].add_patch(rect)\n        \n        # TODO 5: Draw ground truth bounding box (GREEN, dashed)\n        if row['Target'] == 1 and not pd.isna(row['x']):\n            #  Create Rectangle patch with green dashed line\n            rect_gt = # YOUR CODE HERE\n            axes[idx].add_patch(rect_gt)\n            axes[idx].legend(loc='upper right', fontsize=8)\n    \n    plt.tight_layout()\n    plt.savefig('model_predictions.png', dpi=150, bbox_inches='tight')\n    plt.show()\n\n\ndef calculate_metrics(model, dataloader, device, confidence_threshold=0.5):\n    \"\"\"\n    Calculate classification metrics\n    \n    YOUR TASK: Complete the metrics calculation\n    \"\"\"\n    model.eval()\n    \n    all_predictions = []\n    all_labels = []\n    all_probs = []\n    \n    with torch.no_grad():\n        for images, (labels, boxes) in tqdm(dataloader, desc=\"Calculating metrics\"):\n            images = images.to(device)\n            \n            # TODO 6: Get predictions\n            class_preds, box_preds = # YOUR CODE HERE\n            \n            # Get probabilities\n            probs = F.softmax(class_preds, dim=2)\n            pneumonia_probs = probs[:, :, 1].max(dim=1)[0]\n            \n            # TODO 7: Convert probabilities to binary predictions\n            #  If prob > threshold, predict 1, else 0\n            binary_preds = # YOUR CODE HERE\n            \n            all_predictions.extend(binary_preds)\n            all_labels.extend(labels.cpu().numpy().flatten())\n            all_probs.extend(pneumonia_probs.cpu().numpy())\n    \n    # Calculate metrics\n    from sklearn.metrics import (accuracy_score, precision_score, recall_score, \n                                  f1_score, confusion_matrix, roc_auc_score, \n                                  classification_report)\n    \n    accuracy = accuracy_score(all_labels, all_predictions)\n    precision = precision_score(all_labels, all_predictions, zero_division=0)\n    recall = recall_score(all_labels, all_predictions, zero_division=0)\n    f1 = f1_score(all_labels, all_predictions, zero_division=0)\n    cm = confusion_matrix(all_labels, all_predictions)\n    \n    try:\n        auc = roc_auc_score(all_labels, all_probs)\n    except:\n        auc = 0.0\n    \n    # Print metrics\n    print(\"\\n\" + \"=\"*60)\n    print(\"MODEL EVALUATION METRICS\")\n    print(\"=\"*60)\n    print(f\"Accuracy:  {accuracy:.4f}\")\n    print(f\"Precision: {precision:.4f}\")\n    print(f\"Recall:    {recall:.4f}\")\n    print(f\"F1 Score:  {f1:.4f}\")\n    print(f\"AUC Score: {auc:.4f}\")\n    print(\"\\nConfusion Matrix:\")\n    print(\"                Predicted\")\n    print(\"              Normal  Pneumonia\")\n    print(f\"Actual Normal    {cm[0,0]:6d}  {cm[0,1]:6d}\")\n    print(f\"     Pneumonia   {cm[1,0]:6d}  {cm[1,1]:6d}\")\n    print(\"\\nClassification Report:\")\n    print(classification_report(all_labels, all_predictions, \n                                target_names=['Normal', 'Pneumonia'],\n                                zero_division=0))\n    print(\"=\"*60)\n    \n    # Plot confusion matrix\n    plt.figure(figsize=(8, 6))\n    plt.imshow(cm, interpolation='nearest', cmap=plt.cm.Blues)\n    plt.title('Confusion Matrix', fontsize=14, fontweight='bold')\n    plt.colorbar()\n    tick_marks = np.arange(2)\n    plt.xticks(tick_marks, ['Normal', 'Pneumonia'], fontsize=12)\n    plt.yticks(tick_marks, ['Normal', 'Pneumonia'], fontsize=12)\n    \n    # Add text annotations\n    thresh = cm.max() / 2.\n    for i in range(2):\n        for j in range(2):\n            plt.text(j, i, format(cm[i, j], 'd'),\n                    ha=\"center\", va=\"center\",\n                    fontsize=20, fontweight='bold',\n                    color=\"white\" if cm[i, j] > thresh else \"black\")\n    \n    plt.ylabel('True Label', fontsize=12)\n    plt.xlabel('Predicted Label', fontsize=12)\n    plt.tight_layout()\n    plt.savefig('confusion_matrix.png', dpi=150, bbox_inches='tight')\n    plt.show()\n    \n    return {\n        'accuracy': accuracy,\n        'precision': precision,\n        'recall': recall,\n        'f1': f1,\n        'auc': auc,\n        'confusion_matrix': cm\n    }\n\n\n\n\nprint(\"TESTING THE MODEL\\n\")\n\n\n# Load best model\nmodel = load_best_model(model)\nmodel = model.to(device)\n\n# Visualize predictions on validation set\nprint(\"\\nVisualizing predictions...\")\nvisualize_predictions(model, val_df, TRAIN_IMAGES, num_samples=6, confidence_threshold=0.5)\n\n# Calculate metrics\nprint(\"\\nCalculating metrics on validation set...\")\nmetrics = calculate_metrics(model, val_loader, device, confidence_threshold=0.5)\n\n# Test on specific images\nprint(\"\\n\" + \"=\"*60)\nprint(\"TESTING ON INDIVIDUAL SAMPLES\")\nprint(\"=\"*60)\n\ntest_patient_ids = val_df.sample(n=5, random_state=42)['patientId'].values\n\nfor i, patient_id in enumerate(test_patient_ids):\n    image_path = os.path.join(TRAIN_IMAGES, f\"{patient_id}.dcm\")\n    result = predict_single_image(model, image_path, device, confidence_threshold=0.5)\n    \n    # Get actual label\n    actual = val_df[val_df['patientId'] == patient_id]['Target'].values[0]\n    \n    print(f\"\\n{i+1}. Patient: {patient_id}\")\n    print(f\"   Actual:      {'Pneumonia' if actual == 1 else 'Normal'}\")\n    print(f\"   Predicted:   {'Pneumonia' if result['has_pneumonia'] else 'Normal'}\")\n    print(f\"   Confidence:  {result['confidence']:.4f}\")\n    print(f\"   Correct:     {'✓' if (result['has_pneumonia'] == (actual == 1)) else '✗'}\")\n    if result['box']:\n        print(f\"   Bounding Box: x={result['box'][0]}, y={result['box'][1]}, \"\n              f\"w={result['box'][2]}, h={result['box'][3]}\")\n\n\nprint(\"TESTING COMPLETED!\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}