{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-06-28T19:17:48.649026Z","iopub.execute_input":"2026-06-28T19:17:48.649754Z","iopub.status.idle":"2026-06-28T19:18:00.928782Z","shell.execute_reply.started":"2026-06-28T19:17:48.649719Z","shell.execute_reply":"2026-06-28T19:18:00.927964Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 1\n!pip install -q pydicom opencv-contrib-python-headless\n\nimport os, random\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport pydicom\nimport matplotlib.pyplot as plt\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torchvision\nfrom torchvision.models.detection import fasterrcnn_resnet50_fpn\nfrom torchvision.models.detection.faster_rcnn import FastRCNNPredictor\nfrom torchvision.models.detection._utils import BoxCoder\nfrom torchvision.ops import RoIPool, box_iou, nms\nfrom torch.utils.data import Dataset, DataLoader\n\nSEED = 42\nrandom.seed(SEED); np.random.seed(SEED); torch.manual_seed(SEED)\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-28T19:18:35.795004Z","iopub.execute_input":"2026-06-28T19:18:35.795836Z","iopub.status.idle":"2026-06-28T19:18:39.011039Z","shell.execute_reply.started":"2026-06-28T19:18:35.795799Z","shell.execute_reply":"2026-06-28T19:18:39.010231Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 3\nDATA_DIR = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'  # adjust based on Cell 2 output\nTRAIN_IMG_DIR = os.path.join(DATA_DIR, 'stage_2_train_images')\nLABELS_CSV = os.path.join(DATA_DIR, 'stage_2_train_labels.csv')\nCLASS_CSV = os.path.join(DATA_DIR, 'stage_2_detailed_class_info.csv')\n\nlabels_df = pd.read_csv(LABELS_CSV)\nclass_df = pd.read_csv(CLASS_CSV)\n\nprint(labels_df.shape, class_df.shape)\ndisplay(labels_df.head())\nprint('Unique patients:', labels_df.patientId.nunique())\nprint(labels_df.Target.value_counts())\nprint(class_df['class'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-28T19:18:42.996939Z","iopub.execute_input":"2026-06-28T19:18:42.997713Z","iopub.status.idle":"2026-06-28T19:18:43.103321Z","shell.execute_reply.started":"2026-06-28T19:18:42.997677Z","shell.execute_reply":"2026-06-28T19:18:43.102443Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 4 -- patient-level train/val split (stratified by whether they have pneumonia)\nfrom sklearn.model_selection import train_test_split\n\npatient_ids = labels_df.patientId.unique()\n\n# --- subsample to 10% of patients ---\nSUBSET_FRACTION = 0.1\nrng = np.random.RandomState(SEED)\nsubset_size = int(len(patient_ids) * SUBSET_FRACTION)\npatient_ids = rng.choice(patient_ids, size=subset_size, replace=False)\nprint(f\"Using {len(patient_ids)} patients ({SUBSET_FRACTION*100:.0f}% of full dataset)\")\n# --- end subsample ---\n\npatient_target = labels_df.groupby('patientId').Target.max()\n\ntrain_ids, val_ids = train_test_split(\n    patient_ids, test_size=0.15, random_state=SEED,\n    stratify=patient_target.loc[patient_ids].values\n)\nprint(len(train_ids), len(val_ids))\n\nboxes_by_patient = {\n    pid: g[['x', 'y', 'width', 'height']].dropna().values\n    for pid, g in labels_df.groupby('patientId')\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-28T19:18:43.817717Z","iopub.execute_input":"2026-06-28T19:18:43.818371Z","iopub.status.idle":"2026-06-28T19:19:07.650762Z","shell.execute_reply.started":"2026-06-28T19:18:43.818343Z","shell.execute_reply":"2026-06-28T19:19:07.650097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 5\nIMG_SIZE = 512\n\nclass RSNADataset(Dataset):\n    def __init__(self, patient_ids, img_dir, boxes_by_patient, img_size=IMG_SIZE):\n        self.patient_ids = list(patient_ids)\n        self.img_dir = img_dir\n        self.boxes_by_patient = boxes_by_patient\n        self.img_size = img_size\n\n    def __len__(self):\n        return len(self.patient_ids)\n\n    def __getitem__(self, idx):\n        pid = self.patient_ids[idx]\n        dcm = pydicom.dcmread(os.path.join(self.img_dir, pid + '.dcm'))\n        img = dcm.pixel_array.astype(np.float32)\n        scale = self.img_size / img.shape[0]\n        img = cv2.resize(img, (self.img_size, self.img_size)) / 255.0\n        img_tensor = torch.tensor(np.stack([img, img, img], axis=0), dtype=torch.float32)\n\n        boxes = self.boxes_by_patient.get(pid, np.zeros((0, 4)))\n        if len(boxes) > 0:\n            boxes = boxes * scale\n            boxes_xyxy = boxes.copy()\n            boxes_xyxy[:, 2] = boxes[:, 0] + boxes[:, 2]\n            boxes_xyxy[:, 3] = boxes[:, 1] + boxes[:, 3]\n            labels = np.ones((len(boxes),), dtype=np.int64)\n        else:\n            boxes_xyxy = np.zeros((0, 4), dtype=np.float32)\n            labels = np.zeros((0,), dtype=np.int64)\n\n        target = {\n            'boxes': torch.tensor(boxes_xyxy, dtype=torch.float32),\n            'labels': torch.tensor(labels, dtype=torch.int64),\n            'image_id': torch.tensor([idx]),\n            'patient_id': pid\n        }\n        return img_tensor, target\n\ndef collate_fn(batch):\n    return tuple(zip(*batch))\n\ntrain_dataset = RSNADataset(train_ids, TRAIN_IMG_DIR, boxes_by_patient)\nval_dataset = RSNADataset(val_ids, TRAIN_IMG_DIR, boxes_by_patient)\n\ntrain_loader = DataLoader(train_dataset, batch_size=4, shuffle=True, collate_fn=collate_fn, num_workers=2)\nval_loader = DataLoader(val_dataset, batch_size=4, shuffle=False, collate_fn=collate_fn, num_workers=2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-28T19:19:07.652113Z","iopub.execute_input":"2026-06-28T19:19:07.652555Z","iopub.status.idle":"2026-06-28T19:19:07.662364Z","shell.execute_reply.started":"2026-06-28T19:19:07.652529Z","shell.execute_reply":"2026-06-28T19:19:07.661724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 6 -- sanity-check a few samples visually\ndef show_sample(dataset, n=6):\n    fig, axes = plt.subplots(1, n, figsize=(4 * n, 4))\n    for i in range(n):\n        img, target = dataset[i]\n        ax = axes[i]\n        ax.imshow(img[0].numpy(), cmap='gray')\n        for box in target['boxes']:\n            x1, y1, x2, y2 = box.tolist()\n            ax.add_patch(plt.Rectangle((x1, y1), x2 - x1, y2 - y1, fill=False, edgecolor='red', linewidth=2))\n        ax.set_title(target['patient_id'][:10])\n        ax.axis('off')\n    plt.show()\n\nshow_sample(train_dataset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-28T19:19:07.663228Z","iopub.execute_input":"2026-06-28T19:19:07.663486Z","iopub.status.idle":"2026-06-28T19:19:08.277449Z","shell.execute_reply.started":"2026-06-28T19:19:07.663450Z","shell.execute_reply":"2026-06-28T19:19:08.276559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 7\ndef get_faster_rcnn_model(num_classes=2, pretrained=True):\n    model = fasterrcnn_resnet50_fpn(weights='DEFAULT' if pretrained else None)\n    in_features = model.roi_heads.box_predictor.cls_score.in_features\n    model.roi_heads.box_predictor = FastRCNNPredictor(in_features, num_classes)\n    return model\n\nfaster_rcnn_model = get_faster_rcnn_model(num_classes=2).to(device)  # 2 = background + pneumonia","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-28T19:19:08.279113Z","iopub.execute_input":"2026-06-28T19:19:08.279481Z","iopub.status.idle":"2026-06-28T19:19:08.922626Z","shell.execute_reply.started":"2026-06-28T19:19:08.279452Z","shell.execute_reply":"2026-06-28T19:19:08.921979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 8\nfrom tqdm.auto import tqdm\n\ndef train_one_epoch_torchvision(model, optimizer, data_loader, device, epoch):\n    model.train()\n    total_loss = 0.0\n    pbar = tqdm(data_loader, desc=f\"Epoch {epoch}\", unit=\"batch\")\n    for images, targets in pbar:\n        images = [img.to(device) for img in images]\n        targets = [{k: v.to(device) if torch.is_tensor(v) else v for k, v in t.items()} for t in targets]\n\n        loss_dict = model(images, targets)\n        losses = sum(loss_dict.values())\n\n        optimizer.zero_grad()\n        losses.backward()\n        optimizer.step()\n\n        total_loss += losses.item()\n        pbar.set_postfix(loss=f\"{losses.item():.4f}\")\n\n    return total_loss / len(data_loader)\n\noptimizer = torch.optim.AdamW(\n    [p for p in faster_rcnn_model.parameters() if p.requires_grad],\n    lr=3e-4, \n    weight_decay=0.01\n)\nlr_scheduler = torch.optim.lr_scheduler.StepLR(optimizer, step_size=3, gamma=0.1)\n\nNUM_EPOCHS = 5  # raise if time allows\nfaster_rcnn_losses = []\nfor epoch in range(NUM_EPOCHS):\n    avg_loss = train_one_epoch_torchvision(faster_rcnn_model, optimizer, train_loader, device, epoch)\n    faster_rcnn_losses.append(avg_loss)\n    lr_scheduler.step()\n    print(f\"Epoch {epoch} avg loss: {avg_loss:.4f}\")\n\ntorch.save(faster_rcnn_model.state_dict(), '/kaggle/working/faster_rcnn.pth')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-28T19:19:08.923500Z","iopub.execute_input":"2026-06-28T19:19:08.923834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 9 -- inference helper\n@torch.no_grad()\ndef predict_faster_rcnn(model, images, score_thresh=0.3):\n    model.eval()\n    images = [img.to(device) for img in images]\n    outputs = model(images)\n    results = []\n    for out in outputs:\n        keep = out['scores'] >= score_thresh\n        results.append({\n            'boxes': out['boxes'][keep].cpu().numpy(),\n            'scores': out['scores'][keep].cpu().numpy()\n        })\n    return results","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 10 (corrected)\nss = cv2.ximgproc.segmentation.createSelectiveSearchSegmentation()\n\ndef generate_proposals_selective_search(img_uint8, max_proposals=300):\n    ss.setBaseImage(img_uint8)\n    ss.switchToSelectiveSearchFast()\n    rects = ss.process()\n    rects = rects[:max_proposals]\n    boxes_xyxy = np.zeros_like(rects, dtype=np.float32)\n    boxes_xyxy[:, 0] = rects[:, 0]\n    boxes_xyxy[:, 1] = rects[:, 1]\n    boxes_xyxy[:, 2] = rects[:, 0] + rects[:, 2]\n    boxes_xyxy[:, 3] = rects[:, 1] + rects[:, 3]\n    return boxes_xyxy","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 11 -- precompute & cache proposals (this is the slow step; tqdm shows progress)\nFASTRCNN_MAX_TRAIN = 3000\nFASTRCNN_MAX_VAL = 500\n\n# Drawn from the SAME train_ids/val_ids as Faster R-CNN (Cell 4) for a fair, matched comparison\nfastrcnn_train_ids = train_ids[:FASTRCNN_MAX_TRAIN] if len(train_ids) > FASTRCNN_MAX_TRAIN else train_ids\nfastrcnn_val_ids = val_ids[:FASTRCNN_MAX_VAL] if len(val_ids) > FASTRCNN_MAX_VAL else val_ids\nprint(f\"Fast R-CNN: {len(fastrcnn_train_ids)} train / {len(fastrcnn_val_ids)} val patients\")\n\nproposal_cache = {}\n\ndef precompute_proposals(patient_ids, img_dir, img_size=IMG_SIZE):\n    for pid in tqdm(patient_ids, desc=\"Selective Search proposals\"):\n        if pid in proposal_cache:\n            continue\n        dcm = pydicom.dcmread(os.path.join(img_dir, pid + '.dcm'))\n        img = dcm.pixel_array.astype(np.float32)\n        img = cv2.resize(img, (img_size, img_size))\n        img_uint8 = np.stack([img, img, img], axis=-1).astype(np.uint8)\n        proposal_cache[pid] = generate_proposals_selective_search(img_uint8)\n\nprecompute_proposals(fastrcnn_train_ids, TRAIN_IMG_DIR)\nprecompute_proposals(fastrcnn_val_ids, TRAIN_IMG_DIR)\nprint(f\"Cached proposals for {len(proposal_cache)} images\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 12 -- backbone + Fast R-CNN head (RoIPool over fixed proposals)\ndef build_resnet_backbone():\n    resnet = torchvision.models.resnet50(weights='DEFAULT')\n    return nn.Sequential(*list(resnet.children())[:-2])  # up to layer4, stride 32, 2048 channels\n\nclass FastRCNNHead(nn.Module):\n    def __init__(self, backbone, num_classes, roi_output_size=7, representation_size=1024):\n        super().__init__()\n        self.backbone = backbone\n        self.roi_pool = RoIPool(output_size=roi_output_size, spatial_scale=1/32)\n        feat_channels = 2048\n        pooled_dim = feat_channels * roi_output_size * roi_output_size\n        self.fc1 = nn.Linear(pooled_dim, representation_size)\n        self.fc2 = nn.Linear(representation_size, representation_size)\n        self.cls_score = nn.Linear(representation_size, num_classes)\n        self.bbox_pred = nn.Linear(representation_size, num_classes * 4)\n\n    def forward(self, images, proposals):\n        feats = self.backbone(images)\n        pooled = self.roi_pool(feats, proposals)\n        x = pooled.flatten(start_dim=1)\n        x = F.relu(self.fc1(x))\n        x = F.relu(self.fc2(x))\n        return self.cls_score(x), self.bbox_pred(x)\n\nbackbone = build_resnet_backbone().to(device)\nfast_rcnn_head = FastRCNNHead(backbone, num_classes=2).to(device)\nbox_coder = BoxCoder(weights=(10., 10., 5., 5.))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 13 -- Fast R-CNN dataset (samples proposals into FG/BG per the original paper: IoU>=0.5 -> FG)\nclass FastRCNNDataset(Dataset):\n    def __init__(self, patient_ids, img_dir, boxes_by_patient, proposal_cache, img_size=IMG_SIZE,\n                 fg_iou_thresh=0.5, bg_iou_thresh_hi=0.5, bg_iou_thresh_lo=0.0,\n                 num_samples=64, fg_fraction=0.25):\n        self.patient_ids = list(patient_ids)\n        self.img_dir = img_dir\n        self.boxes_by_patient = boxes_by_patient\n        self.proposal_cache = proposal_cache\n        self.img_size = img_size\n        self.fg_iou_thresh = fg_iou_thresh\n        self.bg_iou_thresh_hi = bg_iou_thresh_hi\n        self.bg_iou_thresh_lo = bg_iou_thresh_lo\n        self.num_samples = num_samples\n        self.fg_fraction = fg_fraction\n\n    def __len__(self):\n        return len(self.patient_ids)\n\n    def __getitem__(self, idx):\n        pid = self.patient_ids[idx]\n        dcm = pydicom.dcmread(os.path.join(self.img_dir, pid + '.dcm'))\n        img = dcm.pixel_array.astype(np.float32)\n        scale = self.img_size / img.shape[0]\n        img = cv2.resize(img, (self.img_size, self.img_size)) / 255.0\n        img_tensor = torch.tensor(np.stack([img, img, img], axis=0), dtype=torch.float32)\n\n        gt_boxes = self.boxes_by_patient.get(pid, np.zeros((0, 4)))\n        if len(gt_boxes) > 0:\n            gt_boxes = gt_boxes * scale\n            gt_xyxy = gt_boxes.copy()\n            gt_xyxy[:, 2] = gt_boxes[:, 0] + gt_boxes[:, 2]\n            gt_xyxy[:, 3] = gt_boxes[:, 1] + gt_boxes[:, 3]\n        else:\n            gt_xyxy = np.zeros((0, 4), dtype=np.float32)\n        gt_xyxy = torch.tensor(gt_xyxy, dtype=torch.float32)\n\n        proposals = torch.tensor(self.proposal_cache[pid], dtype=torch.float32)\n\n        if len(gt_xyxy) > 0:\n            ious = box_iou(proposals, gt_xyxy)\n            max_iou, max_idx = ious.max(dim=1)\n        else:\n            max_iou = torch.zeros(len(proposals))\n            max_idx = torch.zeros(len(proposals), dtype=torch.long)\n\n        fg_idx = torch.where(max_iou >= self.fg_iou_thresh)[0]\n        bg_idx = torch.where((max_iou < self.bg_iou_thresh_hi) & (max_iou >= self.bg_iou_thresh_lo))[0]\n\n        num_fg = min(int(self.num_samples * self.fg_fraction), len(fg_idx))\n        num_bg = self.num_samples - num_fg\n\n        if len(fg_idx) > 0:\n            fg_idx = fg_idx[torch.randperm(len(fg_idx))[:num_fg]]\n        if len(bg_idx) > 0:\n            bg_idx = bg_idx[torch.randperm(len(bg_idx))[:min(num_bg, len(bg_idx))]]\n\n        sampled_idx = torch.cat([fg_idx, bg_idx])\n        if len(sampled_idx) == 0:\n            sampled_idx = torch.randint(0, len(proposals), (1,))\n\n        sampled_proposals = proposals[sampled_idx]\n        sampled_labels = torch.zeros(len(sampled_idx), dtype=torch.int64)\n        sampled_labels[:len(fg_idx)] = 1\n\n        if len(fg_idx) > 0 and len(gt_xyxy) > 0:\n            matched_gt = gt_xyxy[max_idx[fg_idx]]\n            regression_targets = box_coder.encode_single(matched_gt, sampled_proposals[:len(fg_idx)])\n        else:\n            regression_targets = torch.zeros((0, 4))\n\n        return img_tensor, sampled_proposals, sampled_labels, regression_targets\n\ndef fastrcnn_collate_fn(batch):\n    images = torch.stack([b[0] for b in batch])\n    proposals = [b[1] for b in batch]\n    labels = [b[2] for b in batch]\n    reg_targets = [b[3] for b in batch]\n    return images, proposals, labels, reg_targets\n\nfastrcnn_train_dataset = FastRCNNDataset(fastrcnn_train_ids, TRAIN_IMG_DIR, boxes_by_patient, proposal_cache)\nfastrcnn_val_dataset = FastRCNNDataset(fastrcnn_val_ids, TRAIN_IMG_DIR, boxes_by_patient, proposal_cache)\n\nfastrcnn_train_loader = DataLoader(fastrcnn_train_dataset, batch_size=4, shuffle=True, collate_fn=fastrcnn_collate_fn, num_workers=2)\nfastrcnn_val_loader = DataLoader(fastrcnn_val_dataset, batch_size=4, shuffle=False, collate_fn=fastrcnn_collate_fn, num_workers=2)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 14 -- training loop (tqdm, same style as Faster R-CNN)\ndef train_one_epoch_fastrcnn(head_model, optimizer, data_loader, device, epoch):\n    head_model.train()\n    total_loss = 0.0\n    pbar = tqdm(data_loader, desc=f\"Fast R-CNN Epoch {epoch}\", unit=\"batch\")\n    for images, proposals, labels, reg_targets in pbar:\n        images = images.to(device)\n        proposals = [p.to(device) for p in proposals]\n        labels_cat = torch.cat(labels).to(device)\n        nonempty = [r for r in reg_targets if len(r) > 0]\n        reg_targets_cat = torch.cat(nonempty).to(device) if nonempty else torch.zeros((0, 4)).to(device)\n\n        scores, bbox_deltas = head_model(images, proposals)\n        cls_loss = F.cross_entropy(scores, labels_cat)\n\n        fg_mask = labels_cat == 1\n        if fg_mask.sum() > 0:\n            bbox_deltas_fg = bbox_deltas[fg_mask].view(-1, 2, 4)[:, 1, :]\n            reg_loss = F.smooth_l1_loss(bbox_deltas_fg, reg_targets_cat)\n        else:\n            reg_loss = torch.tensor(0.0, device=device)\n\n        loss = cls_loss + reg_loss\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n\n        total_loss += loss.item()\n        pbar.set_postfix(loss=f\"{loss.item():.4f}\", cls=f\"{cls_loss.item():.4f}\", reg=f\"{reg_loss.item():.4f}\")\n\n    return total_loss / len(data_loader)\n\noptimizer_frcnn = torch.optim.SGD(fast_rcnn_head.parameters(), lr=0.001, momentum=0.9, weight_decay=0.0005)\n\nNUM_EPOCHS_FRCNN = 5\nfast_rcnn_losses = []\nfor epoch in range(NUM_EPOCHS_FRCNN):\n    avg_loss = train_one_epoch_fastrcnn(fast_rcnn_head, optimizer_frcnn, fastrcnn_train_loader, device, epoch)\n    fast_rcnn_losses.append(avg_loss)\n    print(f\"Epoch {epoch} avg loss: {avg_loss:.4f}\")\n\ntorch.save(fast_rcnn_head.state_dict(), '/kaggle/working/fast_rcnn.pth')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 15 -- Fast R-CNN inference helpers\ndef load_image_and_proposals(pid, img_dir, proposal_cache, img_size=IMG_SIZE):\n    dcm = pydicom.dcmread(os.path.join(img_dir, pid + '.dcm'))\n    img = dcm.pixel_array.astype(np.float32)\n    img = cv2.resize(img, (img_size, img_size)) / 255.0\n    img_tensor = torch.tensor(np.stack([img, img, img], axis=0), dtype=torch.float32)\n    proposals = torch.tensor(proposal_cache[pid], dtype=torch.float32)\n    return img_tensor, proposals\n\n@torch.no_grad()\ndef predict_fast_rcnn(head_model, images, proposals_list, score_thresh=0.3, nms_thresh=0.3):\n    head_model.eval()\n    images = images.to(device)\n    proposals_list = [p.to(device) for p in proposals_list]\n    scores, bbox_deltas = head_model(images, proposals_list)\n    probs = F.softmax(scores, dim=1)\n\n    results = []\n    start = 0\n    for proposals in proposals_list:\n        n = len(proposals)\n        p_scores = probs[start:start + n]\n        p_deltas = bbox_deltas[start:start + n].view(-1, 2, 4)[:, 1, :]\n        boxes = box_coder.decode_single(p_deltas, proposals)\n        cls_scores = p_scores[:, 1]\n\n        keep = cls_scores >= score_thresh\n        boxes, cls_scores = boxes[keep], cls_scores[keep]\n        if len(boxes) > 0:\n            nms_keep = nms(boxes, cls_scores, nms_thresh)\n            boxes, cls_scores = boxes[nms_keep], cls_scores[nms_keep]\n\n        results.append({'boxes': boxes.cpu().numpy(), 'scores': cls_scores.cpu().numpy()})\n        start += n\n    return results","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 16 -- the official RSNA metric: mAP over IoU thresholds 0.40-0.75 (step 0.05),\n# precision(t) = TP/(TP+FP+FN) per image, averaged over thresholds then images\ndef compute_iou_matrix(boxes_true, boxes_pred):\n    if len(boxes_true) == 0 or len(boxes_pred) == 0:\n        return np.zeros((len(boxes_true), len(boxes_pred)))\n    return box_iou(torch.tensor(boxes_true), torch.tensor(boxes_pred)).numpy()\n\ndef rsna_map_single_image(boxes_true, boxes_pred, scores, thresholds=np.arange(0.4, 0.8, 0.05)):\n    if len(boxes_pred) == 0:\n        return 1.0 if len(boxes_true) == 0 else 0.0\n    order = np.argsort(-scores)\n    boxes_pred = boxes_pred[order]\n    iou_matrix = compute_iou_matrix(boxes_true, boxes_pred)\n\n    precisions = []\n    for t in thresholds:\n        matched_gt = set()\n        tp = 0\n        for j in range(len(boxes_pred)):\n            if len(boxes_true) == 0:\n                break\n            ious = iou_matrix[:, j]\n            best_gt = np.argmax(ious)\n            if ious[best_gt] >= t and best_gt not in matched_gt:\n                tp += 1\n                matched_gt.add(best_gt)\n        fp = len(boxes_pred) - tp\n        fn = len(boxes_true) - tp\n        precision_t = tp / (tp + fp + fn) if (tp + fp + fn) > 0 else 1.0\n        precisions.append(precision_t)\n    return float(np.mean(precisions))\n\ndef rsna_map_dataset(all_boxes_true, all_boxes_pred, all_scores):\n    scores_per_image = [\n        rsna_map_single_image(bt, bp, sc)\n        for bt, bp, sc in zip(all_boxes_true, all_boxes_pred, all_scores)\n    ]\n    return float(np.mean(scores_per_image)), scores_per_image","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 17 -- evaluate both models on the SAME validation patients (fastrcnn_val_ids) for a fair comparison\n@torch.no_grad()\ndef evaluate_faster_rcnn_map(model, patient_ids, img_dir, boxes_by_patient, score_thresh=0.3):\n    dataset = RSNADataset(patient_ids, img_dir, boxes_by_patient)\n    all_true, all_pred, all_scores = [], [], []\n    for i in tqdm(range(len(dataset)), desc=\"Evaluating Faster R-CNN\"):\n        img, target = dataset[i]\n        pred = predict_faster_rcnn(model, [img], score_thresh=score_thresh)[0]\n        all_true.append(target['boxes'].numpy())\n        all_pred.append(pred['boxes'])\n        all_scores.append(pred['scores'])\n    return rsna_map_dataset(all_true, all_pred, all_scores)\n\n@torch.no_grad()\ndef evaluate_fast_rcnn_map(head_model, patient_ids, img_dir, boxes_by_patient, proposal_cache, score_thresh=0.3):\n    all_true, all_pred, all_scores = [], [], []\n    for pid in tqdm(patient_ids, desc=\"Evaluating Fast R-CNN\"):\n        img_tensor, proposals = load_image_and_proposals(pid, img_dir, proposal_cache)\n        pred = predict_fast_rcnn(head_model, img_tensor.unsqueeze(0), [proposals], score_thresh=score_thresh)[0]\n\n        gt_boxes = boxes_by_patient.get(pid, np.zeros((0, 4)))\n        scale = IMG_SIZE / 1024\n        if len(gt_boxes) > 0:\n            gtb = gt_boxes * scale\n            gt_xyxy = gtb.copy()\n            gt_xyxy[:, 2] = gtb[:, 0] + gtb[:, 2]\n            gt_xyxy[:, 3] = gtb[:, 1] + gtb[:, 3]\n        else:\n            gt_xyxy = np.zeros((0, 4))\n\n        all_true.append(gt_xyxy)\n        all_pred.append(pred['boxes'])\n        all_scores.append(pred['scores'])\n    return rsna_map_dataset(all_true, all_pred, all_scores)\n\nfaster_rcnn_map, faster_rcnn_per_image = evaluate_faster_rcnn_map(\n    faster_rcnn_model, fastrcnn_val_ids, TRAIN_IMG_DIR, boxes_by_patient\n)\nfast_rcnn_map, fast_rcnn_per_image = evaluate_fast_rcnn_map(\n    fast_rcnn_head, fastrcnn_val_ids, TRAIN_IMG_DIR, boxes_by_patient, proposal_cache\n)\n\nprint(f\"Faster R-CNN validation mAP: {faster_rcnn_map:.4f}\")\nprint(f\"Fast R-CNN   validation mAP: {fast_rcnn_map:.4f}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 18 -- side-by-side predictions (green = ground truth, red = predicted)\ndef visualize_comparison(patient_ids, n=4, score_thresh=0.3):\n    fig, axes = plt.subplots(2, n, figsize=(4 * n, 8))\n    for col, pid in enumerate(patient_ids[:n]):\n        img_tensor, proposals = load_image_and_proposals(pid, TRAIN_IMG_DIR, proposal_cache)\n        gt_boxes = boxes_by_patient.get(pid, np.zeros((0, 4)))\n        scale = IMG_SIZE / 1024\n        if len(gt_boxes) > 0:\n            gtb = gt_boxes * scale\n            gt_xyxy = gtb.copy()\n            gt_xyxy[:, 2] = gtb[:, 0] + gtb[:, 2]\n            gt_xyxy[:, 3] = gtb[:, 1] + gtb[:, 3]\n        else:\n            gt_xyxy = np.zeros((0, 4))\n\n        pred_frcnn = predict_faster_rcnn(faster_rcnn_model, [img_tensor], score_thresh=score_thresh)[0]\n        ax = axes[0, col]\n        ax.imshow(img_tensor[0].numpy(), cmap='gray')\n        for x1, y1, x2, y2 in gt_xyxy:\n            ax.add_patch(plt.Rectangle((x1, y1), x2 - x1, y2 - y1, fill=False, edgecolor='lime', linewidth=2))\n        for x1, y1, x2, y2 in pred_frcnn['boxes']:\n            ax.add_patch(plt.Rectangle((x1, y1), x2 - x1, y2 - y1, fill=False, edgecolor='red', linewidth=2))\n        ax.set_title(f\"Faster R-CNN\\n{pid[:10]}\")\n        ax.axis('off')\n\n        pred_fast = predict_fast_rcnn(fast_rcnn_head, img_tensor.unsqueeze(0), [proposals], score_thresh=score_thresh)[0]\n        ax = axes[1, col]\n        ax.imshow(img_tensor[0].numpy(), cmap='gray')\n        for x1, y1, x2, y2 in gt_xyxy:\n            ax.add_patch(plt.Rectangle((x1, y1), x2 - x1, y2 - y1, fill=False, edgecolor='lime', linewidth=2))\n        for x1, y1, x2, y2 in pred_fast['boxes']:\n            ax.add_patch(plt.Rectangle((x1, y1), x2 - x1, y2 - y1, fill=False, edgecolor='red', linewidth=2))\n        ax.set_title(f\"Fast R-CNN\\n{pid[:10]}\")\n        ax.axis('off')\n\n    plt.tight_layout()\n    plt.show()\n\npneumonia_val_ids = [pid for pid in fastrcnn_val_ids if len(boxes_by_patient.get(pid, [])) > 0]\nvisualize_comparison(pneumonia_val_ids, n=4)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Report: Fast R-CNN vs. Faster R-CNN for Pneumonia Detection (RSNA Challenge)\n\n### 1. Model Architecture\n- **Backbone**: ResNet-50 pretrained on ImageNet for both models, ensuring a fair architectural comparison.\n- **Faster R-CNN**: torchvision's `fasterrcnn_resnet50_fpn` — ResNet-50 + FPN backbone, a trainable Region Proposal Network (RPN) generating proposals end-to-end, followed by RoIAlign + a two-layer FC head for classification and box regression. Proposals are *learned jointly* with the detection head.\n- **Fast R-CNN**: Plain ResNet-50 (no FPN) up to the last conv block (stride 32, 2048 channels), with RoIPool (7×7) over **fixed, externally generated region proposals** from OpenCV Selective Search, followed by the same two-layer FC head design (1024-d FC ×2 → classifier + box regressor). Proposals are *not* learned — this is the defining architectural difference from Faster R-CNN.\n- Both models output 2 classes: background vs. pneumonia opacity.\n\n### 2. Training Process\n- Faster R-CNN trained on [N] patients for [E] epochs, SGD (lr=0.005, momentum=0.9), images resized to 512×512.\n- Fast R-CNN trained on a subset of [N'] patients (capped due to the cost of precomputing Selective Search proposals — ~[X] proposals/image), same 512×512 resolution, SGD (lr=0.001). Proposals were precomputed once and cached, then sampled per-image into 25% foreground (IoU≥0.5) / 75% background per the original Fast R-CNN sampling strategy.\n- [Insert your loss curves / plots here, describe convergence behavior.]\n\n### 3. Results and Comparison (RSNA mAP, IoU 0.40–0.75)\n| Model | Validation mAP |\n|---|---|\n| Faster R-CNN | [0.8000] |\n| Fast R-CNN | [0.0250] |\n\n[Discuss: which performed better and why — e.g., Faster R-CNN's learned RPN typically adapts proposal quality to the detection task, while Fast R-CNN is bottlenecked by Selective Search's class-agnostic, fixed proposals, which may poorly localize subtle lung opacities. Also discuss inference speed: Fast R-CNN requires the separate Selective Search step at inference time, which is slower than Faster R-CNN's integrated RPN.]","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}