{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# RSNA Pneumonia Detection — 2-Stage Pipeline\n## Fast R-CNN + Faster R-CNN Ensemble\n\n**Strategy (1st-place informed):**\n1. **Classification** — DenseNet121 (ImageNet pretrained) scores each image for pneumonia probability\n2. **Fast R-CNN** — Selective Search proposals + ResNet50-FPN + RoI Align\n3. **Faster R-CNN** — Built-in RPN + ResNet50-FPN (same backbone, independent detector)\n4. **Ensemble** — WBF merges both detectors; scores multiplied by classifier probability\n5. **Box resize ×0.875** — Key 1st-place trick: test boxes annotated as intersections are smaller than training boxes\n\n**Expected score:** 0.22–0.25 vs. 0.155 baseline (Mask-RCNN at 256×256)\n\n---\n**Kaggle inputs required:**\n- `rsna-pneumonia-detection-challenge` (competition data)","metadata":{}},{"cell_type":"code","source":"!pip install timm pydicom ensemble-boxes albumentations opencv-contrib-python -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-28T18:42:30.371799Z","iopub.execute_input":"2026-06-28T18:42:30.372547Z","iopub.status.idle":"2026-06-28T18:42:36.192571Z","shell.execute_reply.started":"2026-06-28T18:42:30.372513Z","shell.execute_reply":"2026-06-28T18:42:36.191541Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport gc\nimport pickle\nimport random\nimport warnings\nfrom pathlib import Path\nfrom collections import OrderedDict\n\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport pydicom\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision\nimport torchvision.transforms.functional as TF\nfrom torchvision.models.detection import FasterRCNN\nfrom torchvision.models.detection.faster_rcnn import FastRCNNPredictor\nfrom torchvision.models.detection.backbone_utils import resnet_fpn_backbone\nfrom torchvision.models.detection.rpn import AnchorGenerator\nimport torchvision.ops as ops\n\nimport timm\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score\nfrom ensemble_boxes import weighted_boxes_fusion\nfrom tqdm.auto import tqdm\n\nwarnings.filterwarnings('ignore')\nprint(f'PyTorch {torch.__version__} | torchvision {torchvision.__version__}')\nprint(f'CUDA available: {torch.cuda.is_available()}')\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(f'Device: {device}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-28T18:42:36.194538Z","iopub.execute_input":"2026-06-28T18:42:36.194986Z","iopub.status.idle":"2026-06-28T18:42:49.761334Z","shell.execute_reply.started":"2026-06-28T18:42:36.194938Z","shell.execute_reply":"2026-06-28T18:42:49.760531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Config:\n    # ── Kaggle paths ────────────────────────────────────────────────────\n    DATA_DIR   = Path('/kaggle/input/competitions/rsna-pneumonia-detection-challenge')\n    WORK_DIR   = Path('/kaggle/working')\n\n    TRAIN_DCM  = DATA_DIR / 'stage_2_train_images'\n    TEST_DCM   = DATA_DIR / 'stage_2_test_images'\n    LABELS_CSV = DATA_DIR / 'stage_2_train_labels.csv'\n\n    CLS_IMGS   = WORK_DIR / 'cls_images'\n    DET_IMGS   = WORK_DIR / 'det_images'\n    PROPOSALS  = WORK_DIR / 'proposals'\n    MODELS     = WORK_DIR / 'models'\n\n    # ── Image sizes ─────────────────────────────────────────────────────\n    CLS_SIZE   = 320\n    DET_SIZE   = 512\n\n    # ── Classification ──────────────────────────────────────────────────\n    CLS_MODEL  = 'densenet121'\n    CLS_EPOCHS = 10\n    CLS_BATCH  = 32\n    CLS_LR     = 1e-4\n    CLS_TTA    = 10\n\n    # ── Detection (shared) ──────────────────────────────────────────────\n    DET_EPOCHS  = 5\n    DET_BATCH   = 4\n    DET_LR      = 0.005\n    DET_LR_STEP = 5\n\n    # ── Fast R-CNN ──────────────────────────────────────────────────────\n    SS_PROPOSALS = 2000\n    SS_MIN_SIZE  = 10\n\n    # ── Post-processing ─────────────────────────────────────────────────\n    BOX_SCALE   = 0.875\n    WBF_IOU     = 0.4\n    CONF_THRESH = 0.15\n\n    # ── General ─────────────────────────────────────────────────────────\n    VAL_SPLIT = 0.1\n    SEED      = 42\n    DEBUG     = False  # True → 200 images, 2 epochs (smoke test)\n\n\nfor d in [Config.CLS_IMGS, Config.DET_IMGS, Config.PROPOSALS, Config.MODELS]:\n    d.mkdir(parents=True, exist_ok=True)\n\n\ndef seed_everything(seed):\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n\n\nseed_everything(Config.SEED)\n\nif Config.DEBUG:\n    Config.CLS_EPOCHS = 2\n    Config.DET_EPOCHS = 2\n    print('DEBUG mode active — limited epochs and images')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-28T18:42:49.762308Z","iopub.execute_input":"2026-06-28T18:42:49.76306Z","iopub.status.idle":"2026-06-28T18:42:49.777366Z","shell.execute_reply.started":"2026-06-28T18:42:49.763035Z","shell.execute_reply":"2026-06-28T18:42:49.776715Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1 · Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(Config.LABELS_CSV)\npatient_df = df.groupby('patientId').agg(Target=('Target', 'max')).reset_index()\n\nn_pos = (patient_df['Target'] == 1).sum()\nn_neg = (patient_df['Target'] == 0).sum()\nprint(f'Total patients : {len(patient_df)}')\nprint(f'Positive (pneumonia): {n_pos}  ({n_pos/len(patient_df):.1%})')\nprint(f'Negative            : {n_neg}  ({n_neg/len(patient_df):.1%})')\n\ndf_pos = df[df['Target'] == 1].copy()\nprint(f'\\nBoxes per positive patient:')\nprint(df_pos.groupby('patientId').size().value_counts().sort_index().to_string())\n\nfig, axes = plt.subplots(1, 3, figsize=(15, 4))\n\naxes[0].bar(['Negative', 'Positive'], [n_neg, n_pos], color=['steelblue', 'tomato'])\naxes[0].set_title('Class Distribution')\naxes[0].set_ylabel('Count')\n\nbox_counts = df_pos.groupby('patientId').size()\naxes[1].hist(box_counts, bins=range(1, box_counts.max() + 2), color='tomato', align='left', rwidth=0.8)\naxes[1].set_title('Boxes per Positive Image')\naxes[1].set_xlabel('Number of boxes')\n\naxes[2].scatter(df_pos['width'], df_pos['height'], alpha=0.25, s=5, color='tomato')\naxes[2].set_title('Box Width vs Height (pixels, 1024×1024 space)')\naxes[2].set_xlabel('Width'); axes[2].set_ylabel('Height')\n\nplt.tight_layout()\nplt.savefig(Config.WORK_DIR / 'eda.png', dpi=100, bbox_inches='tight')\nplt.show()\n\n# Sample images\npos_pids = df_pos['patientId'].unique()[:4]\nneg_pids = patient_df[patient_df['Target'] == 0]['patientId'].values[:4]\n\nfig, axes = plt.subplots(2, 4, figsize=(16, 8))\nfor i, pid in enumerate([*pos_pids, *neg_pids]):\n    ax = axes.flatten()[i]\n    try:\n        dcm = pydicom.dcmread(str(Config.TRAIN_DCM / f'{pid}.dcm'))\n        img = dcm.pixel_array\n        if img.max() > 255:\n            img = ((img - img.min()) / (img.max() - img.min()) * 255).astype(np.uint8)\n        ax.imshow(img, cmap='gray')\n        if i < 4:\n            for _, r in df[df['patientId'] == pid].iterrows():\n                rect = patches.Rectangle(\n                    (r['x'], r['y']), r['width'], r['height'],\n                    linewidth=2, edgecolor='red', facecolor='none'\n                )\n                ax.add_patch(rect)\n    except Exception:\n        ax.text(0.5, 0.5, 'No image', ha='center', transform=ax.transAxes)\n    ax.set_title(f\"{'POS' if i < 4 else 'NEG'}: {pid[:10]}...\")\n    ax.axis('off')\n\nplt.suptitle('Sample Images  (top: positive with boxes, bottom: negative)', fontsize=12)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-28T18:42:49.779229Z","iopub.execute_input":"2026-06-28T18:42:49.77958Z","iopub.status.idle":"2026-06-28T18:42:53.093091Z","shell.execute_reply.started":"2026-06-28T18:42:49.779557Z","shell.execute_reply":"2026-06-28T18:42:53.09226Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2 · DICOM → PNG Preprocessing\n\nEach DICOM is:\n1. Loaded with pydicom\n2. Normalised to uint8 (handles 16-bit DICOMs)\n3. Inverted if PhotometricInterpretation = MONOCHROME1\n4. CLAHE applied (improves lung contrast — standard for chest X-ray models)\n5. Resized to 320×320 for classification, 512×512 for detection\n6. Saved as 3-channel BGR PNG (required by COCO-pretrained backbones)","metadata":{}},{"cell_type":"code","source":"def dicom_to_png(dcm_path, out_path, size):\n    try:\n        dcm = pydicom.dcmread(str(dcm_path))\n        img = dcm.pixel_array\n\n        # Normalise to uint8\n        if img.dtype != np.uint8:\n            mn, mx = img.min(), img.max()\n            img = ((img - mn) / max(mx - mn, 1) * 255).astype(np.uint8)\n\n        # Invert MONOCHROME1 (white = 0) so lung fields are dark\n        if hasattr(dcm, 'PhotometricInterpretation') and \\\n                dcm.PhotometricInterpretation == 'MONOCHROME1':\n            img = 255 - img\n\n        # CLAHE contrast enhancement\n        clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\n        img   = clahe.apply(img)\n\n        img = cv2.resize(img, (size, size), interpolation=cv2.INTER_AREA)\n        img = cv2.cvtColor(img, cv2.COLOR_GRAY2BGR)\n        cv2.imwrite(str(out_path), img)\n        return True\n    except Exception as e:\n        print(f'Error {dcm_path.name}: {e}')\n        return False\n\n\ndef convert_split(dcm_dir, out_dir, size, patient_ids=None):\n    out_dir.mkdir(parents=True, exist_ok=True)\n    dcm_files = sorted(dcm_dir.glob('*.dcm'))\n    if patient_ids is not None:\n        patient_ids = set(patient_ids)\n        dcm_files = [f for f in dcm_files if f.stem in patient_ids]\n    skipped = 0\n    for dcm_path in tqdm(dcm_files, desc=f'{out_dir.parent.name}/{out_dir.name}'):\n        out_path = out_dir / f'{dcm_path.stem}.png'\n        if out_path.exists():\n            skipped += 1\n            continue\n        dicom_to_png(dcm_path, out_path, size)\n    print(f'  Skipped {skipped} cached files.')\n\n\n# Patient IDs\nall_pids = patient_df['patientId'].values\nif Config.DEBUG:\n    all_pids = all_pids[:200]\n    print(f'DEBUG: using {len(all_pids)} train patients')\n\ntest_pids = [f.stem for f in sorted(Config.TEST_DCM.glob('*.dcm'))]\nif Config.DEBUG:\n    test_pids = test_pids[:50]\n\nprint('Converting to 320×320 (classification)...')\nconvert_split(Config.TRAIN_DCM, Config.CLS_IMGS / 'train', Config.CLS_SIZE, all_pids)\nconvert_split(Config.TEST_DCM,  Config.CLS_IMGS / 'test',  Config.CLS_SIZE)\n\nprint('Converting to 512×512 (detection)...')\nconvert_split(Config.TRAIN_DCM, Config.DET_IMGS / 'train', Config.DET_SIZE, all_pids)\nconvert_split(Config.TEST_DCM,  Config.DET_IMGS / 'test',  Config.DET_SIZE)\n\nprint('Done converting images.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-28T18:42:53.094072Z","iopub.execute_input":"2026-06-28T18:42:53.094284Z","iopub.status.idle":"2026-06-28T18:42:55.370653Z","shell.execute_reply.started":"2026-06-28T18:42:53.094264Z","shell.execute_reply":"2026-06-28T18:42:55.369686Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4 · Classification Stage\n\nDenseNet121 trained to predict *pneumonia probability* per image.  \nThis score is later multiplied by each detection box score to penalise false-positive boxes in negative images.","metadata":{}},{"cell_type":"code","source":"# Train / val split stratified by Target\nlabels_dict = patient_df.set_index('patientId')['Target'].to_dict()\npids_subset = [p for p in all_pids if p in labels_dict]\n\ntrain_pids, val_pids = train_test_split(\n    pids_subset,\n    test_size=Config.VAL_SPLIT,\n    stratify=[labels_dict[p] for p in pids_subset],\n    random_state=Config.SEED\n)\nprint(f'Train: {len(train_pids)} | Val: {len(val_pids)}')\n\n# Albumentations transforms\ntrain_cls_tfm = A.Compose([\n    A.Resize(Config.CLS_SIZE, Config.CLS_SIZE),\n    A.HorizontalFlip(p=0.5),\n    A.RandomBrightnessContrast(brightness_limit=0.2, contrast_limit=0.2, p=0.5),\n    A.ShiftScaleRotate(shift_limit=0.05, scale_limit=0.1, rotate_limit=10, p=0.5),\n    A.CoarseDropout(max_holes=8, max_height=32, max_width=32, p=0.3),\n    A.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n    ToTensorV2(),\n])\nval_cls_tfm = A.Compose([\n    A.Resize(Config.CLS_SIZE, Config.CLS_SIZE),\n    A.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n    ToTensorV2(),\n])\n\n\nclass RSNAClassDataset(Dataset):\n    def __init__(self, pids, img_dir, labels, transform, is_train=False):\n        self.pids      = list(pids)\n        self.img_dir   = Path(img_dir)\n        self.labels    = labels\n        self.transform = transform\n        self.is_train  = is_train\n\n    def __len__(self):\n        return len(self.pids)\n\n    def __getitem__(self, idx):\n        pid = self.pids[idx]\n        img = cv2.imread(str(self.img_dir / f'{pid}.png'))\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n        # 1st-place technique: 50% probability of channel inversion\n        if self.is_train and random.random() < 0.5:\n            img = 255 - img\n\n        img   = self.transform(image=img)['image']\n        label = torch.tensor(self.labels.get(pid, 0), dtype=torch.long)\n        return img, label\n\n\ncls_train_ds = RSNAClassDataset(train_pids, Config.CLS_IMGS / 'train',\n                                 labels_dict, train_cls_tfm, is_train=True)\ncls_val_ds   = RSNAClassDataset(val_pids,   Config.CLS_IMGS / 'train',\n                                 labels_dict, val_cls_tfm,   is_train=False)\n\ncls_train_loader = DataLoader(cls_train_ds, batch_size=Config.CLS_BATCH,\n                               shuffle=True,  num_workers=4, pin_memory=True)\ncls_val_loader   = DataLoader(cls_val_ds,   batch_size=Config.CLS_BATCH,\n                               shuffle=False, num_workers=4, pin_memory=True)\n\ncls_model = timm.create_model(Config.CLS_MODEL, pretrained=True, num_classes=2).to(device)\nprint(f'Classifier: {Config.CLS_MODEL} | params: {sum(p.numel() for p in cls_model.parameters())/1e6:.1f}M')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-28T18:42:55.371655Z","iopub.execute_input":"2026-06-28T18:42:55.372125Z","iopub.status.idle":"2026-06-28T18:42:57.831862Z","shell.execute_reply.started":"2026-06-28T18:42:55.372101Z","shell.execute_reply":"2026-06-28T18:42:57.831071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# optimizer = optim.AdamW(cls_model.parameters(), lr=Config.CLS_LR, weight_decay=1e-5)\n# scheduler = optim.lr_scheduler.CosineAnnealingLR(optimizer, T_max=Config.CLS_EPOCHS)\n# criterion = nn.CrossEntropyLoss()\n\n# best_auc    = 0.0\n# cls_history = []\n\n# for epoch in range(Config.CLS_EPOCHS):\n#     # ── Train ────────────────────────────────────────────────────────\n#     cls_model.train()\n#     train_loss = 0.0\n#     for imgs, labels in tqdm(cls_train_loader, desc=f'CLS {epoch+1}/{Config.CLS_EPOCHS}'):\n#         imgs, labels = imgs.to(device), labels.to(device)\n#         loss = criterion(cls_model(imgs), labels)\n#         optimizer.zero_grad()\n#         loss.backward()\n#         optimizer.step()\n#         train_loss += loss.item()\n#     scheduler.step()\n\n#     # ── Validate ─────────────────────────────────────────────────────\n#     cls_model.eval()\n#     all_probs, all_labels = [], []\n#     with torch.no_grad():\n#         for imgs, labels in cls_val_loader:\n#             probs = torch.softmax(cls_model(imgs.to(device)), dim=1)[:, 1].cpu()\n#             all_probs.extend(probs.tolist())\n#             all_labels.extend(labels.tolist())\n\n#     val_auc  = roc_auc_score(all_labels, all_probs)\n#     avg_loss = train_loss / len(cls_train_loader)\n#     cls_history.append({'epoch': epoch + 1, 'loss': avg_loss, 'auc': val_auc})\n#     print(f'  Epoch {epoch+1:3d} | loss {avg_loss:.4f} | val AUC {val_auc:.4f}')\n\n#     if val_auc > best_auc:\n#         best_auc = val_auc\n#         torch.save(cls_model.state_dict(), Config.MODELS / 'cls_best.pth')\n\n# print(f'\\nBest val AUC: {best_auc:.4f}')\n# cls_model.load_state_dict(torch.load(Config.MODELS / 'cls_best.pth', map_location=device))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-28T14:59:15.234107Z","iopub.execute_input":"2026-06-28T14:59:15.234702Z","iopub.status.idle":"2026-06-28T14:59:15.239362Z","shell.execute_reply.started":"2026-06-28T14:59:15.234668Z","shell.execute_reply":"2026-06-28T14:59:15.23857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # ── TTA Inference (10×: 5 scales × 2 flips) ─────────────────────────────────\n\n# def cls_tta_predict(model, img_bgr):\n#     \"\"\"Return mean pneumonia probability over 10 TTA passes.\"\"\"\n#     model.eval()\n#     probs = []\n#     base  = Config.CLS_SIZE\n#     with torch.no_grad():\n#         for scale in [0.85, 0.925, 1.0, 1.075, 1.15]:\n#             sz  = int(base * scale)\n#             for flip in (False, True):\n#                 img = cv2.resize(img_bgr, (sz, sz))\n#                 img = cv2.resize(img, (base, base))\n#                 if flip:\n#                     img = cv2.flip(img, 1)\n#                 img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n#                 t   = val_cls_tfm(image=img)['image'].unsqueeze(0).to(device)\n#                 probs.append(torch.softmax(model(t), dim=1)[0, 1].item())\n#     return float(np.mean(probs))\n\n\n# cls_scores = {}\n\n# for pid in tqdm(list(all_pids), desc='CLS scoring train+val'):\n#     img = cv2.imread(str(Config.CLS_IMGS / 'train' / f'{pid}.png'))\n#     if img is not None:\n#         cls_scores[pid] = cls_tta_predict(cls_model, img)\n\n# for pid in tqdm(test_pids, desc='CLS scoring test'):\n#     img = cv2.imread(str(Config.CLS_IMGS / 'test' / f'{pid}.png'))\n#     if img is not None:\n#         cls_scores[pid] = cls_tta_predict(cls_model, img)\n\n# with open(Config.MODELS / 'cls_scores.pkl', 'wb') as f:\n#     pickle.dump(cls_scores, f)\n\n# pos_scores = [cls_scores[p] for p in train_pids if labels_dict.get(p) == 1 and p in cls_scores]\n# neg_scores = [cls_scores[p] for p in train_pids if labels_dict.get(p) == 0 and p in cls_scores]\n# print(f'Train positive score: {np.mean(pos_scores):.3f} ± {np.std(pos_scores):.3f}')\n# print(f'Train negative score: {np.mean(neg_scores):.3f} ± {np.std(neg_scores):.3f}')\n\n# gc.collect()\n# torch.cuda.empty_cache()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-28T14:59:15.24061Z","iopub.execute_input":"2026-06-28T14:59:15.240886Z","iopub.status.idle":"2026-06-28T14:59:15.361735Z","shell.execute_reply.started":"2026-06-28T14:59:15.240865Z","shell.execute_reply":"2026-06-28T14:59:15.361028Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5 · Selective Search Proposal Generation (for Fast R-CNN)\n\nSelective Search generates ~2000 region proposals per image using colour and texture similarity grouping.  \nProposals are **pre-computed and cached** as `.npy` files — this takes 1–3 hours but only runs once.","metadata":{}},{"cell_type":"code","source":"# def generate_proposals(pid, img_path, out_path, n=2000, min_size=10):\n#     \"\"\"Selective Search → xyxy proposals, cached as .npy.\"\"\"\n#     out_path = Path(out_path)\n#     if out_path.exists():\n#         return 'cached'\n\n#     img = cv2.imread(str(img_path))\n#     if img is None:\n#         np.save(str(out_path), np.zeros((0, 4), dtype=np.float32))\n#         return 'empty'\n\n#     ss = cv2.ximgproc.segmentation.createSelectiveSearchSegmentation()\n#     ss.setBaseImage(img)\n#     ss.switchToSelectiveSearchFast()\n#     rects = ss.process()  # (x, y, w, h)\n\n#     proposals = []\n#     for x, y, w, h in rects[:n]:\n#         if w >= min_size and h >= min_size:\n#             proposals.append([float(x), float(y), float(x + w), float(y + h)])\n\n#     arr = np.array(proposals, dtype=np.float32) if proposals \\\n#           else np.zeros((0, 4), dtype=np.float32)\n#     np.save(str(out_path), arr)\n#     return len(arr)\n\n\n# # Build work list: train + val + test images\n# ss_tasks = []\n# for pid in list(all_pids):\n#     img_p = Config.DET_IMGS / 'train' / f'{pid}.png'\n#     ss_tasks.append((pid, img_p, Config.PROPOSALS / f'{pid}.npy'))\n# for pid in test_pids:\n#     img_p = Config.DET_IMGS / 'test' / f'{pid}.png'\n#     ss_tasks.append((pid, img_p, Config.PROPOSALS / f'{pid}.npy'))\n\n# print(f'Generating proposals for {len(ss_tasks)} images...')\n# print('Cached images are skipped. Expected time: 1-3 h on a 4-core CPU.')\n\n# new_count = 0\n# for pid, img_p, out_p in tqdm(ss_tasks):\n#     result = generate_proposals(pid, img_p, out_p,\n#                                  Config.SS_PROPOSALS, Config.SS_MIN_SIZE)\n#     if result != 'cached':\n#         new_count += 1\n\n# print(f'Done. Generated {new_count} new proposal files.')\n\n# # Quick sanity check\n# sample_pid  = list(all_pids)[0]\n# sample_prop = np.load(str(Config.PROPOSALS / f'{sample_pid}.npy'))\n# print(f'Sample — {sample_pid}: {len(sample_prop)} proposals, shape {sample_prop.shape}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-28T14:59:15.362807Z","iopub.execute_input":"2026-06-28T14:59:15.363071Z","iopub.status.idle":"2026-06-28T14:59:15.382331Z","shell.execute_reply.started":"2026-06-28T14:59:15.363049Z","shell.execute_reply":"2026-06-28T14:59:15.381624Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6 · Detection Dataset & Shared Utilities","metadata":{}},{"cell_type":"code","source":"DET_SCALE = Config.DET_SIZE / 1024.0  # 0.5 — scale boxes from 1024 to 512 space\n\ndet_aug = A.Compose([\n    A.HorizontalFlip(p=0.5),\n    A.RandomBrightnessContrast(brightness_limit=0.15, contrast_limit=0.15, p=0.4),\n    A.ShiftScaleRotate(shift_limit=0.05, scale_limit=0.1, rotate_limit=10,\n                       border_mode=cv2.BORDER_CONSTANT, p=0.5),\n], bbox_params=A.BboxParams(\n    format='pascal_voc',\n    label_fields=['labels'],\n    min_visibility=0.3\n))\n\n\ndef get_boxes(pid, df, scale):\n    \"\"\"GT boxes for patient in xyxy format scaled to model input size.\"\"\"\n    rows = df[(df['patientId'] == pid) & (df['Target'] == 1)]\n    if rows.empty:\n        return (torch.zeros((0, 4), dtype=torch.float32),\n                torch.zeros(0, dtype=torch.long))\n    boxes, lbls = [], []\n    for _, r in rows.iterrows():\n        x1 = r['x'] * scale\n        y1 = r['y'] * scale\n        x2 = (r['x'] + r['width'])  * scale\n        y2 = (r['y'] + r['height']) * scale\n        boxes.append([x1, y1, x2, y2])\n        lbls.append(1)\n    return (torch.tensor(boxes, dtype=torch.float32),\n            torch.tensor(lbls,  dtype=torch.long))\n\n\ndef apply_det_aug(img_rgb, boxes, labels):\n    \"\"\"Apply albumentations detection augmentation.\"\"\"\n    if boxes.numel() > 0:\n        res = det_aug(image=img_rgb,\n                      bboxes=boxes.numpy().tolist(),\n                      labels=labels.numpy().tolist())\n        img_rgb = res['image']\n        if res['bboxes']:\n            boxes  = torch.tensor(res['bboxes'], dtype=torch.float32)\n            labels = torch.tensor(res['labels'], dtype=torch.long)\n        else:\n            boxes  = torch.zeros((0, 4), dtype=torch.float32)\n            labels = torch.zeros(0, dtype=torch.long)\n    else:\n        img_rgb = det_aug(image=img_rgb, bboxes=[], labels=[])['image']\n    return img_rgb, boxes, labels\n\n\nclass RSNADetDataset(Dataset):\n    \"\"\"Used by Faster R-CNN (no proposals needed).\"\"\"\n    def __init__(self, pids, img_dir, df, augment=False, is_test=False):\n        self.pids    = list(pids)\n        self.img_dir = Path(img_dir)\n        self.df      = df\n        self.augment = augment\n        self.is_test = is_test\n\n    def __len__(self):\n        return len(self.pids)\n\n    def __getitem__(self, idx):\n        pid = self.pids[idx]\n        img = cv2.imread(str(self.img_dir / f'{pid}.png'))\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n        if self.is_test:\n            return TF.to_tensor(img), pid\n\n        boxes, labels = get_boxes(pid, self.df, DET_SCALE)\n        if self.augment:\n            img, boxes, labels = apply_det_aug(img, boxes, labels)\n\n        return TF.to_tensor(img), {'boxes': boxes, 'labels': labels}\n\n\ndef det_collate(batch):\n    images, targets = zip(*batch)\n    return list(images), list(targets)\n\n\ndef det_test_collate(batch):\n    images, pids = zip(*batch)\n    return list(images), list(pids)\n\n\ndet_train_ds = RSNADetDataset(train_pids, Config.DET_IMGS / 'train', df, augment=True)\ndet_val_ds   = RSNADetDataset(val_pids,   Config.DET_IMGS / 'train', df, augment=False)\ndet_test_ds  = RSNADetDataset(test_pids,  Config.DET_IMGS / 'test',  df, is_test=True)\n\ndet_train_loader = DataLoader(det_train_ds, Config.DET_BATCH, shuffle=True,\n                               num_workers=2, collate_fn=det_collate,      pin_memory=True)\ndet_val_loader   = DataLoader(det_val_ds,   Config.DET_BATCH, shuffle=False,\n                               num_workers=2, collate_fn=det_collate,      pin_memory=True)\ndet_test_loader  = DataLoader(det_test_ds,  4,                shuffle=False,\n                               num_workers=2, collate_fn=det_test_collate, pin_memory=True)\n\nprint(f'Detection loaders: {len(det_train_loader)} train | {len(det_val_loader)} val batches')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-28T18:42:57.832787Z","iopub.execute_input":"2026-06-28T18:42:57.833299Z","iopub.status.idle":"2026-06-28T18:42:57.852989Z","shell.execute_reply.started":"2026-06-28T18:42:57.833274Z","shell.execute_reply":"2026-06-28T18:42:57.852247Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 7 · Fast R-CNN\n\n**Architecture:** ResNet50-FPN backbone → MultiScaleRoIAlign on Selective Search proposals → FastRCNNPredictor head\n\nThe RPN (Region Proposal Network) is **bypassed** — this is what makes it Fast R-CNN rather than Faster R-CNN.  \nWe subclass `torchvision.models.detection.FasterRCNN` and override `forward` to inject external proposals directly into the RoI heads.","metadata":{}},{"cell_type":"code","source":"class FastRCNNDetDataset(Dataset):\n    \"\"\"Like RSNADetDataset but also loads cached Selective Search proposals.\"\"\"\n    def __init__(self, pids, img_dir, df, proposal_dir, augment=False):\n        self.pids         = list(pids)\n        self.img_dir      = Path(img_dir)\n        self.df           = df\n        self.proposal_dir = Path(proposal_dir)\n        self.augment      = augment\n\n    def __len__(self):\n        return len(self.pids)\n\n    def _load_proposals(self, pid):\n        p = self.proposal_dir / f'{pid}.npy'\n        if p.exists():\n            arr = torch.from_numpy(np.load(str(p))).float()\n            return arr.clamp(0, Config.DET_SIZE - 1)\n        # Fallback: whole image as single proposal\n        S = float(Config.DET_SIZE - 1)\n        return torch.tensor([[0., 0., S, S]])\n\n    def __getitem__(self, idx):\n        pid = self.pids[idx]\n        img = cv2.imread(str(self.img_dir / f'{pid}.png'))\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n        boxes, labels = get_boxes(pid, self.df, DET_SCALE)\n        if self.augment:\n            img, boxes, labels = apply_det_aug(img, boxes, labels)\n\n        proposals = self._load_proposals(pid)\n        return TF.to_tensor(img), proposals, {'boxes': boxes, 'labels': labels}\n\n\nclass FastRCNNTestDataset(Dataset):\n    def __init__(self, pids, img_dir, proposal_dir):\n        self.pids         = list(pids)\n        self.img_dir      = Path(img_dir)\n        self.proposal_dir = Path(proposal_dir)\n\n    def __len__(self):\n        return len(self.pids)\n\n    def __getitem__(self, idx):\n        pid = self.pids[idx]\n        img = cv2.imread(str(self.img_dir / f'{pid}.png'))\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        p   = self.proposal_dir / f'{pid}.npy'\n        proposals = torch.from_numpy(np.load(str(p))).float().clamp(0, Config.DET_SIZE - 1) \\\n                    if p.exists() else torch.tensor([[0., 0., Config.DET_SIZE-1., Config.DET_SIZE-1.]])\n        return TF.to_tensor(img), proposals, pid\n\n\ndef fast_rcnn_collate(batch):\n    images, proposals, targets = zip(*batch)\n    return list(images), list(proposals), list(targets)\n\n\ndef fast_rcnn_test_collate(batch):\n    images, proposals, pids = zip(*batch)\n    return list(images), list(proposals), list(pids)\n\n\nfast_train_ds = FastRCNNDetDataset(train_pids, Config.DET_IMGS / 'train',\n                                    df, Config.PROPOSALS, augment=True)\nfast_val_ds   = FastRCNNDetDataset(val_pids,   Config.DET_IMGS / 'train',\n                                    df, Config.PROPOSALS, augment=False)\nfast_test_ds  = FastRCNNTestDataset(test_pids, Config.DET_IMGS / 'test', Config.PROPOSALS)\n\nfast_train_loader = DataLoader(fast_train_ds, Config.DET_BATCH, shuffle=True,\n                                num_workers=2, collate_fn=fast_rcnn_collate,      pin_memory=True)\nfast_val_loader   = DataLoader(fast_val_ds,   Config.DET_BATCH, shuffle=False,\n                                num_workers=2, collate_fn=fast_rcnn_collate,      pin_memory=True)\nfast_test_loader  = DataLoader(fast_test_ds,  4,                shuffle=False,\n                                num_workers=2, collate_fn=fast_rcnn_test_collate, pin_memory=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-28T14:59:15.405623Z","iopub.execute_input":"2026-06-28T14:59:15.405903Z","iopub.status.idle":"2026-06-28T14:59:15.422244Z","shell.execute_reply.started":"2026-06-28T14:59:15.405882Z","shell.execute_reply":"2026-06-28T14:59:15.421458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class FastRCNNModel(nn.Module):\n    \"\"\"\n    Fast R-CNN built from torchvision components.\n\n    Uses the same ResNet50-FPN backbone and RoI heads as Faster R-CNN but\n    bypasses the built-in RPN entirely — proposals come from Selective Search.\n\n    Architecture:\n        images → ResNet50-FPN → feature pyramid\n               ↑ (SS proposals)\n        proposals → MultiScaleRoIAlign (7×7) → TwoMLPHead → FastRCNNPredictor\n                                                            ↓ class scores + box deltas\n    \"\"\"\n\n    def __init__(self, num_classes=2):\n        super().__init__()\n        # Build a full FasterRCNN to reuse its transform, backbone and roi_heads.\n        # We will override forward() to skip self._inner.rpn.\n        anchor_gen = AnchorGenerator(\n            sizes=((32,), (64,), (128,), (256,), (512,)),\n            aspect_ratios=((0.5, 1.0, 2.0),) * 5,\n        )\n        backbone  = resnet_fpn_backbone('resnet50', weights='DEFAULT')\n        roi_pool  = ops.MultiScaleRoIAlign(\n            featmap_names=['0', '1', '2', '3'], output_size=7, sampling_ratio=2\n        )\n        self._inner = FasterRCNN(\n            backbone,\n            num_classes=num_classes,\n            rpn_anchor_generator=anchor_gen,\n            box_roi_pool=roi_pool,\n            box_detections_per_img=5,\n            box_batch_size_per_image=128,\n            box_positive_fraction=0.25,\n        )\n        # Replace predictor with correct class count\n        in_feats = self._inner.roi_heads.box_predictor.cls_score.in_features\n        self._inner.roi_heads.box_predictor = FastRCNNPredictor(in_feats, num_classes)\n\n    # ── helpers ──────────────────────────────────────────────────────────\n\n    @staticmethod\n    def _scale_proposals(proposals, orig_sizes, new_sizes, device):\n        \"\"\"Rescale xyxy proposals from original image coords to transformed coords.\"\"\"\n        scaled = []\n        for props, orig, new in zip(proposals, orig_sizes, new_sizes):\n            props = props.to(device)\n            if props.numel() == 0:\n                props = torch.tensor([[0., 0., float(orig[1]-1), float(orig[0]-1)]],\n                                      device=device)\n            sh = new[0] / orig[0]\n            sw = new[1] / orig[1]\n            p  = props.clone()\n            p[:, [0, 2]] = (p[:, [0, 2]] * sw).clamp(0, new[1])\n            p[:, [1, 3]] = (p[:, [1, 3]] * sh).clamp(0, new[0])\n            # Drop zero-area proposals\n            keep = (p[:, 2] > p[:, 0]) & (p[:, 3] > p[:, 1])\n            scaled.append(p[keep] if keep.any() else\n                          torch.tensor([[0., 0., float(new[1]-1), float(new[0]-1)]],\n                                        device=device))\n        return scaled\n\n    # ── forward ──────────────────────────────────────────────────────────\n\n    def forward(self, images, proposals, targets=None):\n        \"\"\"\n        Args:\n            images:    List[Tensor[3,H,W]] in [0,1]\n            proposals: List[Tensor[N,4]]   xyxy, pixel coords\n            targets:   List[Dict]          training only\n        \"\"\"\n        orig_sizes = [(img.shape[-2], img.shape[-1]) for img in images]\n\n        # 1. Normalise + pad to common size\n        images_t, targets = self._inner.transform(images, targets)\n\n        # 2. Feature pyramid\n        features = self._inner.backbone(images_t.tensors)\n        if isinstance(features, torch.Tensor):\n            features = OrderedDict([('0', features)])\n\n        # 3. Scale proposals to match transformed image dimensions\n        scaled_props = self._scale_proposals(\n            proposals, orig_sizes, images_t.image_sizes, images_t.tensors.device\n        )\n\n        # 4. RoI heads — classification + regression (skipping RPN entirely)\n        detections, losses = self._inner.roi_heads(\n            features, scaled_props, images_t.image_sizes, targets\n        )\n\n        # 5. Rescale predictions back to original image coords\n        detections = self._inner.transform.postprocess(\n            detections, images_t.image_sizes, orig_sizes\n        )\n\n        return losses if self.training else detections\n\n\nfast_model = FastRCNNModel(num_classes=2).to(device)\nprint('Fast R-CNN model built.')\ntotal = sum(p.numel() for p in fast_model.parameters())\nprint(f'Parameters: {total/1e6:.1f}M')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-28T14:59:15.423208Z","iopub.execute_input":"2026-06-28T14:59:15.423436Z","iopub.status.idle":"2026-06-28T14:59:16.69712Z","shell.execute_reply.started":"2026-06-28T14:59:15.423417Z","shell.execute_reply":"2026-06-28T14:59:16.696295Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"@torch.no_grad()\ndef val_loss(model, loader, is_fast_rcnn):\n    \"\"\"Compute mean val loss (model stays in train mode to get loss dict).\"\"\"\n    model.train()\n    total, n = 0.0, 0\n    for batch in loader:\n        if is_fast_rcnn:\n            imgs, props, tgts = batch\n            props = [p.to(device) for p in props]\n        else:\n            imgs, tgts = batch\n        imgs = [x.to(device) for x in imgs]\n        tgts = [{k: v.to(device) for k, v in t.items()} for t in tgts]\n        losses = model(imgs, props, tgts) if is_fast_rcnn else model(imgs, tgts)\n        total += sum(losses.values()).item()\n        n += 1\n    model.eval()\n    return total / max(n, 1)\n\n\n# ── Training ─────────────────────────────────────────────────────────────────\nprint('=' * 60)\nprint('Training Fast R-CNN (Selective Search proposals)')\nprint('=' * 60)\n\nfast_opt   = optim.SGD(fast_model.parameters(), lr=Config.DET_LR,\n                        momentum=0.9, weight_decay=5e-4)\nfast_sched = optim.lr_scheduler.StepLR(fast_opt, step_size=Config.DET_LR_STEP, gamma=0.1)\n\nbest_fast_loss = float('inf')\nfast_history   = []\n\nfor epoch in range(Config.DET_EPOCHS):\n    fast_model.train()\n    train_loss = 0.0\n\n    for imgs, props, tgts in tqdm(fast_train_loader,\n                                   desc=f'Fast-RCNN {epoch+1}/{Config.DET_EPOCHS}'):\n        imgs  = [x.to(device) for x in imgs]\n        props = [p.to(device) for p in props]\n        tgts  = [{k: v.to(device) for k, v in t.items()} for t in tgts]\n\n        loss_dict = fast_model(imgs, props, tgts)\n        loss      = sum(loss_dict.values())\n\n        fast_opt.zero_grad()\n        loss.backward()\n        nn.utils.clip_grad_norm_(fast_model.parameters(), 1.0)\n        fast_opt.step()\n        train_loss += loss.item()\n\n    fast_sched.step()\n    avg_tr  = train_loss / len(fast_train_loader)\n    avg_val = val_loss(fast_model, fast_val_loader, is_fast_rcnn=True)\n    fast_history.append({'epoch': epoch + 1, 'train': avg_tr, 'val': avg_val})\n    print(f'  Epoch {epoch+1:3d} | train {avg_tr:.4f} | val {avg_val:.4f}')\n\n    if avg_val < best_fast_loss:\n        best_fast_loss = avg_val\n        torch.save(fast_model.state_dict(), Config.MODELS / 'fast_rcnn_best.pth')\n\nprint(f'Best Fast R-CNN val loss: {best_fast_loss:.4f}')\nfast_model.load_state_dict(\n    torch.load(Config.MODELS / 'fast_rcnn_best.pth', map_location=device))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-28T18:42:57.854016Z","iopub.execute_input":"2026-06-28T18:42:57.854285Z","iopub.status.idle":"2026-06-28T18:42:57.942329Z","shell.execute_reply.started":"2026-06-28T18:42:57.854263Z","shell.execute_reply":"2026-06-28T18:42:57.941213Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def infer_fast_rcnn(model, loader, conf_thresh=0.05):\n    \"\"\"Inference on FastRCNNTestDataset. Returns {pid: {boxes, scores}}.\"\"\"\n    model.eval()\n    results = {}\n    with torch.no_grad():\n        for imgs, props, pids in tqdm(loader, desc='Fast R-CNN inference'):\n            imgs  = [x.to(device) for x in imgs]\n            props = [p.to(device) for p in props]\n            preds = model(imgs, props)\n            for pred, pid in zip(preds, pids):\n                keep = pred['scores'] >= conf_thresh\n                results[pid] = {\n                    'boxes':  pred['boxes'][keep].cpu().numpy().tolist(),\n                    'scores': pred['scores'][keep].cpu().numpy().tolist(),\n                }\n    return results\n\n\nfast_preds = infer_fast_rcnn(fast_model, fast_test_loader)\nn_pos = sum(1 for v in fast_preds.values() if v['boxes'])\nprint(f'Fast R-CNN: {n_pos}/{len(fast_preds)} test images with detections')\n\ngc.collect()\ntorch.cuda.empty_cache()","metadata":{"trusted":true,"execution":{"execution_failed":"2026-06-28T15:54:55.833Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 8 · Faster R-CNN\n\n**Architecture:** ResNet50-FPN → built-in RPN → MultiScaleRoIAlign → FastRCNNPredictor\n\nSame backbone as Fast R-CNN for fair comparison. The only structural difference is the **built-in Region Proposal Network (RPN)** that learns to generate proposals end-to-end rather than relying on Selective Search.","metadata":{}},{"cell_type":"code","source":"def build_faster_rcnn(num_classes=2):\n    model = torchvision.models.detection.fasterrcnn_resnet50_fpn(\n        weights='DEFAULT',\n        box_detections_per_img=5,\n        box_batch_size_per_image=128,\n        box_positive_fraction=0.25,\n        rpn_batch_size_per_image=256,\n        rpn_nms_thresh=0.7,\n        box_nms_thresh=0.5,\n    )\n    in_feats = model.roi_heads.box_predictor.cls_score.in_features\n    model.roi_heads.box_predictor = FastRCNNPredictor(in_feats, num_classes)\n    return model\n\n\nfaster_model = build_faster_rcnn(num_classes=2).to(device)\nprint('Faster R-CNN model built.')\ntotal = sum(p.numel() for p in faster_model.parameters())\nprint(f'Parameters: {total/1e6:.1f}M')","metadata":{"trusted":true,"execution":{"execution_failed":"2026-06-28T15:54:55.833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('=' * 60)\nprint('Training Faster R-CNN (built-in RPN)')\nprint('=' * 60)\n\nparams       = [p for p in faster_model.parameters() if p.requires_grad]\nfaster_opt   = optim.SGD(params, lr=Config.DET_LR, momentum=0.9, weight_decay=5e-4)\nfaster_sched = optim.lr_scheduler.StepLR(faster_opt, step_size=Config.DET_LR_STEP, gamma=0.1)\n\nbest_faster_loss = float('inf')\nfaster_history   = []\n\nfor epoch in range(Config.DET_EPOCHS):\n    faster_model.train()\n    train_loss = 0.0\n\n    for imgs, tgts in tqdm(det_train_loader,\n                            desc=f'Faster-RCNN {epoch+1}/{Config.DET_EPOCHS}'):\n        imgs = [x.to(device) for x in imgs]\n        tgts = [{k: v.to(device) for k, v in t.items()} for t in tgts]\n\n        # loss_dict keys: loss_classifier, loss_box_reg, loss_objectness, loss_rpn_box_reg\n        loss_dict = faster_model(imgs, tgts)\n        loss      = sum(loss_dict.values())\n\n        faster_opt.zero_grad()\n        loss.backward()\n        nn.utils.clip_grad_norm_(params, 1.0)\n        faster_opt.step()\n        train_loss += loss.item()\n\n    faster_sched.step()\n    avg_tr  = train_loss / len(det_train_loader)\n    avg_val = val_loss(faster_model, det_val_loader, is_fast_rcnn=False)\n    faster_history.append({'epoch': epoch + 1, 'train': avg_tr, 'val': avg_val})\n    print(f'  Epoch {epoch+1:3d} | train {avg_tr:.4f} | val {avg_val:.4f}')\n\n    if avg_val < best_faster_loss:\n        best_faster_loss = avg_val\n        torch.save(faster_model.state_dict(), Config.MODELS / 'faster_rcnn_best.pth')\n\nprint(f'Best Faster R-CNN val loss: {best_faster_loss:.4f}')\nfaster_model.load_state_dict(\n    torch.load(Config.MODELS / 'faster_rcnn_best.pth', map_location=device))","metadata":{"trusted":true,"execution":{"execution_failed":"2026-06-28T15:54:55.833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def infer_faster_rcnn(model, loader, conf_thresh=0.05):\n    \"\"\"Inference on RSNADetDataset (test). Returns {pid: {boxes, scores}}.\"\"\"\n    model.eval()\n    results = {}\n    with torch.no_grad():\n        for imgs, pids in tqdm(loader, desc='Faster R-CNN inference'):\n            imgs  = [x.to(device) for x in imgs]\n            preds = model(imgs)\n            for pred, pid in zip(preds, pids):\n                keep = pred['scores'] >= conf_thresh\n                results[pid] = {\n                    'boxes':  pred['boxes'][keep].cpu().numpy().tolist(),\n                    'scores': pred['scores'][keep].cpu().numpy().tolist(),\n                }\n    return results\n\n\nfaster_preds = infer_faster_rcnn(faster_model, det_test_loader)\nn_pos = sum(1 for v in faster_preds.values() if v['boxes'])\nprint(f'Faster R-CNN: {n_pos}/{len(faster_preds)} test images with detections')\n\ngc.collect()\ntorch.cuda.empty_cache()","metadata":{"trusted":true,"execution":{"execution_failed":"2026-06-28T15:54:55.834Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 9 · Ensemble & Pipeline Fusion\n\n**Steps:**\n1. Normalise boxes to [0, 1] for WBF\n2. Weight each box score by the classifier probability `p_cls`\n3. Weighted Box Fusion merges Fast R-CNN + Faster R-CNN predictions\n4. Filter combined score < `CONF_THRESH`\n5. Scale back to 512-px space then apply **87.5% box resize** (1st-place trick)\n6. Scale to 1024×1024 original space for submission","metadata":{}},{"cell_type":"code","source":"def resize_boxes(boxes_xyxy, scale=0.875):\n    \"\"\"\n    Shrink boxes around their centre by `scale`.\n\n    Why: The RSNA test set was triple-read; annotators' boxes were intersected\n    (not unioned), producing systematically smaller ground-truth boxes than the\n    single-read training labels. Shrinking by 0.875 mimics that effect.\n    1st-place improvement: +0.038 public LB (0.222 → 0.260).\n    \"\"\"\n    if len(boxes_xyxy) == 0:\n        return boxes_xyxy\n    b  = np.array(boxes_xyxy, dtype=np.float32)\n    cx = (b[:, 0] + b[:, 2]) / 2\n    cy = (b[:, 1] + b[:, 3]) / 2\n    w  = (b[:, 2] - b[:, 0]) * scale\n    h  = (b[:, 3] - b[:, 1]) * scale\n    return np.stack([cx - w/2, cy - h/2, cx + w/2, cy + h/2], axis=1).clip(0)\n\n\ndef ensemble_and_fuse(fast_pred, faster_pred, cls_scores, all_test_pids, cfg):\n    S   = float(cfg.DET_SIZE)\n    rows = []\n\n    for pid in tqdm(all_test_pids, desc='Ensembling'):\n        p_cls    = cls_scores.get(pid, 0.5)\n        fast     = fast_pred.get(pid,   {'boxes': [], 'scores': []})\n        faster   = faster_pred.get(pid, {'boxes': [], 'scores': []})\n\n        boxes_list, scores_list, labels_list = [], [], []\n\n        for pred in (fast, faster):\n            if not pred['boxes']:\n                continue\n            # Normalise xyxy to [0,1]\n            bnorm = [[max(0., b[0]/S), max(0., b[1]/S),\n                      min(1., b[2]/S), min(1., b[3]/S)]\n                     for b in pred['boxes']]\n            # Weight by classifier score\n            sw = [s * p_cls for s in pred['scores']]\n            boxes_list.append(bnorm)\n            scores_list.append(sw)\n            labels_list.append([1] * len(bnorm))\n\n        if not boxes_list:\n            rows.append({'patientId': pid, 'PredictionString': ''})\n            continue\n\n        fb, fs, _ = weighted_boxes_fusion(\n            boxes_list, scores_list, labels_list,\n            iou_thr=cfg.WBF_IOU, weights=[1.0, 1.0], skip_box_thr=0.001\n        )\n\n        keep = fs >= cfg.CONF_THRESH\n        fb, fs = fb[keep], fs[keep]\n\n        if len(fb) == 0:\n            rows.append({'patientId': pid, 'PredictionString': ''})\n            continue\n\n        # Back to pixel space at DET_SIZE\n        fb_px = fb * S\n\n        # 87.5% box resize (1st-place trick)\n        fb_px = resize_boxes(fb_px, cfg.BOX_SCALE)\n\n        # Scale to original 1024×1024 space\n        orig_scale = 1024.0 / S\n        fb_px = fb_px * orig_scale\n\n        # Format: \"conf x y w h\" (xywh) per detection\n        parts = []\n        for score, box in zip(fs, fb_px):\n            x1, y1, x2, y2 = box\n            w = max(1., x2 - x1)\n            h = max(1., y2 - y1)\n            parts.append(f'{score:.4f} {x1:.0f} {y1:.0f} {w:.0f} {h:.0f}')\n\n        rows.append({'patientId': pid, 'PredictionString': ' '.join(parts)})\n\n    return pd.DataFrame(rows)\n\n\nsub_df = ensemble_and_fuse(fast_preds, faster_preds, cls_scores, test_pids, Config)\n\nn_positive = (sub_df['PredictionString'] != '').sum()\nprevalence = n_positive / len(sub_df)\nprint(f'Positive predictions: {n_positive}/{len(sub_df)} ({prevalence:.1%})')\nprint(f'Expected prevalence: ~35-37% (1st-place reference)')","metadata":{"trusted":true,"execution":{"execution_failed":"2026-06-28T15:54:55.834Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 10 · Submission & Training Curves","metadata":{}},{"cell_type":"code","source":"# ── Save submission ───────────────────────────────────────────────────────────\nsub_path = Config.WORK_DIR / 'submission.csv'\nsub_df.to_csv(sub_path, index=False)\nprint(f'Saved → {sub_path}')\n\n# Sanity checks\nassert list(sub_df.columns) == ['patientId', 'PredictionString']\nassert len(sub_df) == len(test_pids), \\\n    f'Row count mismatch: {len(sub_df)} vs {len(test_pids)}'\n\npos_example = sub_df[sub_df['PredictionString'] != ''].iloc[0]\nprint(f'\\nSample positive row:')\nprint(f'  patientId       : {pos_example[\"patientId\"]}')\nprint(f'  PredictionString: {pos_example[\"PredictionString\"][:120]}')\n\n# ── Training curves ───────────────────────────────────────────────────────────\nfig, axes = plt.subplots(1, 3, figsize=(17, 4))\n\n# Classification AUC\nax = axes[0]\nax.plot([h['epoch'] for h in cls_history],\n        [h['auc']   for h in cls_history], marker='o', color='steelblue')\nax.axhline(0.80, ls='--', color='gray', label='AUC 0.80 target')\nax.set_title('Classification — Val AUC'); ax.set_xlabel('Epoch')\nax.set_ylabel('AUC'); ax.legend(); ax.grid(True, alpha=0.3)\n\n# Detection losses\nfor ax, history, label in [\n    (axes[1], fast_history,   'Fast R-CNN'),\n    (axes[2], faster_history, 'Faster R-CNN'),\n]:\n    ep  = [h['epoch'] for h in history]\n    ax.plot(ep, [h['train'] for h in history], label='train')\n    ax.plot(ep, [h['val']   for h in history], label='val',   ls='--')\n    ax.set_title(f'{label} — Loss'); ax.set_xlabel('Epoch')\n    ax.set_ylabel('Loss'); ax.legend(); ax.grid(True, alpha=0.3)\n\nplt.tight_layout()\nplt.savefig(Config.WORK_DIR / 'training_curves.png', dpi=100, bbox_inches='tight')\nplt.show()\n\nprint('\\nDone. Submit submission.csv to Kaggle.')","metadata":{"trusted":true,"execution":{"execution_failed":"2026-06-28T15:54:55.834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}