{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":29653,"databundleVersionId":2420395,"isSourceIdPinned":false},{"sourceType":"datasetVersion","sourceId":14943783,"datasetId":9563220,"databundleVersionId":15813126}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install numpy pydicom scipy tqdm opencv-python","metadata":{"_uuid":"f0070b64-ae23-4394-bea4-b0c82e174f05","_cell_guid":"b377905d-54f4-4efe-8027-f4064345a420","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2026-03-05T18:30:14.466836Z","iopub.execute_input":"2026-03-05T18:30:14.467492Z","iopub.status.idle":"2026-03-05T18:30:17.890348Z","shell.execute_reply.started":"2026-03-05T18:30:14.46746Z","shell.execute_reply":"2026-03-05T18:30:17.889518Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"![IMAGE PREPROCESSING](https://www.c-sharpcorner.com/article/data-preprocessing-in-machine-learning/Images/Model%20Training3.jpg)","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pydicom\nfrom pathlib import Path\nfrom tqdm import tqdm\nfrom scipy.ndimage import zoom\nfrom concurrent.futures import ProcessPoolExecutor\nimport multiprocessing\nimport csv\nimport json\n\nDATA = Path(r\"/kaggle/input/competitions/rsna-miccai-brain-tumor-radiogenomic-classification\")\nTRAIN = DATA / \"train\"\n\nOUT = Path(r\"/kaggle/working/Preprocessed\")\nOUT.mkdir(exist_ok=True)\n\nEXCLUDE = [\"00109\", \"00123\", \"00709\"]\n\nMODS = [\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"]\n\nIMG_SIZE = 128\nDEPTH = 64\n\nINTERP_ORDER = 1\n\n\n# -----------------------------\n# Load DICOM volume\n# -----------------------------\ndef load_dicom_volume(mod_path):\n\n    files = list(mod_path.glob(\"*.dcm\"))\n\n    if len(files) == 0:\n        return None\n\n    nodes = []\n\n    for f in files:\n\n        d = pydicom.dcmread(f, stop_before_pixels=True)\n\n        try:\n            z = float(d.ImagePositionPatient[2])\n        except:\n            z = getattr(d, \"InstanceNumber\", 0)\n\n        nodes.append((f, z))\n\n    nodes.sort(key=lambda x: x[1])\n\n    volume = []\n\n    for f, _ in nodes:\n\n        d = pydicom.dcmread(f)\n\n        img = d.pixel_array.astype(np.float32)\n\n        img = img * getattr(d, \"RescaleSlope\", 1) + getattr(d, \"RescaleIntercept\", 0)\n\n        volume.append(img)\n\n    if len(volume) == 0:\n        return None\n\n    return np.stack(volume)\n\n\n# -----------------------------\n# Crop brain region\n# -----------------------------\ndef get_crop_box(volume):\n\n    try:\n\n        mask = volume > np.percentile(volume, 5)\n\n        coords = np.where(mask)\n\n        if len(coords[0]) == 0:\n            return volume\n\n        zmin, zmax = coords[0].min(), coords[0].max()\n        ymin, ymax = coords[1].min(), coords[1].max()\n        xmin, xmax = coords[2].min(), coords[2].max()\n\n        pad = 3\n\n        zmin = max(zmin - pad, 0)\n        ymin = max(ymin - pad, 0)\n        xmin = max(xmin - pad, 0)\n\n        zmax = min(zmax + pad, volume.shape[0])\n        ymax = min(ymax + pad, volume.shape[1])\n        xmax = min(xmax + pad, volume.shape[2])\n\n        cropped = volume[zmin:zmax, ymin:ymax, xmin:xmax]\n\n        if cropped.shape[0] < 10:\n            return volume\n\n        return cropped\n\n    except:\n        return volume\n\n\n# -----------------------------\n# 3D resampling\n# -----------------------------\ndef resample_volume(volume):\n\n    z, h, w = volume.shape\n\n    if z < 2 or h < 2 or w < 2:\n        return np.zeros((DEPTH, IMG_SIZE, IMG_SIZE), dtype=np.float32)\n\n    z_factor = DEPTH / z\n    h_factor = IMG_SIZE / h\n    w_factor = IMG_SIZE / w\n\n    volume = zoom(volume, (z_factor, h_factor, w_factor), order=INTERP_ORDER)\n\n    return volume\n\n\n# -----------------------------\n# Safe normalization\n# -----------------------------\ndef normalize(volume):\n\n    if np.all(volume == 0):\n        return volume\n\n    low = np.percentile(volume, 0.5)\n    high = np.percentile(volume, 99.5)\n\n    volume = np.clip(volume, low, high)\n\n    std = volume.std()\n\n    if std < 1e-6:\n        return np.zeros_like(volume)\n\n    volume = (volume - volume.mean()) / std\n\n    return volume\n\n\n# -----------------------------\n# Process patient\n# -----------------------------\ndef process_patient(patient_path):\n\n    patient_id = patient_path.name\n\n    if patient_id in EXCLUDE:\n        return []\n\n    patient_volumes = []\n\n    local_missing = []\n\n    try:\n\n        for mod in MODS:\n\n            mod_path = patient_path / mod\n\n            volume = load_dicom_volume(mod_path)\n\n            if volume is None:\n\n                local_missing.append((patient_id, mod))\n\n                volume = np.zeros((DEPTH, IMG_SIZE, IMG_SIZE), dtype=np.float32)\n\n                patient_volumes.append(volume)\n\n                continue\n\n            volume = get_crop_box(volume)\n\n            volume = resample_volume(volume)\n\n            volume = normalize(volume)\n\n            patient_volumes.append(volume)\n\n        tensor = np.stack(patient_volumes).astype(np.float16)\n\n        np.save(OUT / f\"{patient_id}.npy\", tensor)\n\n    except Exception as e:\n\n        print(f\"Skipping {patient_id}: {e}\")\n\n    return local_missing\n\n\n# -----------------------------\n# Main multiprocessing runner\n# -----------------------------\ndef main():\n    # 1. Generate Metadata for Kaggle Upload\n    # This prevents the \"Title must be between 6 and 50 chars\" error\n    metadata = {\n        \"title\": \"brain-tumor-preprocessed-v1\", # Title: 6-50 chars\n        \"id\": \"your-username/brain-tumor-preprocessed-v1\", # Replace with your username\n        \"licenses\": [{\"name\": \"CC0-1.0\"}]\n    }\n    \n    with open(OUT / \"dataset-metadata.json\", \"w\") as f:\n        json.dump(metadata, f)\n\n    # 2. Run Processing\n    patients = list(TRAIN.glob(\"*\"))\n    workers = max(1, multiprocessing.cpu_count() - 2)\n\n    print(f\"Starting preprocessing with {workers} workers...\")\n    with ProcessPoolExecutor(workers) as executor:\n        # This captures the return values from process_patient\n        results = list(tqdm(executor.map(process_patient, patients), total=len(patients)))\n\n    # 3. Handle Missing Modalities\n    # Flatten the list of lists returned by the workers\n    flat_missing = [item for sublist in results if sublist for item in sublist]\n\n    if len(flat_missing) > 0:\n        csv_path = OUT / \"missing_modalities.csv\"\n        with open(csv_path, \"w\", newline=\"\") as f:\n            writer = csv.writer(f)\n            writer.writerow([\"PatientID\", \"Modality\"])\n            writer.writerows(flat_missing)\n        print(f\"\\n✅ Created: {csv_path}\")\n    else:\n        print(\"\\n✨ All modalities found for all patients.\")\n\nif __name__ == \"__main__\":\n    main()","metadata":{"_uuid":"16fc4a0a-a95a-449c-a520-f1ea36b5319a","_cell_guid":"d990e378-7937-4f11-b11b-527d6e48338a","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"![Image](https://i.ibb.co/20ttQ5wB/Chat-GPT-Image-Mar-6-2026-12-12-04-AM.png)","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\n\n# Update this to your actual output directory\nOUT_DIR = Path(\"/kaggle/input/datasets/raj072004/preprocessed-final\")\n\ndef check_processed_sample(patient_id):\n    # Construct the full path\n    file_path = OUT_DIR / f\"{patient_id}.npy\"\n    \n    if not file_path.exists():\n        print(f\"Error: File {file_path} not found. Check if preprocessing finished.\")\n        return\n\n    # Load the data (4, 64, 128, 128)\n    # We use mmap_mode='r' to be memory efficient\n    data = np.load(file_path, mmap_mode='r') \n    \n    fig, axes = plt.subplots(1, 4, figsize=(20, 5))\n    titles = [\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"]\n    \n    # We'll look at the middle slice of the depth (64 // 2 = 32)\n    slice_idx = data.shape[1] // 2 \n    \n    for i in range(4):\n        # Indexing: [Modality, Depth, Height, Width]\n        img_slice = data[i, slice_idx, :, :]\n        \n        # Convert from float16 to float32 for plotting compatibility\n        axes[i].imshow(img_slice.astype(np.float32), cmap=\"bone\")\n        axes[i].set_title(f\"{titles[i]} (Slice {slice_idx})\")\n        axes[i].axis(\"off\")\n        \n    plt.tight_layout()\n    plt.show()\n\n# Example Usage:\n# Replace \"00002\" with an actual ID from your output folder\ncheck_processed_sample(\"00006\")","metadata":{"_uuid":"c806d828-4c5c-4ea8-93dc-5c77290260c4","_cell_guid":"222fcf7e-02f7-42b6-b4fd-ddd5b130d491","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2026-03-05T18:31:26.31875Z","iopub.execute_input":"2026-03-05T18:31:26.319073Z","iopub.status.idle":"2026-03-05T18:31:26.681191Z","shell.execute_reply.started":"2026-03-05T18:31:26.319049Z","shell.execute_reply":"2026-03-05T18:31:26.680316Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"  Initializing Database and Uploading it to Kaggle ","metadata":{}},{"cell_type":"code","source":"!kaggle datasets init -p /kaggle/working/Preprocessed","metadata":{"_uuid":"dc21f12b-5871-47d2-8aca-6075bbeda222","_cell_guid":"9ae2136a-d994-4845-8dc0-673b9f844ec1","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2026-02-24T08:36:31.34759Z","iopub.execute_input":"2026-02-24T08:36:31.348289Z","iopub.status.idle":"2026-02-24T08:36:32.52595Z","shell.execute_reply.started":"2026-02-24T08:36:31.348244Z","shell.execute_reply":"2026-02-24T08:36:32.525238Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\n\npath = \"/kaggle/working/Preprocessed/dataset-metadata.json\"\n\nwith open(path) as f:\n    data = json.load(f)\n\ndata[\"title\"] = \"Preprocessed Dataset Ready for training\"\ndata[\"id\"] = \"raj072004/preprocessed-final\"\n\nwith open(path, \"w\") as f:\n    json.dump(data, f)\n\nprint(\"Metadata updated\")","metadata":{"_uuid":"623c97dc-50bb-40d6-804a-849d2dd40374","_cell_guid":"7097ab7b-80ed-4f86-82c0-81be27b44b13","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2026-02-24T08:37:32.711931Z","iopub.execute_input":"2026-02-24T08:37:32.712612Z","iopub.status.idle":"2026-02-24T08:37:32.718589Z","shell.execute_reply.started":"2026-02-24T08:37:32.712579Z","shell.execute_reply":"2026-02-24T08:37:32.717871Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!kaggle datasets create -p /kaggle/working/Preprocessed","metadata":{"_uuid":"4bdfafeb-b8c2-488c-8d03-91950433b630","_cell_guid":"91e9d17f-ad4f-43ce-ba1e-c5499bb42266","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2026-02-24T08:39:27.156159Z","iopub.execute_input":"2026-02-24T08:39:27.156916Z","iopub.status.idle":"2026-02-24T08:45:35.317442Z","shell.execute_reply.started":"2026-02-24T08:39:27.156876Z","shell.execute_reply":"2026-02-24T08:45:35.316611Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==============================\n# RSNA MGMT TRAINING SCRIPT\n# ==============================\n\nimport os\nimport pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom torch.utils.data import Dataset, DataLoader, WeightedRandomSampler\nfrom torch.amp import autocast, GradScaler\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score\n\nfrom pathlib import Path\nimport random\nfrom tqdm import tqdm\n\n\n# -----------------------------\n# Configuration\n# -----------------------------\nclass CFG:\n\n    seed = 42\n\n    img_size = 128\n    depth = 64\n\n    batch_size = 16\n    epochs = 100\n\n    lr = 2e-4\n    num_workers = 2\n\n    npy_path = Path(\"/kaggle/input/datasets/raj072004/preprocessed-final\")\n    labels_path = Path(\"/kaggle/input/datasets/raj072004/preprocessed-final/train_labels.csv\")\n\n\n# -----------------------------\n# Seed Everything\n# -----------------------------\ndef seed_everything(seed):\n\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n\n    np.random.seed(seed)\n\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n\n    torch.backends.cudnn.deterministic = False\n    torch.backends.cudnn.benchmark = True\n\n\nseed_everything(CFG.seed)\n\n\n# -----------------------------\n# Dataset\n# -----------------------------\nclass BrainTumorDataset(Dataset):\n\n    def __init__(self, df, npy_path, transform=None):\n\n        self.df = df\n        self.npy_path = npy_path\n        self.transform = transform\n\n\n    def __len__(self):\n        return len(self.df)\n\n\n    def __getitem__(self, idx):\n\n        row = self.df.iloc[idx]\n\n        patient_id = str(int(row['BraTS21ID'])).zfill(5)\n\n        file_path = self.npy_path / f\"{patient_id}.npy\"\n\n        raw = np.load(file_path, mmap_mode=\"r\")\n\n        data = np.asarray(raw, dtype=np.float32)\n\n        if self.transform:\n            data = self.transform(data)\n\n        label = torch.tensor(row[\"MGMT_value\"], dtype=torch.float32)\n\n        return torch.from_numpy(data), label\n\n\n# -----------------------------\n# MRI Augmentation\n# -----------------------------\ndef train_transform(data):\n\n    if random.random() > 0.5:\n        data = np.flip(data, axis=1).copy()\n\n    if random.random() > 0.5:\n        data = np.flip(data, axis=2).copy()\n\n    if random.random() > 0.5:\n        data = np.flip(data, axis=3).copy()\n\n    if random.random() > 0.5:\n        noise = np.random.normal(0, 0.02, data.shape)\n        data += noise\n\n    if random.random() > 0.5:\n\n        gamma = random.uniform(0.7, 1.3)\n\n        data = np.clip(data, 0, None)\n\n        data = data ** gamma\n\n    if random.random() > 0.5:\n        data += random.uniform(-0.05, 0.05)\n\n    return data\n\n\n# -----------------------------\n# ResNet 3D Block\n# -----------------------------\nclass ResNet3dBlock(nn.Module):\n\n    def __init__(self, in_channels, out_channels, stride=1):\n\n        super().__init__()\n\n        self.conv1 = nn.Conv3d(\n            in_channels,\n            out_channels,\n            kernel_size=3,\n            stride=stride,\n            padding=1,\n            bias=False\n        )\n\n        self.bn1 = nn.BatchNorm3d(out_channels)\n\n        self.relu = nn.ReLU(inplace=True)\n\n        self.conv2 = nn.Conv3d(\n            out_channels,\n            out_channels,\n            kernel_size=3,\n            padding=1,\n            bias=False\n        )\n\n        self.bn2 = nn.BatchNorm3d(out_channels)\n\n        self.shortcut = nn.Sequential()\n\n        if stride != 1 or in_channels != out_channels:\n\n            self.shortcut = nn.Sequential(\n\n                nn.Conv3d(\n                    in_channels,\n                    out_channels,\n                    kernel_size=1,\n                    stride=stride,\n                    bias=False\n                ),\n\n                nn.BatchNorm3d(out_channels)\n            )\n\n\n    def forward(self, x):\n\n        out = self.relu(self.bn1(self.conv1(x)))\n\n        out = self.bn2(self.conv2(out))\n\n        out += self.shortcut(x)\n\n        return self.relu(out)\n\n\n# -----------------------------\n# Model\n# -----------------------------\nclass BrainNet3D(nn.Module):\n\n    def __init__(self):\n\n        super().__init__()\n\n        self.stem = nn.Sequential(\n\n            nn.Conv3d(4, 32, kernel_size=7, stride=(1,2,2), padding=3, bias=False),\n\n            nn.BatchNorm3d(32),\n\n            nn.ReLU(inplace=True),\n\n            nn.MaxPool3d(kernel_size=3, stride=2, padding=1)\n\n        )\n\n        self.layer1 = ResNet3dBlock(32, 64, stride=2)\n\n        self.layer2 = ResNet3dBlock(64, 128, stride=2)\n\n        self.layer3 = ResNet3dBlock(128, 256, stride=2)\n\n        self.pool = nn.AdaptiveAvgPool3d((1,1,1))\n\n        self.fc = nn.Linear(256, 1)\n\n\n    def forward(self, x):\n\n        x = self.stem(x)\n\n        x = self.layer1(x)\n\n        x = self.layer2(x)\n\n        x = self.layer3(x)\n\n        x = self.pool(x)\n\n        x = torch.flatten(x,1)\n\n        return self.fc(x)\n\n\n# -----------------------------\n# Training\n# -----------------------------\ndef train_model():\n\n    print(\"=\"*40)\n\n    print(\"GPU:\", torch.cuda.get_device_name(0))\n\n    print(\"Total GPUs:\", torch.cuda.device_count())\n\n    print(\"=\"*40)\n\n\n    df = pd.read_csv(CFG.labels_path)\n\n    df[\"exists\"] = df[\"BraTS21ID\"].apply(\n        lambda x: (CFG.npy_path / f\"{str(x).zfill(5)}.npy\").exists()\n    )\n\n    df = df[df[\"exists\"]]\n\n\n    train_df, valid_df = train_test_split(\n\n        df,\n\n        test_size=0.15,\n\n        stratify=df[\"MGMT_value\"],\n\n        random_state=CFG.seed\n\n    )\n\n\n    # -----------------------------\n    # Class balancing\n    # -----------------------------\n    class_counts = train_df[\"MGMT_value\"].value_counts().to_dict()\n\n    weights = train_df[\"MGMT_value\"].map({\n\n        0: 1.0 / class_counts[0],\n\n        1: 1.0 / class_counts[1]\n\n    }).values\n\n\n    sampler = WeightedRandomSampler(\n\n        weights,\n\n        num_samples=len(weights),\n\n        replacement=True\n\n    )\n\n\n    train_loader = DataLoader(\n\n        BrainTumorDataset(train_df, CFG.npy_path, train_transform),\n\n        batch_size=CFG.batch_size,\n\n        sampler=sampler,\n\n        num_workers=CFG.num_workers,\n\n        pin_memory=True,\n\n        persistent_workers=True,\n\n        prefetch_factor=2\n\n    )\n\n\n    valid_loader = DataLoader(\n\n        BrainTumorDataset(valid_df, CFG.npy_path),\n\n        batch_size=CFG.batch_size,\n\n        shuffle=False,\n\n        num_workers=CFG.num_workers,\n\n        pin_memory=True\n\n    )\n\n\n    model = BrainNet3D()\n\n\n    if torch.cuda.device_count() > 1:\n        model = nn.DataParallel(model)\n\n\n    model = model.cuda()\n\n\n    # -----------------------------\n    # Loss\n    # -----------------------------\n    pos_weight = torch.tensor(\n\n        [class_counts[0] / class_counts[1]],\n\n        dtype=torch.float32\n\n    ).cuda()\n\n\n    criterion = nn.BCEWithLogitsLoss(pos_weight=pos_weight)\n\n\n    optimizer = optim.AdamW(\n\n        model.parameters(),\n\n        lr=CFG.lr,\n\n        weight_decay=1e-5\n\n    )\n\n\n    scheduler = optim.lr_scheduler.CosineAnnealingWarmRestarts(\n\n        optimizer,\n\n        T_0=10,\n\n        T_mult=2\n\n    )\n\n\n    scaler = GradScaler(\"cuda\")\n\n\n    best_auc = 0\n\n\n    for epoch in range(CFG.epochs):\n\n        print(f\"\\nEpoch {epoch+1}/{CFG.epochs}\")\n\n\n        # -----------------------------\n        # TRAIN\n        # -----------------------------\n        model.train()\n\n        losses = []\n\n\n        loop = tqdm(train_loader)\n\n\n        for imgs, labels in loop:\n\n            imgs = imgs.cuda(non_blocking=True)\n\n            labels = labels.cuda(non_blocking=True).unsqueeze(1)\n\n\n            optimizer.zero_grad(set_to_none=True)\n\n\n            with autocast(\"cuda\"):\n\n                outputs = model(imgs)\n\n                loss = criterion(outputs, labels)\n\n\n            scaler.scale(loss).backward()\n\n            scaler.unscale_(optimizer)\n\n            torch.nn.utils.clip_grad_norm_(model.parameters(),1.0)\n\n            scaler.step(optimizer)\n\n            scaler.update()\n\n\n            losses.append(loss.item())\n\n            loop.set_postfix(loss=np.mean(losses))\n\n\n        # -----------------------------\n        # VALIDATION\n        # -----------------------------\n        model.eval()\n\n        preds = []\n\n        targets = []\n\n\n        with torch.no_grad():\n\n            for imgs, labels in valid_loader:\n\n                imgs = imgs.cuda()\n\n\n                with autocast(\"cuda\"):\n\n                    outputs = model(imgs)\n\n\n                preds.extend(torch.sigmoid(outputs).cpu().numpy().ravel())\n\n                targets.extend(labels.numpy())\n\n\n        auc = roc_auc_score(targets, preds)\n\n        scheduler.step()\n\n\n        print(f\"Validation AUC: {auc:.4f}\")\n\n\n        if auc > best_auc:\n\n            best_auc = auc\n\n            state = model.module.state_dict() if torch.cuda.device_count()>1 else model.state_dict()\n\n            torch.save(state, \"best_brain_model.pth\")\n\n            print(\"💾 Best model saved!\")\n\n\n# -----------------------------\n# Main\n# -----------------------------\nif __name__ == \"__main__\":\n\n    train_model()","metadata":{"_uuid":"c14ba2a1-7a9b-4b82-9d4d-452a331e4b6b","_cell_guid":"94dfc121-5703-44d7-b7b3-8ab1d9e93d95","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2026-03-05T18:59:50.615667Z","iopub.execute_input":"2026-03-05T18:59:50.616059Z","iopub.status.idle":"2026-03-05T20:09:43.065103Z","shell.execute_reply.started":"2026-03-05T18:59:50.616022Z","shell.execute_reply":"2026-03-05T20:09:43.064108Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls /kaggle/working/","metadata":{"_uuid":"9606c30a-2163-4adb-b24f-19f9ab1910d1","_cell_guid":"255d010b-bc12-413b-b2cc-69ab7e99a0a8","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2026-03-05T20:09:43.067059Z","iopub.execute_input":"2026-03-05T20:09:43.067421Z","iopub.status.idle":"2026-03-05T20:09:43.226721Z","shell.execute_reply.started":"2026-03-05T20:09:43.06739Z","shell.execute_reply":"2026-03-05T20:09:43.225793Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"/kaggle/working/best_brain_model.pth\n/kaggle/working/best_val_preds.npy\n/kaggle/working/best_val_targets.npy","metadata":{"_uuid":"59d926f0-d20b-4c5a-b3df-4258adc6dd12","_cell_guid":"7881ad9c-59bc-4fa1-9df0-8581d08dca01","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"","metadata":{"_uuid":"d93977aa-ebe6-4511-b353-f3fb205ed1dc","_cell_guid":"177c4fe1-7513-4e7c-99da-a8c15b306f9a","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2026-02-24T16:48:56.574371Z","iopub.execute_input":"2026-02-24T16:48:56.575183Z","iopub.status.idle":"2026-02-24T16:48:58.026798Z","shell.execute_reply.started":"2026-02-24T16:48:56.575149Z","shell.execute_reply":"2026-02-24T16:48:58.0261Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pydicom\nimport torch\nimport torch.nn as nn\nfrom pathlib import Path\nfrom scipy.ndimage import zoom\nfrom torch.cuda.amp import autocast\n\n# -----------------------------\n# Configuration & Hyperparameters\n# -----------------------------\nclass CFG:\n    model_path = \"/kaggle/working/best_brain_model.pth\"\n    # The threshold we discovered during evaluation\n    threshold = 0.1561 \n    \n    img_size = 128\n    depth = 64\n    mods = [\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"]\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# -----------------------------\n# Preprocessing Functions (Must match training)\n# -----------------------------\ndef load_dicom_volume(mod_path):\n    files = list(mod_path.glob(\"*.dcm\"))\n    if len(files) == 0: return None\n    \n    nodes = []\n    for f in files:\n        d = pydicom.dcmread(f, stop_before_pixels=True)\n        try:\n            z = float(d.ImagePositionPatient[2])\n        except:\n            z = getattr(d, \"InstanceNumber\", 0)\n        nodes.append((f, z))\n    \n    nodes.sort(key=lambda x: x[1])\n    volume = []\n    for f, _ in nodes:\n        d = pydicom.dcmread(f)\n        img = d.pixel_array.astype(np.float32)\n        img = img * getattr(d, \"RescaleSlope\", 1) + getattr(d, \"RescaleIntercept\", 0)\n        volume.append(img)\n    return np.stack(volume) if len(volume) > 0 else None\n\ndef get_crop_box(volume):\n    try:\n        mask = volume > np.percentile(volume, 5)\n        coords = np.where(mask)\n        if len(coords[0]) == 0: return volume\n        zmin, zmax = coords[0].min(), coords[0].max()\n        ymin, ymax = coords[1].min(), coords[1].max()\n        xmin, xmax = coords[2].min(), coords[2].max()\n        pad = 3\n        zmin, zmax = max(zmin-pad, 0), min(zmax+pad, volume.shape[0])\n        ymin, ymax = max(ymin-pad, 0), min(ymax+pad, volume.shape[1])\n        xmin, xmax = max(xmin-pad, 0), min(xmax+pad, volume.shape[2])\n        return volume[zmin:zmax, ymin:ymax, xmin:xmax]\n    except: return volume\n\ndef resample_volume(volume):\n    z, h, w = volume.shape\n    factors = (CFG.depth / z, CFG.img_size / h, CFG.img_size / w)\n    return zoom(volume, factors, order=1)\n\ndef normalize(volume):\n    if np.all(volume == 0): return volume\n    low, high = np.percentile(volume, 0.5), np.percentile(volume, 99.5)\n    volume = np.clip(volume, low, high)\n    std = volume.std()\n    return (volume - volume.mean()) / std if std > 1e-6 else np.zeros_like(volume)\n\n# -----------------------------\n# Model Architecture\n# -----------------------------\nclass ResNet3dBlock(nn.Module):\n    def __init__(self, in_channels, out_channels, stride=1):\n        super().__init__()\n        self.conv1 = nn.Conv3d(in_channels, out_channels, 3, stride=stride, padding=1, bias=False)\n        self.bn1 = nn.BatchNorm3d(out_channels)\n        self.relu = nn.ReLU(inplace=True)\n        self.conv2 = nn.Conv3d(out_channels, out_channels, 3, padding=1, bias=False)\n        self.bn2 = nn.BatchNorm3d(out_channels)\n        self.shortcut = nn.Sequential()\n        if stride != 1 or in_channels != out_channels:\n            self.shortcut = nn.Sequential(\n                nn.Conv3d(in_channels, out_channels, 1, stride=stride, bias=False),\n                nn.BatchNorm3d(out_channels)\n            )\n    def forward(self, x):\n        out = self.relu(self.bn1(self.conv1(x)))\n        out = self.bn2(self.conv2(out))\n        out += self.shortcut(x)\n        return self.relu(out)\n\nclass BrainNet3D(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.stem = nn.Sequential(\n            nn.Conv3d(4, 32, 7, stride=(1, 2, 2), padding=3, bias=False),\n            nn.BatchNorm3d(32), nn.ReLU(inplace=True),\n            nn.MaxPool3d(3, stride=2, padding=1)\n        )\n        self.layer1 = ResNet3dBlock(32, 64, 2)\n        self.layer2 = ResNet3dBlock(64, 128, 2)\n        self.layer3 = ResNet3dBlock(128, 256, 2)\n        self.avgpool = nn.AdaptiveAvgPool3d((1, 1, 1))\n        self.fc = nn.Linear(256, 1)\n\n    def forward(self, x):\n        x = self.stem(x); x = self.layer1(x); x = self.layer2(x); x = self.layer3(x)\n        x = self.avgpool(x); x = torch.flatten(x, 1)\n        return self.fc(x)\n\n# -----------------------------\n# Single Patient Inference\n# -----------------------------\ndef predict_patient(patient_path):\n    patient_path = Path(patient_path)\n    print(f\"🔄 Processing Patient: {patient_path.name}...\")\n    \n    # 1. Preprocess DICOMs\n    patient_volumes = []\n    for mod in CFG.mods:\n        mod_path = patient_path / mod\n        vol = load_dicom_volume(mod_path)\n        if vol is None:\n            vol = np.zeros((CFG.depth, CFG.img_size, CFG.img_size), dtype=np.float32)\n        else:\n            vol = get_crop_box(vol)\n            vol = resample_volume(vol)\n            vol = normalize(vol)\n        patient_volumes.append(vol)\n    \n    # Create input tensor (1, 4, 64, 128, 128)\n    input_tensor = torch.from_numpy(np.stack(patient_volumes)).float().unsqueeze(0).to(CFG.device)\n    \n    # 2. Load Model\n    model = BrainNet3D()\n    # If trained with DataParallel, we load the state_dict carefully\n    checkpoint = torch.load(CFG.model_path, map_location=CFG.device)\n    model.load_state_dict(checkpoint)\n    model.to(CFG.device).eval()\n    \n    # 3. Predict\n    with torch.no_grad():\n        with autocast():\n            logit = model(input_tensor)\n            prob = torch.sigmoid(logit).cpu().item()\n    \n    # 4. Result\n    prediction = \"PRESENT (1)\" if prob >= CFG.threshold else \"ABSENT (0)\"\n    \n    print(\"\\n\" + \"=\"*35)\n    print(f\"      PREDICTION RESULT      \")\n    print(\"=\"*35)\n    print(f\"Raw Probability:  {prob:.4f}\")\n    print(f\"Applied Threshold: {CFG.threshold:.4f}\")\n    print(f\"MGMT Status:      {prediction}\")\n    print(\"=\"*35)\n\nif __name__ == \"__main__\":\n    # CHANGE THIS to your manual patient folder path\n    manual_path = \"/kaggle/input/competitions/rsna-miccai-brain-tumor-radiogenomic-classification/test/00015\"\n    predict_patient(manual_path)","metadata":{"_uuid":"cf3cefd3-93b7-4c33-9728-fc4a72b0bc45","_cell_guid":"60cb56fb-3040-4b65-be56-ef1be26ca13b","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2026-03-05T20:13:58.183331Z","iopub.execute_input":"2026-03-05T20:13:58.184165Z","iopub.status.idle":"2026-03-05T20:14:16.134804Z","shell.execute_reply.started":"2026-03-05T20:13:58.184111Z","shell.execute_reply":"2026-03-05T20:14:16.133944Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader, WeightedRandomSampler\nfrom torch.cuda.amp import autocast, GradScaler\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score\nfrom pathlib import Path\nimport random\nfrom tqdm import tqdm\n\n# -----------------------------\n# Configuration\n# -----------------------------\nclass CFG:\n    seed = 42\n    img_size = 128\n    depth = 64\n    batch_size = 16 \n    epochs = 60 # Reduced epochs as Sampler converges faster\n    lr = 1e-4   # Slightly lower LR for stability\n    weight_decay = 1e-2 # Standard regularization\n    num_workers = 4 \n    \n    npy_path = Path(\"/kaggle/input/datasets/raj072004/preprocessed-final\")\n    labels_path = Path(\"/kaggle/input/datasets/raj072004/preprocessed-final/train_labels.csv\")\n\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed) \n    torch.backends.cudnn.deterministic = False \n    torch.backends.cudnn.benchmark = True      \n\nseed_everything(CFG.seed)\n\n# -----------------------------\n# Dataset\n# -----------------------------\nclass BrainTumorDataset(Dataset):\n    def __init__(self, df, npy_path, transform=None):\n        self.df = df\n        self.npy_path = npy_path\n        self.transform = transform\n\n    def __len__(self): return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        patient_id = str(int(row['BraTS21ID'])).zfill(5)\n        file_path = self.npy_path / f\"{patient_id}.npy\"\n        \n        data = np.load(file_path).astype(np.float32)\n        \n        if self.transform:\n            data = self.transform(data)\n            \n        label = torch.tensor(row['MGMT_value'], dtype=torch.float32)\n        return torch.from_numpy(data), label\n\ndef train_transform(data):\n    if random.random() > 0.5: data = np.flip(data, axis=1).copy()\n    if random.random() > 0.5: data = np.flip(data, axis=2).copy()\n    if random.random() > 0.5: data += np.random.normal(0, 0.005, data.shape)\n    return data\n\n# -----------------------------\n# Architecture\n# -----------------------------\nclass ResNet3dBlock(nn.Module):\n    def __init__(self, in_channels, out_channels, stride=1):\n        super().__init__()\n        self.conv1 = nn.Conv3d(in_channels, out_channels, 3, stride=stride, padding=1, bias=False)\n        self.bn1 = nn.BatchNorm3d(out_channels)\n        self.relu = nn.ReLU(inplace=True)\n        self.conv2 = nn.Conv3d(out_channels, out_channels, 3, padding=1, bias=False)\n        self.bn2 = nn.BatchNorm3d(out_channels)\n        self.shortcut = nn.Sequential()\n        if stride != 1 or in_channels != out_channels:\n            self.shortcut = nn.Sequential(\n                nn.Conv3d(in_channels, out_channels, 1, stride=stride, bias=False),\n                nn.BatchNorm3d(out_channels)\n            )\n    def forward(self, x):\n        out = self.relu(self.bn1(self.conv1(x)))\n        out = self.bn2(self.conv2(out))\n        out += self.shortcut(x)\n        return self.relu(out)\n\nclass BrainNet3D(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.stem = nn.Sequential(\n            nn.Conv3d(4, 32, 7, stride=(1, 2, 2), padding=3, bias=False),\n            nn.BatchNorm3d(32), nn.ReLU(inplace=True),\n            nn.MaxPool3d(3, stride=2, padding=1)\n        )\n        self.layer1 = ResNet3dBlock(32, 64, 2)\n        self.layer2 = ResNet3dBlock(64, 128, 2)\n        self.layer3 = ResNet3dBlock(128, 256, 2)\n        self.dropout = nn.Dropout3d(p=0.1) # Reduced dropout\n        self.avgpool = nn.AdaptiveAvgPool3d((1, 1, 1))\n        self.fc = nn.Linear(256, 1)\n\n    def forward(self, x):\n        x = self.stem(x); x = self.layer1(x); x = self.layer2(x); x = self.layer3(x)\n        x = self.dropout(x); x = self.avgpool(x); x = torch.flatten(x, 1)\n        return self.fc(x)\n\n# -----------------------------\n# Training Routine\n# -----------------------------\ndef train_model():\n    df = pd.read_csv(CFG.labels_path)\n    df['exists'] = df['BraTS21ID'].apply(lambda x: (CFG.npy_path / f\"{str(x).zfill(5)}.npy\").exists())\n    df = df[df['exists'] == True].reset_index(drop=True)\n    \n    train_df, valid_df = train_test_split(df, test_size=0.15, stratify=df['MGMT_value'], random_state=CFG.seed)\n\n    # --- BALANCING STRATEGY: WeightedSampler ---\n    class_counts = train_df['MGMT_value'].value_counts().to_dict()\n    weights = [1.0/class_counts[label] for label in train_df['MGMT_value']]\n    sampler = WeightedRandomSampler(torch.DoubleTensor(weights), len(weights))\n\n    train_loader = DataLoader(\n        BrainTumorDataset(train_df, CFG.npy_path, transform=train_transform),\n        batch_size=CFG.batch_size, sampler=sampler, num_workers=CFG.num_workers, pin_memory=True\n    )\n    valid_loader = DataLoader(\n        BrainTumorDataset(valid_df, CFG.npy_path),\n        batch_size=CFG.batch_size, shuffle=False, num_workers=CFG.num_workers\n    )\n\n    model = BrainNet3D().to(\"cuda\")\n    if torch.cuda.device_count() > 1: model = nn.DataParallel(model)\n\n    # Changed pos_weight to 0.9 to give Class 0 (Absent) slightly more importance\n    criterion = nn.BCEWithLogitsLoss(pos_weight=torch.tensor([0.9]).to(\"cuda\"))\n    \n    optimizer = optim.AdamW(model.parameters(), lr=CFG.lr, weight_decay=CFG.weight_decay)\n    scheduler = optim.lr_scheduler.CosineAnnealingLR(optimizer, T_max=CFG.epochs)\n    scaler = GradScaler()\n    \n    best_auc = 0\n    \n    for epoch in range(CFG.epochs):\n        model.train()\n        train_loss = []\n        for imgs, labels in tqdm(train_loader, desc=f\"Epoch {epoch+1}\"):\n            imgs, labels = imgs.to(\"cuda\"), labels.to(\"cuda\").unsqueeze(1)\n            \n            optimizer.zero_grad(set_to_none=True)\n            with autocast():\n                outputs = model(imgs)\n                loss = criterion(outputs, labels)\n            \n            scaler.scale(loss).backward()\n            scaler.step(optimizer)\n            scaler.update()\n            train_loss.append(loss.item())\n\n        model.eval()\n        preds, targets = [], []\n        with torch.no_grad():\n            for imgs, labels in valid_loader:\n                imgs = imgs.to(\"cuda\")\n                with autocast():\n                    outputs = model(imgs)\n                preds.extend(torch.sigmoid(outputs).cpu().numpy().ravel())\n                targets.extend(labels.numpy())\n        \n        auc = roc_auc_score(targets, preds)\n        scheduler.step()\n        \n        print(f\"Val AUC: {auc:.4f} | Avg Loss: {np.mean(train_loss):.4f}\")\n        \n        if auc > best_auc:\n            best_auc = auc\n            torch.save(model.state_dict(), \"best_balanced_model.pth\")\n            print(\"💾 New Best Balanced Model Saved!\")\n\nif __name__ == \"__main__\":\n    train_model()","metadata":{"_uuid":"345ebbcb-c406-4463-903a-a6e13a7011fc","_cell_guid":"36cd6742-6647-4cea-b92a-14e61d9d144e","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2026-02-25T13:08:49.965795Z","iopub.execute_input":"2026-02-25T13:08:49.966602Z","iopub.status.idle":"2026-02-25T13:34:35.829659Z","shell.execute_reply.started":"2026-02-25T13:08:49.966569Z","shell.execute_reply":"2026-02-25T13:34:35.828795Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pydicom\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nfrom pathlib import Path\nfrom tqdm import tqdm\nfrom scipy.ndimage import zoom\nfrom torch.cuda.amp import autocast\n\n# -----------------------------\n# Configuration\n# -----------------------------\nclass CFG:\n    # 1. Path to your TEST folders (raw .dcm files)\n    test_path = Path(\"/kaggle/input/competitions/rsna-miccai-brain-tumor-radiogenomic-classification/test\")\n    # 2. Path to your NEW regularized model\n    model_path = \"/kaggle/working/best_brain_model.pth\"\n    \n    img_size = 128\n    depth = 64\n    mods = [\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"]\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# -----------------------------\n# Preprocessing Logic (Raw DICOM to Model-Ready)\n# -----------------------------\ndef load_dicom_volume(mod_path):\n    files = list(mod_path.glob(\"*.dcm\"))\n    if not files: return None\n    \n    # Sort slices by position\n    nodes = []\n    for f in files:\n        d = pydicom.dcmread(f, stop_before_pixels=True)\n        z = float(getattr(d, \"ImagePositionPatient\", [0,0,0])[2])\n        nodes.append((f, z))\n    nodes.sort(key=lambda x: x[1])\n    \n    volume = []\n    for f, _ in nodes:\n        d = pydicom.dcmread(f)\n        img = d.pixel_array.astype(np.float32)\n        # Apply rescale if present\n        img = img * getattr(d, \"RescaleSlope\", 1) + getattr(d, \"RescaleIntercept\", 0)\n        volume.append(img)\n    return np.stack(volume)\n\ndef preprocess_volume(vol):\n    # 1. Crop\n    mask = vol > np.percentile(vol, 5)\n    coords = np.where(mask)\n    if len(coords[0]) > 0:\n        vol = vol[coords[0].min():coords[0].max(), coords[1].min():coords[1].max(), coords[2].min():coords[2].max()]\n    # 2. Resample\n    z, h, w = vol.shape\n    vol = zoom(vol, (CFG.depth/z, CFG.img_size/h, CFG.img_size/w), order=1)\n    # 3. Normalize\n    std = vol.std()\n    return (vol - vol.mean()) / std if std > 1e-6 else np.zeros_like(vol)\n\n# -----------------------------\n# Model (Regularized Architecture)\n# -----------------------------\nclass ResNet3dBlock(nn.Module):\n    def __init__(self, in_channels, out_channels, stride=1):\n        super().__init__()\n        self.conv1 = nn.Conv3d(in_channels, out_channels, 3, stride=stride, padding=1, bias=False)\n        self.bn1 = nn.BatchNorm3d(out_channels)\n        self.relu = nn.ReLU(inplace=True)\n        self.conv2 = nn.Conv3d(out_channels, out_channels, 3, padding=1, bias=False)\n        self.bn2 = nn.BatchNorm3d(out_channels)\n        self.shortcut = nn.Sequential()\n        if stride != 1 or in_channels != out_channels:\n            self.shortcut = nn.Sequential(\n                nn.Conv3d(in_channels, out_channels, 1, stride=stride, bias=False),\n                nn.BatchNorm3d(out_channels)\n            )\n    def forward(self, x):\n        out = self.relu(self.bn1(self.conv1(x)))\n        out = self.bn2(self.conv2(out))\n        out += self.shortcut(x)\n        return self.relu(out)\n\nclass BrainNet3D(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.stem = nn.Sequential(\n            nn.Conv3d(4, 32, 7, stride=(1, 2, 2), padding=3, bias=False),\n            nn.BatchNorm3d(32), nn.ReLU(inplace=True),\n            nn.MaxPool3d(3, stride=2, padding=1)\n        )\n        self.layer1 = ResNet3dBlock(32, 64, 2)\n        self.layer2 = ResNet3dBlock(64, 128, 2)\n        self.layer3 = ResNet3dBlock(128, 256, 2)\n        self.dropout = nn.Dropout3d(p=0.2) # Keeping dropout for consistency\n        self.avgpool = nn.AdaptiveAvgPool3d((1, 1, 1))\n        self.fc = nn.Linear(256, 1)\n\n    def forward(self, x):\n        x = self.stem(x); x = self.layer1(x); x = self.layer2(x); x = self.layer3(x)\n        x = self.dropout(x); x = self.avgpool(x); x = torch.flatten(x, 1)\n        return self.fc(x)\n\n# -----------------------------\n# Main Inference Loop\n# -----------------------------\ndef run_test_inference():\n    # 1. Load Model\n    model = BrainNet3D()\n    model.load_state_dict(torch.load(CFG.model_path, map_location=CFG.device))\n    model.to(CFG.device).eval()\n    \n    test_patients = sorted([f.name for f in CFG.test_path.iterdir() if f.is_dir()])\n    results = []\n\n    print(f\"Running inference on {len(test_patients)} patients from Test Split...\")\n\n    for patient_id in tqdm(test_patients):\n        try:\n            patient_vols = []\n            for mod in CFG.mods:\n                mod_path = CFG.test_path / patient_id / mod\n                vol = load_dicom_volume(mod_path)\n                if vol is not None:\n                    vol = preprocess_volume(vol)\n                else:\n                    vol = np.zeros((CFG.depth, CFG.img_size, CFG.img_size))\n                patient_vols.append(vol)\n            \n            # Prepare tensor (1, 4, 64, 128, 128)\n            img_tensor = torch.from_numpy(np.stack(patient_vols)).float().unsqueeze(0).to(CFG.device)\n            \n            with torch.no_grad():\n                with autocast():\n                    output = model(img_tensor)\n                    prob = torch.sigmoid(output).item()\n            \n            results.append({\"BraTS21ID\": int(patient_id), \"MGMT_value\": prob})\n            \n        except Exception as e:\n            print(f\"Error processing {patient_id}: {e}\")\n            results.append({\"BraTS21ID\": int(patient_id), \"MGMT_value\": 0.5})\n\n    # Save to CSV\n    submission = pd.DataFrame(results)\n    submission.to_csv(\"submission.csv\", index=False)\n    print(\"\\n✅ Inference Complete! 'submission.csv' generated.\")\n\nif __name__ == \"__main__\":\n    run_test_inference()","metadata":{"_uuid":"c22f714e-2bea-43cd-b00c-6ad94d0d49ce","_cell_guid":"8811e686-39cc-4108-8920-ed5e8e1b5615","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2026-03-05T20:32:31.022837Z","iopub.execute_input":"2026-03-05T20:32:31.023469Z","iopub.status.idle":"2026-03-05T22:01:41.173725Z","shell.execute_reply.started":"2026-03-05T20:32:31.02344Z","shell.execute_reply":"2026-03-05T22:01:41.172094Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nfrom pathlib import Path\nfrom tqdm import tqdm\nfrom torch.amp import autocast\n\n# -----------------------------\n# Configuration\n# -----------------------------\nclass CFG:\n    # folder containing patient .npy volumes\n    test_path = Path(\"/kaggle/input/datasets/raj072004/preprocessed-final\")  \n    \n    # trained model\n    model_path = \"/kaggle/working/best_brain_model.pth\"\n\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n\n# -----------------------------\n# Model\n# -----------------------------\nclass ResNet3dBlock(nn.Module):\n    def __init__(self, in_channels, out_channels, stride=1):\n        super().__init__()\n\n        self.conv1 = nn.Conv3d(in_channels, out_channels, 3, stride=stride, padding=1, bias=False)\n        self.bn1 = nn.BatchNorm3d(out_channels)\n        self.relu = nn.ReLU(inplace=True)\n\n        self.conv2 = nn.Conv3d(out_channels, out_channels, 3, padding=1, bias=False)\n        self.bn2 = nn.BatchNorm3d(out_channels)\n\n        self.shortcut = nn.Sequential()\n\n        if stride != 1 or in_channels != out_channels:\n            self.shortcut = nn.Sequential(\n                nn.Conv3d(in_channels, out_channels, 1, stride=stride, bias=False),\n                nn.BatchNorm3d(out_channels)\n            )\n\n    def forward(self, x):\n        out = self.relu(self.bn1(self.conv1(x)))\n        out = self.bn2(self.conv2(out))\n        out += self.shortcut(x)\n        return self.relu(out)\n\n\nclass BrainNet3D(nn.Module):\n\n    def __init__(self):\n        super().__init__()\n\n        self.stem = nn.Sequential(\n            nn.Conv3d(4, 32, 7, stride=(1,2,2), padding=3, bias=False),\n            nn.BatchNorm3d(32),\n            nn.ReLU(inplace=True),\n            nn.MaxPool3d(3, stride=2, padding=1)\n        )\n\n        self.layer1 = ResNet3dBlock(32, 64, 2)\n        self.layer2 = ResNet3dBlock(64, 128, 2)\n        self.layer3 = ResNet3dBlock(128, 256, 2)\n\n        self.dropout = nn.Dropout3d(0.2)\n\n        self.avgpool = nn.AdaptiveAvgPool3d((1,1,1))\n        self.fc = nn.Linear(256,1)\n\n    def forward(self,x):\n\n        x = self.stem(x)\n\n        x = self.layer1(x)\n        x = self.layer2(x)\n        x = self.layer3(x)\n\n        x = self.dropout(x)\n\n        x = self.avgpool(x)\n\n        x = torch.flatten(x,1)\n\n        return self.fc(x)\n\n\n# -----------------------------\n# Inference\n# -----------------------------\ndef run_inference():\n\n    # load model\n    model = BrainNet3D()\n    model.load_state_dict(torch.load(CFG.model_path, map_location=CFG.device))\n\n    model.to(CFG.device)\n    model.eval()\n\n    npy_files = sorted(list(CFG.test_path.glob(\"*.npy\")))\n\n    print(f\"Found {len(npy_files)} patients\")\n\n    results = []\n\n    for file in tqdm(npy_files):\n\n        patient_id = file.stem\n\n        try:\n\n            vol = np.load(file)\n\n            # shape should be (4,64,128,128)\n            tensor = torch.from_numpy(vol).float().unsqueeze(0).to(CFG.device)\n\n            with torch.no_grad():\n\n                with autocast(\"cuda\"):\n\n                    output = model(tensor)\n\n                    prob = torch.sigmoid(output).item()\n\n            results.append({\n                \"BraTS21ID\": int(patient_id),\n                \"MGMT_value\": prob\n            })\n\n        except Exception as e:\n\n            print(f\"Error with {patient_id}: {e}\")\n\n            results.append({\n                \"BraTS21ID\": int(patient_id),\n                \"MGMT_value\": 0.5\n            })\n\n    df = pd.DataFrame(results)\n\n    df.to_csv(\"train_predictions.csv\", index=False)\n\n    print(\"Inference finished → train_predictions.csv created\")\n\n\nif __name__ == \"__main__\":\n    run_inference()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T05:14:01.077913Z","iopub.execute_input":"2026-03-06T05:14:01.078499Z","iopub.status.idle":"2026-03-06T05:14:55.841572Z","shell.execute_reply.started":"2026-03-06T05:14:01.078472Z","shell.execute_reply":"2026-03-06T05:14:55.840793Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = pd.read_csv(\"/kaggle/working/train_predictions.csv\")","metadata":{"_uuid":"bb840b67-e530-48aa-abc4-6771f3e291f1","_cell_guid":"8718694b-73f1-4f67-860c-c71609a40c89","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2026-03-06T05:15:01.486168Z","iopub.execute_input":"2026-03-06T05:15:01.486469Z","iopub.status.idle":"2026-03-06T05:15:01.493013Z","shell.execute_reply.started":"2026-03-06T05:15:01.486442Z","shell.execute_reply":"2026-03-06T05:15:01.492399Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data","metadata":{"_uuid":"ffe1ae75-acd5-48b9-85c1-0876975cf739","_cell_guid":"37a87794-e3e4-4c38-95c9-1b9fdff41baa","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2026-03-06T05:15:07.439251Z","iopub.execute_input":"2026-03-06T05:15:07.43955Z","iopub.status.idle":"2026-03-06T05:15:07.467244Z","shell.execute_reply.started":"2026-03-06T05:15:07.439523Z","shell.execute_reply":"2026-03-06T05:15:07.466527Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import (\n    roc_auc_score, confusion_matrix, precision_recall_curve, \n    f1_score, accuracy_score, classification_report\n)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom pathlib import Path\nfrom tqdm import tqdm\n\n# -----------------------------\n# Configuration\n# -----------------------------\nclass CFG:\n    seed = 42\n    batch_size = 16\n    # UPDATE THIS to your new regularized model file name\n    model_path = \"/kaggle/working/best_brain_model_regularized.pth\" \n    npy_path = Path(\"/kaggle/input/datasets/raj072004/preprocessed-final\")\n    labels_path = Path(\"/kaggle/input/datasets/raj072004/preprocessed-final/train_labels.csv\")\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# -----------------------------\n# Architecture (Must match exactly)\n# -----------------------------\nclass ResNet3dBlock(nn.Module):\n    def __init__(self, in_channels, out_channels, stride=1):\n        super().__init__()\n        self.conv1 = nn.Conv3d(in_channels, out_channels, 3, stride=stride, padding=1, bias=False)\n        self.bn1 = nn.BatchNorm3d(out_channels)\n        self.relu = nn.ReLU(inplace=True)\n        self.conv2 = nn.Conv3d(out_channels, out_channels, 3, padding=1, bias=False)\n        self.bn2 = nn.BatchNorm3d(out_channels)\n        self.shortcut = nn.Sequential()\n        if stride != 1 or in_channels != out_channels:\n            self.shortcut = nn.Sequential(\n                nn.Conv3d(in_channels, out_channels, 1, stride=stride, bias=False),\n                nn.BatchNorm3d(out_channels)\n            )\n    def forward(self, x):\n        out = self.relu(self.bn1(self.conv1(x)))\n        out = self.bn2(self.conv2(out))\n        out += self.shortcut(x)\n        return self.relu(out)\n\nclass BrainNet3D(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.stem = nn.Sequential(\n            nn.Conv3d(4, 32, 7, stride=(1, 2, 2), padding=3, bias=False),\n            nn.BatchNorm3d(32), nn.ReLU(inplace=True),\n            nn.MaxPool3d(3, stride=2, padding=1)\n        )\n        self.layer1 = ResNet3dBlock(32, 64, 2)\n        self.layer2 = ResNet3dBlock(64, 128, 2)\n        self.layer3 = ResNet3dBlock(128, 256, 2)\n        self.dropout = nn.Dropout3d(p=0.2) # REGULARIZATION INCLUDED\n        self.avgpool = nn.AdaptiveAvgPool3d((1, 1, 1))\n        self.fc = nn.Linear(256, 1)\n\n    def forward(self, x):\n        x = self.stem(x); x = self.layer1(x); x = self.layer2(x); x = self.layer3(x)\n        x = self.dropout(x); x = self.avgpool(x); x = torch.flatten(x, 1)\n        return self.fc(x)\n\n# -----------------------------\n# Evaluation Pipeline\n# -----------------------------\ndef run_eval():\n    # 1. Split Data\n    df = pd.read_csv(CFG.labels_path)\n    df['exists'] = df['BraTS21ID'].apply(lambda x: (CFG.npy_path / f\"{str(x).zfill(5)}.npy\").exists())\n    df = df[df['exists'] == True]\n    _, valid_df = train_test_split(df, test_size=0.15, stratify=df['MGMT_value'], random_state=CFG.seed)\n    \n    # 2. Dataset/Loader\n    class EvalDataset(Dataset):\n        def __init__(self, df, npy_path):\n            self.df = df\n            self.npy_path = npy_path\n        def __len__(self): return len(self.df)\n        def __getitem__(self, idx):\n            row = self.df.iloc[idx]\n            p_id = str(int(row['BraTS21ID'])).zfill(5)\n            data = np.load(self.npy_path / f\"{p_id}.npy\")\n            return torch.from_numpy(data.astype(np.float32)), torch.tensor(row['MGMT_value'], dtype=torch.float32)\n\n    v_loader = DataLoader(EvalDataset(valid_df, CFG.npy_path), batch_size=CFG.batch_size, shuffle=False)\n\n    # 3. Load & Predict\n    model = BrainNet3D().to(CFG.device)\n    model.load_state_dict(torch.load(CFG.model_path, map_location=CFG.device))\n    model.eval()\n\n    preds, targets = [], []\n    with torch.no_grad():\n        for imgs, labels in tqdm(v_loader):\n            imgs = imgs.to(CFG.device)\n            out = model(imgs)\n            preds.extend(torch.sigmoid(out).cpu().numpy().ravel())\n            targets.extend(labels.numpy())\n    \n    preds, targets = np.array(preds), np.array(targets)\n\n    # 4. Metrics & Plotting\n    # Find best threshold dynamically\n    precision, recall, thresholds = precision_recall_curve(targets, preds)\n    f1_scores = 2 * (precision * recall) / (precision + recall + 1e-8)\n    best_thresh = thresholds[np.argmax(f1_scores)]\n    preds_bin = (preds >= best_thresh).astype(int)\n\n    print(\"\\n\" + \"=\"*30)\n    print(\" REGULARIZED MODEL RESULTS \")\n    print(\"=\"*30)\n    print(f\"AUC Score: {roc_auc_score(targets, preds):.4f}\")\n    print(f\"Accuracy:  {accuracy_score(targets, preds_bin):.4f}\")\n    print(\"-\" * 30)\n    print(classification_report(targets, preds_bin))\n\n    # Confusion Matrix Plot\n    plt.figure(figsize=(6, 5))\n    cm = confusion_matrix(targets, preds_bin)\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Greens', xticklabels=['Absent', 'Present'], yticklabels=['Absent', 'Present'])\n    plt.xlabel('Predicted'); plt.ylabel('Actual'); plt.title('Confusion Matrix: Regularized Model')\n    plt.show()\n\nif __name__ == \"__main__\":\n    run_eval()","metadata":{"_uuid":"dddaf83d-5647-4608-b859-22b963e7d998","_cell_guid":"e97c0985-f4bb-4050-8e06-f7fc660e7fe8","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2026-02-25T13:02:24.327413Z","iopub.execute_input":"2026-02-25T13:02:24.327724Z","iopub.status.idle":"2026-02-25T13:02:42.22536Z","shell.execute_reply.started":"2026-02-25T13:02:24.327698Z","shell.execute_reply":"2026-02-25T13:02:42.224675Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"_uuid":"9b757ff8-2a0d-4814-a3bd-734702c4de53","_cell_guid":"15a4979b-7047-40a3-b78a-18c0f8c7db47","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null}]}