{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport pydicom\nimport torch\nimport torch.nn as nn\nimport torchvision.transforms as transforms\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import models\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n\nDEVICE = \"cuda\" if torch.cuda.is_available() else \"cpu\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-14T14:00:25.777909Z","iopub.execute_input":"2026-04-14T14:00:25.778188Z","iopub.status.idle":"2026-04-14T14:00:40.132114Z","shell.execute_reply.started":"2026-04-14T14:00:25.778154Z","shell.execute_reply":"2026-04-14T14:00:40.131092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nfor root, dirs, files in os.walk('/kaggle/input'):\n    print(root)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-12T15:42:27.377243Z","iopub.execute_input":"2026-04-12T15:42:27.378202Z","iopub.status.idle":"2026-04-12T15:43:19.482297Z","shell.execute_reply.started":"2026-04-12T15:42:27.378159Z","shell.execute_reply":"2026-04-12T15:43:19.481476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class RSNADataset(Dataset):\n    def __init__(self, img_dir, labels_csv, transform=None):\n        self.img_dir = img_dir\n        self.df = pd.read_csv(labels_csv)\n        self.transform = transform\n\n        # convert detection → classification\n        self.labels = self.df.groupby('patientId')['Target'].max().reset_index()\n\n    def __len__(self):\n        return len(self.labels)\n\n    def __getitem__(self, idx):\n        pid = self.labels.iloc[idx]['patientId']\n        label = self.labels.iloc[idx]['Target']\n\n        path = os.path.join(self.img_dir, f\"{pid}.dcm\")\n        dicom = pydicom.dcmread(path)\n        img = dicom.pixel_array\n\n        img = cv2.resize(img, (224,224))\n        img = np.stack([img]*3, axis=-1)\n\n        if self.transform:\n            img = self.transform(img)\n\n        return img.float(), torch.tensor(label, dtype=torch.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-14T14:00:48.965670Z","iopub.execute_input":"2026-04-14T14:00:48.966129Z","iopub.status.idle":"2026-04-14T14:00:48.972861Z","shell.execute_reply.started":"2026-04-14T14:00:48.966099Z","shell.execute_reply":"2026-04-14T14:00:48.972220Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_tf = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomRotation(10)\n])\n\ntest_tf = transforms.Compose([\n    transforms.ToTensor()\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-14T14:00:50.908420Z","iopub.execute_input":"2026-04-14T14:00:50.909053Z","iopub.status.idle":"2026-04-14T14:00:50.913359Z","shell.execute_reply.started":"2026-04-14T14:00:50.909027Z","shell.execute_reply":"2026-04-14T14:00:50.912429Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_PATH = \"/kaggle/input/competitions/rsna-pneumonia-detection-challenge\"\n\ndataset = RSNADataset(\n    img_dir=f\"{DATA_PATH}/stage_2_train_images\",\n    labels_csv=f\"{DATA_PATH}/stage_2_train_labels.csv\",\n    transform=train_tf\n)\n\n# split\ntrain_size = int(0.8 * len(dataset))\ntest_size = len(dataset) - train_size\ntrain_ds, test_ds = torch.utils.data.random_split(dataset, [train_size, test_size])\n\ntrain_loader = DataLoader(train_ds, batch_size=32, shuffle=True)\ntest_loader  = DataLoader(test_ds, batch_size=32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-14T14:00:53.404573Z","iopub.execute_input":"2026-04-14T14:00:53.405018Z","iopub.status.idle":"2026-04-14T14:00:53.523619Z","shell.execute_reply.started":"2026-04-14T14:00:53.404981Z","shell.execute_reply":"2026-04-14T14:00:53.523011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class PneumoniaModel(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.model = models.resnet50(weights=\"IMAGENET1K_V1\")\n\n        self.model.fc = nn.Sequential(\n            nn.Linear(2048, 512),\n            nn.ReLU(),\n            nn.Dropout(0.5),   # MC Dropout\n            nn.Linear(512, 1)\n        )\n\n    def forward(self, x):\n        return self.model(x)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-14T14:00:55.445532Z","iopub.execute_input":"2026-04-14T14:00:55.445991Z","iopub.status.idle":"2026-04-14T14:00:55.450756Z","shell.execute_reply.started":"2026-04-14T14:00:55.445926Z","shell.execute_reply":"2026-04-14T14:00:55.450025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = PneumoniaModel().to(DEVICE)\n\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=1e-4)\n\nEPOCHS = 3\n\nfor epoch in range(EPOCHS):\n    model.train()\n    total_loss = 0\n\n    for imgs, labels in tqdm(train_loader):\n        imgs, labels = imgs.to(DEVICE), labels.to(DEVICE)\n\n        optimizer.zero_grad()\n        outputs = model(imgs).squeeze()\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n\n        total_loss += loss.item()\n\n    print(f\"Epoch {epoch+1} Loss: {total_loss:.4f}\")\n\ntorch.save(model.state_dict(), \"best_model.pth\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-14T14:00:57.557707Z","iopub.execute_input":"2026-04-14T14:00:57.558322Z","iopub.status.idle":"2026-04-14T14:24:47.817330Z","shell.execute_reply.started":"2026-04-14T14:00:57.558290Z","shell.execute_reply":"2026-04-14T14:24:47.816606Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.eval()\n\nbase_preds, base_labels = [], []\n\nwith torch.no_grad():\n    for imgs, labels in tqdm(test_loader):\n        imgs = imgs.to(DEVICE)\n        logits = model(imgs)\n\n        probs = torch.sigmoid(logits).cpu().numpy()\n        base_preds.extend(probs)\n        base_labels.extend(labels.numpy())\n\nbase_preds = np.array(base_preds).flatten()\nbase_labels = np.array(base_labels)\n\nTHRESHOLD = 0.5\nbase_bin = (base_preds >= THRESHOLD).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-14T14:29:41.521318Z","iopub.execute_input":"2026-04-14T14:29:41.522028Z","iopub.status.idle":"2026-04-14T14:31:28.687423Z","shell.execute_reply.started":"2026-04-14T14:29:41.521997Z","shell.execute_reply":"2026-04-14T14:31:28.686607Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def enable_dropout(m):\n    for layer in m.modules():\n        if isinstance(layer, nn.Dropout):\n            layer.train()\n\nmodel.eval()\nenable_dropout(model)\n\nT = 10  # MC samples\n\nall_preds = []\n\nfor _ in tqdm(range(T), desc=\"MC Sampling\"):\n    preds = []\n    with torch.no_grad():\n        for imgs, _ in test_loader:\n            imgs = imgs.to(DEVICE)\n            logits = model(imgs)\n            probs = torch.sigmoid(logits).cpu().numpy()\n            preds.extend(probs)\n    all_preds.append(preds)\n\nall_preds = np.array(all_preds)\n\nmean_preds = all_preds.mean(axis=0).flatten()\nuncertainty = all_preds.std(axis=0).flatten()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-14T14:53:25.044873Z","iopub.execute_input":"2026-04-14T14:53:25.045563Z","iopub.status.idle":"2026-04-14T15:03:45.569663Z","shell.execute_reply.started":"2026-04-14T14:53:25.045533Z","shell.execute_reply":"2026-04-14T15:03:45.569006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# dynamic threshold\nUNCERTAINTY_THRESHOLD = np.percentile(uncertainty, 75)\n\nselect_preds, select_labels, select_probs = [], [], []\n\nfor mp, un, lbl in zip(mean_preds, uncertainty, base_labels):\n    if un > UNCERTAINTY_THRESHOLD:\n        continue   # refer to doctor\n\n    pred = 1 if mp >= THRESHOLD else 0\n    select_preds.append(pred)\n    select_labels.append(lbl)\n    select_probs.append(mp)\n\ncoverage = len(select_preds) / len(base_labels)\n\nselect_preds  = np.array(select_preds)\nselect_labels = np.array(select_labels)\nselect_probs  = np.array(select_probs)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-14T15:07:56.404017Z","iopub.execute_input":"2026-04-14T15:07:56.404673Z","iopub.status.idle":"2026-04-14T15:07:56.421011Z","shell.execute_reply.started":"2026-04-14T15:07:56.404643Z","shell.execute_reply":"2026-04-14T15:07:56.419911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_metrics(y_true, y_pred, probs):\n    return {\n        \"Accuracy\": accuracy_score(y_true, y_pred),\n        \"Precision\": precision_score(y_true, y_pred),\n        \"Recall\": recall_score(y_true, y_pred),\n        \"F1\": f1_score(y_true, y_pred),\n        \"AUC\": roc_auc_score(y_true, probs)\n    }\n\nbaseline_metrics = get_metrics(base_labels, base_bin, base_preds)\nmc_metrics       = get_metrics(base_labels, (mean_preds>=0.5), mean_preds)\nsel_metrics      = get_metrics(select_labels, select_preds, select_probs)\n\ndf = pd.DataFrame([baseline_metrics, mc_metrics, sel_metrics],\n                  index=[\"Baseline\", \"MC Dropout\", \"Selective\"])\n\ndf[\"Coverage\"] = [1.0, 1.0, coverage]\n\nprint(df.round(4))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-14T15:07:59.830223Z","iopub.execute_input":"2026-04-14T15:07:59.831225Z","iopub.status.idle":"2026-04-14T15:07:59.888334Z","shell.execute_reply.started":"2026-04-14T15:07:59.831196Z","shell.execute_reply":"2026-04-14T15:07:59.887423Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[[\"Accuracy\",\"Precision\",\"Recall\",\"F1\"]].plot(kind=\"bar\", figsize=(10,5))\nplt.title(\"Model Comparison\")\nplt.ylim(0.5,1)\nplt.grid()\nplt.show()\n\ndf[\"Coverage\"].plot(kind=\"bar\")\nplt.title(\"Coverage (Doctor Referral Effect)\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-14T15:08:02.382204Z","iopub.execute_input":"2026-04-14T15:08:02.382730Z","iopub.status.idle":"2026-04-14T15:08:02.721504Z","shell.execute_reply.started":"2026-04-14T15:08:02.382702Z","shell.execute_reply":"2026-04-14T15:08:02.720651Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"torch.save({\n    'model_state': model.state_dict(),\n    'threshold': THRESHOLD,\n    'uncertainty_threshold': UNCERTAINTY_THRESHOLD\n}, \"/kaggle/working/pneumonia_model_final.pth\")\n\nprint(\"Model saved successfully\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-14T15:08:07.004769Z","iopub.execute_input":"2026-04-14T15:08:07.005751Z","iopub.status.idle":"2026-04-14T15:08:07.146471Z","shell.execute_reply.started":"2026-04-14T15:08:07.005721Z","shell.execute_reply":"2026-04-14T15:08:07.145538Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}