{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"sourceType":"competition"}],"dockerImageVersionId":31240,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\n\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import models, transforms\n\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\nfrom tqdm import tqdm\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-14T00:24:40.861907Z","iopub.execute_input":"2026-01-14T00:24:40.862542Z","iopub.status.idle":"2026-01-14T00:24:40.866702Z","shell.execute_reply.started":"2026-01-14T00:24:40.862516Z","shell.execute_reply":"2026-01-14T00:24:40.866115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/rsna-pneumonia-detection-challenge\"\nIMAGE_DIR = f\"{DATA_DIR}/stage_2_train_images\"\nLABEL_CSV = f\"{DATA_DIR}/stage_2_train_labels.csv\"\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-14T00:25:18.089214Z","iopub.execute_input":"2026-01-14T00:25:18.089931Z","iopub.status.idle":"2026-01-14T00:25:18.093328Z","shell.execute_reply.started":"2026-01-14T00:25:18.089904Z","shell.execute_reply":"2026-01-14T00:25:18.092659Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(LABEL_CSV)\n\n# Convert multiple bounding boxes → single binary label\nlabels = df.groupby(\"patientId\")[\"Target\"].max().reset_index()\n\nprint(\"Total samples:\", len(labels))\nlabels.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-14T00:25:37.608564Z","iopub.execute_input":"2026-01-14T00:25:37.609311Z","iopub.status.idle":"2026-01-14T00:25:37.707728Z","shell.execute_reply.started":"2026-01-14T00:25:37.609285Z","shell.execute_reply":"2026-01-14T00:25:37.707010Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels = labels.sample(frac=1, random_state=42).reset_index(drop=True)\n\nsplit = int(0.8 * len(labels))\ntrain_df = labels.iloc[:split]\nval_df = labels.iloc[split:]\n\nprint(\"Train:\", len(train_df))\nprint(\"Validation:\", len(val_df))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-14T00:26:25.125862Z","iopub.execute_input":"2026-01-14T00:26:25.126195Z","iopub.status.idle":"2026-01-14T00:26:25.136757Z","shell.execute_reply.started":"2026-01-14T00:26:25.126172Z","shell.execute_reply":"2026-01-14T00:26:25.136164Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class RSNADataset(Dataset):\n    def __init__(self, df, image_dir, transform=None):\n        self.df = df\n        self.image_dir = image_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        patient_id = self.df.iloc[idx][\"patientId\"]\n        label = self.df.iloc[idx][\"Target\"]\n\n        img_path = os.path.join(self.image_dir, patient_id + \".dcm\")\n\n        import pydicom\n        dicom = pydicom.dcmread(img_path)\n        img = dicom.pixel_array\n\n        img = Image.fromarray(img).convert(\"RGB\")\n\n        if self.transform:\n            img = self.transform(img)\n\n        return img, torch.tensor(label, dtype=torch.float32)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-14T00:26:48.757363Z","iopub.execute_input":"2026-01-14T00:26:48.757872Z","iopub.status.idle":"2026-01-14T00:26:48.763477Z","shell.execute_reply.started":"2026-01-14T00:26:48.757846Z","shell.execute_reply":"2026-01-14T00:26:48.762913Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize([0.485, 0.456, 0.406],\n                         [0.229, 0.224, 0.225])\n])\n\ntrain_dataset = RSNADataset(train_df, IMAGE_DIR, transform)\nval_dataset = RSNADataset(val_df, IMAGE_DIR, transform)\n\ntrain_loader = DataLoader(train_dataset, batch_size=16, shuffle=True, num_workers=2)\nval_loader = DataLoader(val_dataset, batch_size=16, shuffle=False, num_workers=2)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-14T00:27:23.062512Z","iopub.execute_input":"2026-01-14T00:27:23.063198Z","iopub.status.idle":"2026-01-14T00:27:23.068022Z","shell.execute_reply.started":"2026-01-14T00:27:23.063171Z","shell.execute_reply":"2026-01-14T00:27:23.067249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nmodel = models.densenet121(pretrained=True)\nmodel.classifier = nn.Linear(model.classifier.in_features, 1)\nmodel = model.to(device)\n\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=1e-4)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-14T00:28:04.771232Z","iopub.execute_input":"2026-01-14T00:28:04.771752Z","iopub.status.idle":"2026-01-14T00:28:05.155116Z","shell.execute_reply.started":"2026-01-14T00:28:04.771723Z","shell.execute_reply":"2026-01-14T00:28:05.154527Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"EPOCHS = 16\n\nfor epoch in range(EPOCHS):\n    model.train()\n    train_loss = 0\n\n    for images, labels in tqdm(train_loader):\n        images = images.to(device)\n        labels = labels.to(device)\n\n        optimizer.zero_grad()\n        outputs = model(images).squeeze()\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n\n        train_loss += loss.item()\n\n    print(f\"Epoch [{epoch+1}/{EPOCHS}] Train Loss: {train_loss/len(train_loader):.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-14T00:28:56.310949Z","iopub.execute_input":"2026-01-14T00:28:56.311236Z","iopub.status.idle":"2026-01-14T01:28:51.989454Z","shell.execute_reply.started":"2026-01-14T00:28:56.311215Z","shell.execute_reply":"2026-01-14T01:28:51.988732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# CELL 9: MODEL EVALUATION\n# =========================\n\nimport torch\nimport numpy as np\n\nmodel.eval()\n\nTP = TN = FP = FN = 0\nall_preds = []\nall_labels = []\n\nwith torch.no_grad():\n    for images, labels in val_loader:\n        images = images.to(device)\n        labels = labels.to(device)\n\n        outputs = model(images).squeeze()\n        probs = torch.sigmoid(outputs)\n        preds = (probs > 0.5).long()\n\n        all_preds.extend(preds.cpu().numpy())\n        all_labels.extend(labels.cpu().numpy())\n\n        TP += ((preds == 1) & (labels == 1)).sum().item()\n        TN += ((preds == 0) & (labels == 0)).sum().item()\n        FP += ((preds == 1) & (labels == 0)).sum().item()\n        FN += ((preds == 0) & (labels == 1)).sum().item()\n\n# Convert to numpy\nall_preds = np.array(all_preds)\nall_labels = np.array(all_labels)\n\n# ===== METRICS =====\naccuracy = (TP + TN) / (TP + TN + FP + FN)\n\nprecision = TP / (TP + FP + 1e-8)\nrecall = TP / (TP + FN + 1e-8)          # Sensitivity\nspecificity = TN / (TN + FP + 1e-8)\n\nf1_score = 2 * (precision * recall) / (precision + recall + 1e-8)\n\n# ===== PRINT RESULTS =====\nprint(\"===== CONFUSION MATRIX =====\")\nprint(f\"TP (Pneumonia → Pneumonia): {TP}\")\nprint(f\"TN (Normal → Normal):       {TN}\")\nprint(f\"FP (Normal → Pneumonia):    {FP}\")\nprint(f\"FN (Pneumonia → Normal):    {FN}\")\n\nprint(\"\\n===== EVALUATION METRICS =====\")\nprint(f\"Accuracy     : {accuracy*100:.2f}%\")\nprint(f\"Precision    : {precision*100:.2f}%\")\nprint(f\"Recall       : {recall*100:.2f}%\")\nprint(f\"Specificity  : {specificity*100:.2f}%\")\nprint(f\"F1 Score     : {f1_score*100:.2f}%\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-14T01:29:14.843601Z","iopub.execute_input":"2026-01-14T01:29:14.844213Z","iopub.status.idle":"2026-01-14T01:30:11.838796Z","shell.execute_reply.started":"2026-01-14T01:29:14.844185Z","shell.execute_reply":"2026-01-14T01:30:11.838034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"MODEL_PATH = \"/kaggle/working/densenet_pneumonia.pth\"\n\ntorch.save(model.state_dict(), MODEL_PATH)\n\nprint(\"Model saved at:\", MODEL_PATH)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-14T01:36:08.045015Z","iopub.execute_input":"2026-01-14T01:36:08.045873Z","iopub.status.idle":"2026-01-14T01:36:08.140823Z","shell.execute_reply.started":"2026-01-14T01:36:08.045841Z","shell.execute_reply":"2026-01-14T01:36:08.140075Z"}},"outputs":[],"execution_count":null}]}