{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"sourceType":"competition"}],"dockerImageVersionId":31240,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport pydicom\n\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import models, transforms\n\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-14T05:18:38.384598Z","iopub.execute_input":"2026-01-14T05:18:38.384865Z","iopub.status.idle":"2026-01-14T05:18:52.411114Z","shell.execute_reply.started":"2026-01-14T05:18:38.384843Z","shell.execute_reply":"2026-01-14T05:18:52.410548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/rsna-pneumonia-detection-challenge\"\nIMAGE_DIR = f\"{DATA_DIR}/stage_2_train_images\"\nLABEL_CSV = f\"{DATA_DIR}/stage_2_train_labels.csv\"\n\ndf = pd.read_csv(LABEL_CSV)\n\n# Target: 1 = Pneumonia, 0 = Non-pneumonia\ndf['label'] = df['Target']\n\n# One label per patient\ndf = df.groupby('patientId')['label'].max().reset_index()\n\nprint(\"Label distribution:\")\nprint(df['label'].value_counts())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-14T05:19:20.392933Z","iopub.execute_input":"2026-01-14T05:19:20.394038Z","iopub.status.idle":"2026-01-14T05:19:20.533209Z","shell.execute_reply.started":"2026-01-14T05:19:20.394008Z","shell.execute_reply":"2026-01-14T05:19:20.532380Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_transform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomRotation(10),\n    transforms.ColorJitter(brightness=0.2, contrast=0.2),\n    transforms.ToTensor(),\n    transforms.Lambda(lambda x: x.repeat(3, 1, 1)),  # Convert 1 channel → 3 channels\n    transforms.Normalize([0.5, 0.5, 0.5], [0.25, 0.25, 0.25])\n])\n\nval_transform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Lambda(lambda x: x.repeat(3, 1, 1)),  # Convert 1 channel → 3 channels\n    transforms.Normalize([0.5, 0.5, 0.5], [0.25, 0.25, 0.25])\n])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-14T05:20:11.278156Z","iopub.execute_input":"2026-01-14T05:20:11.278811Z","iopub.status.idle":"2026-01-14T05:20:11.284941Z","shell.execute_reply.started":"2026-01-14T05:20:11.278785Z","shell.execute_reply":"2026-01-14T05:20:11.284326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class RSNADataset(Dataset):\n    def __init__(self, df, image_dir, transform=None):\n        self.df = df.reset_index(drop=True)\n        self.image_dir = image_dir\n        self.transform = transform\n\n        # Preload all images in memory\n        self.images = []\n        for pid in tqdm(self.df['patientId'], desc=\"Loading DICOMs\"):\n            img_path = os.path.join(self.image_dir, pid + \".dcm\")\n            dicom_image = pydicom.dcmread(img_path)\n            image = dicom_image.pixel_array.astype(np.float32)\n\n            # Normalize to [0,1]\n            image = (image - image.min()) / (image.max() - image.min() + 1e-8)\n            image = Image.fromarray((image * 255).astype(np.uint8)).convert(\"L\")\n            self.images.append(image)\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        image = self.images[idx]\n        label = self.df.loc[idx, 'label']\n\n        if self.transform:\n            image = self.transform(image)\n\n        return image, torch.tensor(label, dtype=torch.long)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-14T05:20:38.709910Z","iopub.execute_input":"2026-01-14T05:20:38.710570Z","iopub.status.idle":"2026-01-14T05:20:38.716599Z","shell.execute_reply.started":"2026-01-14T05:20:38.710544Z","shell.execute_reply":"2026-01-14T05:20:38.715979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df, val_df = train_test_split(\n    df, test_size=0.2, stratify=df['label'], random_state=42\n)\n\ntrain_dataset = RSNADataset(train_df, IMAGE_DIR, train_transform)\nval_dataset = RSNADataset(val_df, IMAGE_DIR, val_transform)\n\ntrain_loader = DataLoader(train_dataset, batch_size=16, shuffle=True, num_workers=2)\nval_loader = DataLoader(val_dataset, batch_size=16, shuffle=False, num_workers=2)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-14T05:21:10.630507Z","iopub.execute_input":"2026-01-14T05:21:10.630802Z","iopub.status.idle":"2026-01-14T05:31:14.654891Z","shell.execute_reply.started":"2026-01-14T05:21:10.630780Z","shell.execute_reply":"2026-01-14T05:31:14.653747Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nmodel = models.densenet121(pretrained=True)\nmodel.classifier = nn.Linear(model.classifier.in_features, 1)\nmodel = model.to(device)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-14T05:32:52.943304Z","iopub.execute_input":"2026-01-14T05:32:52.943606Z","iopub.status.idle":"2026-01-14T05:32:53.209279Z","shell.execute_reply.started":"2026-01-14T05:32:52.943583Z","shell.execute_reply":"2026-01-14T05:32:53.208659Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pos = train_df['label'].sum()\nneg = len(train_df) - pos\npos_weight = torch.tensor([neg / pos]).to(device)\n\ncriterion = nn.BCEWithLogitsLoss(pos_weight=pos_weight)\noptimizer = torch.optim.Adam(model.parameters(), lr=1e-4)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-14T05:33:19.187504Z","iopub.execute_input":"2026-01-14T05:33:19.188250Z","iopub.status.idle":"2026-01-14T05:33:19.194722Z","shell.execute_reply.started":"2026-01-14T05:33:19.188223Z","shell.execute_reply":"2026-01-14T05:33:19.194068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"EPOCHS = 16\n\nfor epoch in range(EPOCHS):\n    model.train()\n    total_loss = 0\n\n    for images, labels in tqdm(train_loader, desc=f\"Epoch {epoch+1}/{EPOCHS}\"):\n        images = images.to(device)\n        labels = labels.float().to(device)\n\n        optimizer.zero_grad()\n        outputs = model(images).view(-1)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n\n        total_loss += loss.item()\n\n    print(f\"Epoch [{epoch+1}/{EPOCHS}] - Loss: {total_loss/len(train_loader):.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-14T05:33:52.716275Z","iopub.execute_input":"2026-01-14T05:33:52.716915Z","iopub.status.idle":"2026-01-14T06:37:07.369132Z","shell.execute_reply.started":"2026-01-14T05:33:52.716888Z","shell.execute_reply":"2026-01-14T06:37:07.368232Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.eval()\n\nTP = TN = FP = FN = 0\nthreshold = 0.35  # tuned for recall\n\nwith torch.no_grad():\n    for images, labels in val_loader:\n        images = images.to(device)\n        labels = labels.to(device)\n\n        outputs = model(images).view(-1)\n        probs = torch.sigmoid(outputs)\n        preds = (probs > threshold).long()\n\n        TP += ((preds == 1) & (labels == 1)).sum().item()\n        TN += ((preds == 0) & (labels == 0)).sum().item()\n        FP += ((preds == 1) & (labels == 0)).sum().item()\n        FN += ((preds == 0) & (labels == 1)).sum().item()\n\naccuracy = (TP + TN) / (TP + TN + FP + FN)\nprecision = TP / (TP + FP + 1e-8)\nrecall = TP / (TP + FN + 1e-8)\nspecificity = TN / (TN + FP + 1e-8)\nf1 = 2 * precision * recall / (precision + recall + 1e-8)\n\nprint(\"CONFUSION MATRIX\")\nprint(f\"TP: {TP}, FP: {FP}\")\nprint(f\"FN: {FN}, TN: {TN}\")\n\nprint(\"\\nMETRICS\")\nprint(f\"Accuracy    : {accuracy*100:.2f}%\")\nprint(f\"Precision   : {precision*100:.2f}%\")\nprint(f\"Recall      : {recall*100:.2f}%\")\nprint(f\"Specificity : {specificity*100:.2f}%\")\nprint(f\"F1-score    : {f1*100:.2f}%\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-14T06:37:42.792456Z","iopub.execute_input":"2026-01-14T06:37:42.793072Z","iopub.status.idle":"2026-01-14T06:38:02.720840Z","shell.execute_reply.started":"2026-01-14T06:37:42.793042Z","shell.execute_reply":"2026-01-14T06:38:02.719945Z"}},"outputs":[],"execution_count":null}]}