{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport copy\nimport time\nimport random\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom PIL import Image\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import Dataset, DataLoader\n\nimport torchvision\nfrom torchvision import models, transforms\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import (\n    classification_report,\n    confusion_matrix,\n    precision_score,\n    recall_score,\n    f1_score,\n    accuracy_score,\n    roc_auc_score\n)\n\nfrom tqdm.auto import tqdm\nimport pydicom\n\n\n# =========================\n# CONFIG\n# =========================\n\nRSNA_DIR = Path(\"/kaggle/input/competitions/rsna-pneumonia-detection-challenge\")\nIMAGE_DIR = RSNA_DIR / \"stage_2_train_images\"\nLABELS_CSV = RSNA_DIR / \"stage_2_train_labels.csv\"\n\nOUTPUT_DIR = Path(\"outputs\")\nOUTPUT_DIR.mkdir(exist_ok=True)\n\nBATCH_SIZE = 8\nNUM_EPOCHS = 5\nLR = 1e-4\nSTEP_SIZE = 2\nGAMMA = 0.1\nINPUT_SIZE = 224\n\nCLASS_NAMES = [\"Normal\", \"Viral Pneumonia\"]\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n\n# =========================\n# SEED\n# =========================\n\ndef set_seed(seed=42):\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n\nset_seed(42)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:32:06.377792Z","iopub.execute_input":"2026-05-06T19:32:06.378419Z","iopub.status.idle":"2026-05-06T19:32:06.386782Z","shell.execute_reply.started":"2026-05-06T19:32:06.378388Z","shell.execute_reply":"2026-05-06T19:32:06.386127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# LOAD RSNA DATASET\n# =========================\n\ndef load_rsna_dataframe():\n    df = pd.read_csv(LABELS_CSV)\n\n    df[\"path\"] = df[\"patientId\"].apply(lambda x: str(IMAGE_DIR / f\"{x}.dcm\"))\n    df[\"label\"] = df[\"Target\"].astype(int)\n    df[\"filename\"] = df[\"patientId\"] + \".dcm\"\n    df[\"class_name\"] = df[\"label\"].map({0: \"Normal\", 1: \"Viral Pneumonia\"})\n    df[\"study_id\"] = df[\"patientId\"]\n\n    df = df.drop_duplicates(subset=[\"patientId\"])\n\n    return df\n\n\nct_df = load_rsna_dataframe()\n\nprint(\"Total:\", len(ct_df))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:32:06.388263Z","iopub.execute_input":"2026-05-06T19:32:06.388571Z","iopub.status.idle":"2026-05-06T19:32:06.674063Z","shell.execute_reply.started":"2026-05-06T19:32:06.388548Z","shell.execute_reply":"2026-05-06T19:32:06.673296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# SPLIT (UNCHANGED)\n# =========================\n\nstudy_df = ct_df[[\"study_id\", \"label\"]].drop_duplicates()\n\ntrain_studies, temp = train_test_split(\n    study_df, test_size=0.2, stratify=study_df[\"label\"], random_state=42\n)\n\nval_studies, test_studies = train_test_split(\n    temp, test_size=0.5, stratify=temp[\"label\"], random_state=42\n)\n\ntrain_df = ct_df[ct_df[\"study_id\"].isin(train_studies[\"study_id\"])]\nval_df = ct_df[ct_df[\"study_id\"].isin(val_studies[\"study_id\"])]\ntest_df = ct_df[ct_df[\"study_id\"].isin(test_studies[\"study_id\"])]\n\n\n# =========================\n# DICOM LOADER\n# =========================\n\ndef load_dicom_as_rgb(path):\n    dcm = pydicom.dcmread(path)\n    img = dcm.pixel_array.astype(np.float32)\n\n    img = (img - img.min()) / (img.max() - img.min() + 1e-8)\n    img = (img * 255).astype(np.uint8)\n\n    img = np.stack([img] * 3, axis=-1)\n\n    return Image.fromarray(img)\n\n\n# =========================\n# DATASET\n# =========================\n\nclass RSNADataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df = df.reset_index(drop=True)\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        path = self.df.loc[idx, \"path\"]\n        label = int(self.df.loc[idx, \"label\"])\n\n        image = load_dicom_as_rgb(path)\n\n        if self.transform:\n            image = self.transform(image)\n\n        return image, label\n\n    def get_filename(self, idx):\n        return self.df.loc[idx, \"filename\"]\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:32:06.674892Z","iopub.execute_input":"2026-05-06T19:32:06.675185Z","iopub.status.idle":"2026-05-06T19:32:06.717947Z","shell.execute_reply.started":"2026-05-06T19:32:06.675161Z","shell.execute_reply":"2026-05-06T19:32:06.717136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# TRANSFORMS\n# =========================\n\ndata_transforms = {\n    \"train\": transforms.Compose([\n        transforms.Resize((INPUT_SIZE, INPUT_SIZE)),\n        transforms.RandomHorizontalFlip(),\n        transforms.RandomRotation(10),\n        transforms.ToTensor(),\n        transforms.Normalize([0.485,0.456,0.406],[0.229,0.224,0.225])\n    ]),\n    \"val\": transforms.Compose([\n        transforms.Resize((INPUT_SIZE, INPUT_SIZE)),\n        transforms.ToTensor(),\n        transforms.Normalize([0.485,0.456,0.406],[0.229,0.224,0.225])\n    ]),\n}\n\n\ntrain_dataset = RSNADataset(train_df, data_transforms[\"train\"])\nval_dataset = RSNADataset(val_df, data_transforms[\"val\"])\ntest_dataset = RSNADataset(test_df, data_transforms[\"val\"])\n\n\ndataloaders = {\n    \"train\": DataLoader(train_dataset, batch_size=BATCH_SIZE, shuffle=True),\n    \"val\": DataLoader(val_dataset, batch_size=BATCH_SIZE),\n    \"test\": DataLoader(test_dataset, batch_size=BATCH_SIZE),\n}\n\ndataset_sizes = {\n    \"train\": len(train_dataset),\n    \"val\": len(val_dataset),\n    \"test\": len(test_dataset),\n}\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:32:06.718901Z","iopub.execute_input":"2026-05-06T19:32:06.719262Z","iopub.status.idle":"2026-05-06T19:32:06.735721Z","shell.execute_reply.started":"2026-05-06T19:32:06.719239Z","shell.execute_reply":"2026-05-06T19:32:06.734906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# MODEL\n# =========================\n\ndef build_model():\n    model = models.resnet18(weights=models.ResNet18_Weights.DEFAULT)\n    model.fc = nn.Linear(model.fc.in_features, 2)\n    return model.to(DEVICE)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:32:06.737754Z","iopub.execute_input":"2026-05-06T19:32:06.738092Z","iopub.status.idle":"2026-05-06T19:32:06.743383Z","shell.execute_reply.started":"2026-05-06T19:32:06.738066Z","shell.execute_reply":"2026-05-06T19:32:06.742467Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# TRAIN\n# =========================\n\ndef train_model(model, criterion, optimizer, scheduler):\n    best_acc = 0\n    best_wts = copy.deepcopy(model.state_dict())\n\n    for epoch in range(NUM_EPOCHS):\n        print(f\"\\nEpoch {epoch+1}\")\n\n        for phase in [\"train\",\"val\"]:\n            model.train() if phase==\"train\" else model.eval()\n\n            running_loss, running_corrects = 0,0\n\n            for x,y in dataloaders[phase]:\n                x,y = x.to(DEVICE), y.to(DEVICE)\n                optimizer.zero_grad()\n\n                with torch.set_grad_enabled(phase==\"train\"):\n                    out = model(x)\n                    loss = criterion(out,y)\n                    _,pred = torch.max(out,1)\n\n                    if phase==\"train\":\n                        loss.backward()\n                        optimizer.step()\n\n                running_loss += loss.item()*x.size(0)\n                running_corrects += (pred==y).sum().item()\n\n            if phase==\"train\":\n                scheduler.step()\n\n            acc = running_corrects/dataset_sizes[phase]\n            print(f\"{phase} Acc: {acc:.4f}\")\n\n            if phase==\"val\" and acc>best_acc:\n                best_acc = acc\n                best_wts = copy.deepcopy(model.state_dict())\n\n    model.load_state_dict(best_wts)\n    return model\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:32:06.744680Z","iopub.execute_input":"2026-05-06T19:32:06.745041Z","iopub.status.idle":"2026-05-06T19:32:06.755649Z","shell.execute_reply.started":"2026-05-06T19:32:06.745019Z","shell.execute_reply":"2026-05-06T19:32:06.754870Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# EXPORT\n# =========================\n\ndef export(model):\n    model.eval()\n    rows = []\n\n    with torch.no_grad():\n        for i,(x,y) in enumerate(dataloaders[\"test\"]):\n            x = x.to(DEVICE)\n            out = model(x)\n            probs = torch.softmax(out,1)\n\n            for j in range(len(x)):\n                rows.append({\n                    \"prob_Normal\": probs[j][0].item(),\n                    \"prob_Pneumonia\": probs[j][1].item(),\n                    \"label\": y[j].item()\n                })\n\n    df = pd.DataFrame(rows)\n    df.to_csv(\"outputs/test.csv\", index=False)\n    return df\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:32:06.756531Z","iopub.execute_input":"2026-05-06T19:32:06.756907Z","iopub.status.idle":"2026-05-06T19:32:06.769352Z","shell.execute_reply.started":"2026-05-06T19:32:06.756875Z","shell.execute_reply":"2026-05-06T19:32:06.768464Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# RUN\n# =========================\n\nmodel = build_model()\ncriterion = nn.CrossEntropyLoss()\noptimizer = optim.Adam(model.parameters(), lr=LR)\nscheduler = lr_scheduler.StepLR(optimizer, STEP_SIZE, GAMMA)\n\nmodel = train_model(model, criterion, optimizer, scheduler)\n \ndf = export(model)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:32:06.770303Z","iopub.execute_input":"2026-05-06T19:32:06.770588Z","iopub.status.idle":"2026-05-06T20:15:53.753990Z","shell.execute_reply.started":"2026-05-06T19:32:06.770557Z","shell.execute_reply":"2026-05-06T20:15:53.752874Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# METRICS\n# =========================\n\npreds = np.argmax(df[[\"prob_Normal\",\"prob_Pneumonia\"]].values, axis=1)\nlabels = df[\"label\"].values\n\naccuracy = accuracy_score(labels, preds)\nprecision = precision_score(labels, preds, zero_division=0)\nrecall = recall_score(labels, preds, zero_division=0)\nf1 = f1_score(labels, preds)\nauc = roc_auc_score(labels, df[\"prob_Pneumonia\"])\n\nprint(f\"Accuracy : {accuracy:.4f}\")\nprint(f\"Precision: {precision:.4f}\")\nprint(f\"Recall   : {recall:.4f}\")\nprint(f\"F1       : {f1:.4f}\")\nprint(f\"AUC      : {auc:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T20:15:53.755202Z","iopub.execute_input":"2026-05-06T20:15:53.755433Z","iopub.status.idle":"2026-05-06T20:15:53.773649Z","shell.execute_reply.started":"2026-05-06T20:15:53.755411Z","shell.execute_reply":"2026-05-06T20:15:53.773068Z"}},"outputs":[],"execution_count":null}]}