{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# PCam (PatchCamelyon dataset)\n## Breast Cancer Detection using histopathological images \n### The system depicted the microscope-mounted digital cameras or scanners used to obtain histopathological images.\n\n#### It’s a medical imaging dataset created from lymph node tissue slides.\n\n#### The task is to detect whether metastatic cancer (tumor spread) is present in the tissue patch or not\n\n#### Task specifically focused on detecting metastatic breast cancer in sentinel lymph node biopsies. \n#### When breast cancer spreads, the first place it usually goes is the sentinel lymph nodes (in the armpit area). Pathologists examine lymph node sections under the microscope to check for metastatic tumor tissue.","metadata":{}},{"cell_type":"markdown","source":"# Configurations","metadata":{}},{"cell_type":"code","source":"import warnings \nwarnings.filterwarnings('ignore')\n\nimport os\nimport cv2\n\nimport random\nimport numpy as np\nimport pandas as pd\n\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import models\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score, roc_curve\n\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\n\nimport matplotlib.pyplot as plt\n\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T14:54:18.444716Z","iopub.execute_input":"2025-09-01T14:54:18.445283Z","iopub.status.idle":"2025-09-01T14:54:18.450152Z","shell.execute_reply.started":"2025-09-01T14:54:18.445260Z","shell.execute_reply":"2025-09-01T14:54:18.449539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CFG:\n    IMG_SIZE = 224\n    BATCH_SIZE = 64\n    EPOCHS = 100\n    LEARNING_RATE = 1e-4\n    SEED = 42\n    DEVICE = \"cuda\" if torch.cuda.is_available() else \"cpu\"  # use \"cpu\" if no GPU\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T14:54:18.454216Z","iopub.execute_input":"2025-09-01T14:54:18.454427Z","iopub.status.idle":"2025-09-01T14:54:18.465148Z","shell.execute_reply.started":"2025-09-01T14:54:18.454408Z","shell.execute_reply":"2025-09-01T14:54:18.464421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set seed\ndef seed_everything(seed=42):\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n\nseed_everything(CFG.SEED)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T14:54:18.466176Z","iopub.execute_input":"2025-09-01T14:54:18.466438Z","iopub.status.idle":"2025-09-01T14:54:18.485201Z","shell.execute_reply.started":"2025-09-01T14:54:18.466417Z","shell.execute_reply":"2025-09-01T14:54:18.484641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Paths \nDATA_DIR = \"/kaggle/input/histopathologic-cancer-detection\"\nTRAIN_DIR = os.path.join(DATA_DIR, \"train\")\nLABELS = pd.read_csv(os.path.join(DATA_DIR, \"train_labels.csv\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T14:54:18.486306Z","iopub.execute_input":"2025-09-01T14:54:18.486566Z","iopub.status.idle":"2025-09-01T14:54:18.697705Z","shell.execute_reply.started":"2025-09-01T14:54:18.486544Z","shell.execute_reply":"2025-09-01T14:54:18.697093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cancer_samples = LABELS[LABELS['label'] == 1].sample(500, random_state=42)\nnormal_samples = LABELS[LABELS['label'] == 0].sample(500, random_state=42)\n\nLABELS = pd.concat([cancer_samples, normal_samples]).reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T14:54:18.698423Z","iopub.execute_input":"2025-09-01T14:54:18.698633Z","iopub.status.idle":"2025-09-01T14:54:18.723493Z","shell.execute_reply.started":"2025-09-01T14:54:18.698617Z","shell.execute_reply":"2025-09-01T14:54:18.722867Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 0 = No metastasis (normal tissue)\n\n### 1 = Metastasis present (Cancer tissue)","metadata":{}},{"cell_type":"code","source":"LABELS['label'].value_counts()[0], LABELS['label'].value_counts()[1] ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T14:54:18.724790Z","iopub.execute_input":"2025-09-01T14:54:18.725019Z","iopub.status.idle":"2025-09-01T14:54:18.731194Z","shell.execute_reply.started":"2025-09-01T14:54:18.725002Z","shell.execute_reply":"2025-09-01T14:54:18.730611Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Number of Normal Tissue samples : {LABELS[LABELS.label == 0].shape[0]}\")\nprint(f\"Number of Cancer Tissue samples : {LABELS[LABELS.label == 1].shape[0]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T14:54:18.732338Z","iopub.execute_input":"2025-09-01T14:54:18.732773Z","iopub.status.idle":"2025-09-01T14:54:18.752059Z","shell.execute_reply.started":"2025-09-01T14:54:18.732757Z","shell.execute_reply":"2025-09-01T14:54:18.751430Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Visualization","metadata":{}},{"cell_type":"code","source":"def show_samples(df, n=9):\n    samples = df.sample(n)\n    plt.figure(figsize=(8, 8))\n    for i, (idx, row) in enumerate(samples.iterrows()):\n        path = os.path.join(TRAIN_DIR, f\"{row['id']}.tif\")\n        img = cv2.imread(path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        plt.subplot(3, 3, i + 1)\n        plt.imshow(img)\n        plt.title(f\"Label: {row['label']}\")\n        plt.axis(\"off\")\n    plt.show()\n\nshow_samples(LABELS)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T14:54:18.752690Z","iopub.execute_input":"2025-09-01T14:54:18.752877Z","iopub.status.idle":"2025-09-01T14:54:19.310265Z","shell.execute_reply.started":"2025-09-01T14:54:18.752862Z","shell.execute_reply":"2025-09-01T14:54:19.309550Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Augmentations","metadata":{}},{"cell_type":"code","source":"train_transforms = A.Compose([\n    A.RandomResizedCrop(size=(CFG.IMG_SIZE, CFG.IMG_SIZE), scale=(0.8, 1.0), ratio=(0.9, 1.1), p=1.0),\n    A.HorizontalFlip(p=0.5),\n    A.VerticalFlip(p=0.5),\n    A.RandomBrightnessContrast(p=0.2),\n    A.Normalize(mean=(0.485, 0.456, 0.406), \n                std=(0.229, 0.224, 0.225)),\n    ToTensorV2(),\n])\n\nvalid_transforms = A.Compose([\n    A.Resize(height=CFG.IMG_SIZE, width=CFG.IMG_SIZE),\n    A.Normalize(mean=(0.485, 0.456, 0.406), \n                std=(0.229, 0.224, 0.225)),\n    ToTensorV2(),\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T14:54:19.311132Z","iopub.execute_input":"2025-09-01T14:54:19.311401Z","iopub.status.idle":"2025-09-01T14:54:19.322944Z","shell.execute_reply.started":"2025-09-01T14:54:19.311365Z","shell.execute_reply":"2025-09-01T14:54:19.322227Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Dataset class","metadata":{}},{"cell_type":"code","source":"class CancerDataset(Dataset):\n    def __init__(self, df, img_dir, transform=None):\n        self.df = df\n        self.img_dir = img_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img_path = os.path.join(self.img_dir, f\"{row['id']}.tif\")\n        image = cv2.imread(img_path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        label = row['label']\n        \n        if self.transform:\n            image = self.transform(image=image)[\"image\"]\n        \n        return image, torch.tensor(label, dtype=torch.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T14:54:19.323685Z","iopub.execute_input":"2025-09-01T14:54:19.323866Z","iopub.status.idle":"2025-09-01T14:54:19.341804Z","shell.execute_reply.started":"2025-09-01T14:54:19.323851Z","shell.execute_reply":"2025-09-01T14:54:19.341089Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df, valid_df = train_test_split(LABELS, test_size=0.2, stratify=LABELS['label'], random_state=CFG.SEED)\n\ntrain_dataset = CancerDataset(train_df, TRAIN_DIR, transform=train_transforms)\nvalid_dataset = CancerDataset(valid_df, TRAIN_DIR, transform=valid_transforms)\n\ntrain_loader = DataLoader(train_dataset, batch_size=CFG.BATCH_SIZE, shuffle=True, num_workers=2)\nvalid_loader = DataLoader(valid_dataset, batch_size=CFG.BATCH_SIZE, shuffle=False, num_workers=2)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T14:54:19.344108Z","iopub.execute_input":"2025-09-01T14:54:19.344313Z","iopub.status.idle":"2025-09-01T14:54:19.367040Z","shell.execute_reply.started":"2025-09-01T14:54:19.344298Z","shell.execute_reply":"2025-09-01T14:54:19.366314Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model architecture","metadata":{}},{"cell_type":"code","source":"class CancerClassifier(nn.Module):\n    def __init__(self, pretrained=True):\n        super().__init__()\n        self.backbone = models.resnet50(pretrained=pretrained)\n        in_features = self.backbone.fc.in_features\n        self.backbone.fc = nn.Linear(in_features, 1)\n    \n    def forward(self, x):\n        return self.backbone(x).squeeze(1)\n\nmodel = CancerClassifier().to(CFG.DEVICE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T14:54:19.367742Z","iopub.execute_input":"2025-09-01T14:54:19.367961Z","iopub.status.idle":"2025-09-01T14:54:19.820471Z","shell.execute_reply.started":"2025-09-01T14:54:19.367945Z","shell.execute_reply":"2025-09-01T14:54:19.819685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"criterion = nn.BCEWithLogitsLoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=CFG.LEARNING_RATE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T14:54:19.821297Z","iopub.execute_input":"2025-09-01T14:54:19.821552Z","iopub.status.idle":"2025-09-01T14:54:19.827053Z","shell.execute_reply.started":"2025-09-01T14:54:19.821526Z","shell.execute_reply":"2025-09-01T14:54:19.826517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_one_epoch(loader, model, optimizer, criterion):\n    model.train()\n    total_loss = 0\n    correct = 0\n    total = 0\n    for images, labels in loader:\n        images, labels = images.to(CFG.DEVICE), labels.to(CFG.DEVICE)\n        optimizer.zero_grad()\n        outputs = model(images)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        \n        total_loss += loss.item() * images.size(0)\n        preds = torch.sigmoid(outputs) > 0.5\n        correct += (preds == labels.int()).sum().item()\n        total += labels.size(0)\n    \n    return total_loss / total, correct / total","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T14:54:19.827899Z","iopub.execute_input":"2025-09-01T14:54:19.828072Z","iopub.status.idle":"2025-09-01T14:54:19.840814Z","shell.execute_reply.started":"2025-09-01T14:54:19.828058Z","shell.execute_reply":"2025-09-01T14:54:19.840033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def validate(loader, model, criterion):\n    model.eval()\n    total_loss = 0\n    correct = 0\n    total = 0\n    with torch.no_grad():\n        for images, labels in loader:\n            images, labels = images.to(CFG.DEVICE), labels.to(CFG.DEVICE)\n            outputs = model(images)\n            loss = criterion(outputs, labels)\n            \n            total_loss += loss.item() * images.size(0)\n            preds = torch.sigmoid(outputs) > 0.5\n            correct += (preds == labels.int()).sum().item()\n            total += labels.size(0)\n    \n    return total_loss / total, correct / total","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T14:54:19.841554Z","iopub.execute_input":"2025-09-01T14:54:19.841863Z","iopub.status.idle":"2025-09-01T14:54:19.860194Z","shell.execute_reply.started":"2025-09-01T14:54:19.841846Z","shell.execute_reply":"2025-09-01T14:54:19.859432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_losses, val_losses, train_accs, val_accs = [], [], [], []\n\nfor epoch in tqdm(range(CFG.EPOCHS)):\n    train_loss, train_acc = train_one_epoch(train_loader, model, optimizer, criterion)\n    val_loss, val_acc = validate(valid_loader, model, criterion)\n\n    train_losses.append(train_loss)\n    val_losses.append(val_loss)\n    train_accs.append(train_acc)\n    val_accs.append(val_acc)\n\n    if epoch%10 == 0:\n        print(f\"Epoch {epoch+1}/{CFG.EPOCHS} - \"\n           f\"Train Loss: {train_loss:.4f}, Train Acc: {train_acc:.4f} | \"\n           f\"Val Loss: {val_loss:.4f}, Val Acc: {val_acc:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T14:54:19.861128Z","iopub.execute_input":"2025-09-01T14:54:19.861352Z","iopub.status.idle":"2025-09-01T15:02:51.829804Z","shell.execute_reply.started":"2025-09-01T14:54:19.861331Z","shell.execute_reply":"2025-09-01T15:02:51.828939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10,4))\nplt.subplot(1,2,1)\nplt.plot(train_losses, label=\"Train\")\nplt.plot(val_losses, label=\"Valid\")\nplt.title(\"Loss\")\nplt.legend()\n\nplt.subplot(1,2,2)\nplt.plot(train_accs, label=\"Train\")\nplt.plot(val_accs, label=\"Valid\")\nplt.title(\"Accuracy\")\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T15:02:51.831067Z","iopub.execute_input":"2025-09-01T15:02:51.831620Z","iopub.status.idle":"2025-09-01T15:02:52.130726Z","shell.execute_reply.started":"2025-09-01T15:02:51.831596Z","shell.execute_reply":"2025-09-01T15:02:52.130049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"torch.save(model.state_dict(), \"cancer_detector.pth\")\n\n# Inference\ndef predict_single(img_path, model, transform):\n    model.eval()\n    image = cv2.imread(img_path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    image = transform(image=image)[\"image\"].unsqueeze(0).to(CFG.DEVICE)\n    with torch.no_grad():\n        output = model(image)\n        prob = torch.sigmoid(output).item()\n    return prob, int(prob > 0.5)\n\ntest_path = os.path.join(TRAIN_DIR, f\"{valid_df.iloc[0]['id']}.tif\")\nprob, pred = predict_single(test_path, model, valid_transforms)\nprint(f\"Prediction: {pred}, Probability: {prob:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T15:02:52.131820Z","iopub.execute_input":"2025-09-01T15:02:52.132289Z","iopub.status.idle":"2025-09-01T15:02:52.398751Z","shell.execute_reply.started":"2025-09-01T15:02:52.132270Z","shell.execute_reply":"2025-09-01T15:02:52.398002Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"valid_df.iloc[0]['label']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T15:14:13.675276Z","iopub.execute_input":"2025-09-01T15:14:13.676003Z","iopub.status.idle":"2025-09-01T15:14:13.681089Z","shell.execute_reply.started":"2025-09-01T15:14:13.675979Z","shell.execute_reply":"2025-09-01T15:14:13.680415Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"just experiments ","metadata":{}},{"cell_type":"code","source":"# def validate(loader, model, criterion):\n#     model.eval()\n#     total_loss = 0\n#     correct = 0\n#     total = 0\n\n#     all_labels = []\n#     all_probs = []\n\n#     with torch.no_grad():\n#         for images, labels in loader:\n#             images, labels = images.to(CFG.DEVICE), labels.to(CFG.DEVICE)\n#             outputs = model(images)\n\n#             # loss\n#             loss = criterion(outputs, labels)\n#             total_loss += loss.item() * images.size(0)\n\n#             # probabilities\n#             probs = torch.sigmoid(outputs).squeeze()\n#             preds = (probs > 0.5).long()\n\n#             # accuracy\n#             correct += (preds == labels.long()).sum().item()\n#             total += labels.size(0)\n\n#             # collect for AUC\n#             all_labels.extend(labels.cpu().numpy())\n#             all_probs.extend(probs.cpu().numpy())\n\n#     avg_loss = total_loss / total\n#     accuracy = correct / total\n\n#     # ROC-AUC\n#     try:\n#         auc = roc_auc_score(all_labels, all_probs)\n#     except ValueError:\n#         auc = float(\"nan\")\n\n#     return avg_loss, accuracy, auc, all_labels, all_probs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T15:02:52.399815Z","iopub.execute_input":"2025-09-01T15:02:52.400112Z","iopub.status.idle":"2025-09-01T15:02:52.403888Z","shell.execute_reply.started":"2025-09-01T15:02:52.400088Z","shell.execute_reply":"2025-09-01T15:02:52.403144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # After training, run validation once\n# val_loss, val_acc, val_auc, all_labels, all_probs = validate(val_loader, model, criterion)\n\n# print(f\"Validation Loss: {val_loss:.4f}\")\n# print(f\"Validation Accuracy: {val_acc:.4f}\")\n# print(f\"Validation AUC: {val_auc:.4f}\")\n\n# # Plot ROC Curve\n# fpr, tpr, _ = roc_curve(all_labels, all_probs)\n# plt.figure(figsize=(6, 6))\n# plt.plot(fpr, tpr, label=f\"AUC = {val_auc:.4f}\")\n# plt.plot([0, 1], [0, 1], \"k--\")\n# plt.xlabel(\"False Positive Rate\")\n# plt.ylabel(\"True Positive Rate\")\n# plt.title(\"ROC Curve\")\n# plt.legend(loc=\"lower right\")\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-01T15:02:52.404554Z","iopub.execute_input":"2025-09-01T15:02:52.404793Z","iopub.status.idle":"2025-09-01T15:02:52.425038Z","shell.execute_reply.started":"2025-09-01T15:02:52.404771Z","shell.execute_reply":"2025-09-01T15:02:52.424265Z"}},"outputs":[],"execution_count":null}]}