{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-01T21:10:24.773473Z","iopub.execute_input":"2025-06-01T21:10:24.773656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─── Cell 1: verify that Kaggle has mounted everything for you ─────────────────────────\nimport os\n\nINPUT_DIR = \"/kaggle/input/histopathologic-cancer-detection\"\nprint(\"Contents of /kaggle/input/histopathologic-cancer-detection/:\")\nfor fname in sorted(os.listdir(INPUT_DIR)):\n    print(\"  \", fname)\n\n# You should see something like:\n#    sample_submission.csv\n#    test\n#    train\n#    train_labels.csv\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T21:21:08.377342Z","iopub.execute_input":"2025-06-01T21:21:08.377962Z","iopub.status.idle":"2025-06-01T21:21:08.384336Z","shell.execute_reply.started":"2025-06-01T21:21:08.377918Z","shell.execute_reply":"2025-06-01T21:21:08.383516Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─── Cell 2: read the CSV of training labels ───────────────────────────────────────────\nimport pandas as pd\n\nlabels_df = pd.read_csv(os.path.join(INPUT_DIR, \"train_labels.csv\"))\nprint(\"Total train_labels rows:\", len(labels_df))\nprint(labels_df[\"label\"].value_counts())\nlabels_df.head()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─── Cell 3: inspect “train/” folder structure and show a few random images ─────────────\nimport random\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom PIL import Image\n\nTRAIN_DIR = os.path.join(INPUT_DIR, \"train\")\nall_train_files = [f for f in os.listdir(TRAIN_DIR) if f.lower().endswith(\".tif\")]\nprint(\"Number of .tif files in train/:\", len(all_train_files))\nprint(\"Example filenames:\", all_train_files[:5])\n\n# Define a small helper to load a patch directly from train/\ndef load_patch(train_dir, patch_id):\n    \"\"\"\n    patch_id is the 40-char string (without “.tif”). \n    We assume train/ contains exactly files named \"<patch_id>.tif\".\n    \"\"\"\n    full_path = os.path.join(train_dir, patch_id + \".tif\")\n    img = Image.open(full_path).convert(\"RGB\")\n    return img\n\n# Display 4 random train patches (2 positives, 2 negatives)\nplt.figure(figsize=(8, 8))\nfor i in range(4):\n    # pick a random row in labels_df \n    idx = random.randint(0, len(labels_df) - 1)\n    pid = labels_df.loc[idx, \"id\"]\n    lab = labels_df.loc[idx, \"label\"]\n    patch_img = load_patch(TRAIN_DIR, pid)\n    ax = plt.subplot(2, 2, i+1)\n    ax.imshow(patch_img)\n    ax.set_title(f\"id={pid[:8]}…  label={lab}\")\n    ax.axis(\"off\")\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─── Cell 4: same for “test/”—peek at a few test images (unlabeled) ───────────────────\nTEST_DIR = os.path.join(INPUT_DIR, \"test\")\nall_test_files = [f for f in os.listdir(TEST_DIR) if f.lower().endswith(\".tif\")]\nprint(\"Number of .tif files in test/:\", len(all_test_files))\nprint(\"First 5 test filenames:\", all_test_files[:5])\n\n# Display 4 random “test” patches (though unlabeled, just for sanity check)\nplt.figure(figsize=(8, 8))\nfor i in range(4):\n    tfn = random.choice(all_test_files)\n    img = Image.open(os.path.join(TEST_DIR, tfn)).convert(\"RGB\")\n    ax = plt.subplot(2, 2, i+1)\n    ax.imshow(img)\n    ax.set_title(f\"test/{tfn[:8]}…\")\n    ax.axis(\"off\")\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─── Cell 5: Quick class‐balance bar chart for the train set ────────────────────────────\ncounts = labels_df[\"label\"].value_counts()\nplt.figure(figsize=(4, 4))\nplt.bar([\"Non-Metastasis (0)\", \"Metastasis (1)\"], counts.values, color=[\"steelblue\",\"crimson\"])\nplt.ylabel(\"Count of Patches\")\nplt.title(\"Class Distribution in the 220k-patch Train Set\")\nfor i, v in enumerate(counts.values):\n    plt.text(i, v + 2000, str(v), ha=\"center\")\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─── Cell 6: build a PyTorch Dataset/output DataLoader (since \"train/\" is already unzipped)\n\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\n\nclass HistopathFolderDataset(Dataset):\n    def __init__(self, labels_df, train_folder, transform=None):\n        super().__init__()\n        self.df = labels_df.reset_index(drop=True)\n        self.train_folder = train_folder\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        patch_id = row[\"id\"]\n        label = torch.tensor(row[\"label\"], dtype=torch.float32)\n        img_path = os.path.join(self.train_folder, patch_id + \".tif\")\n        img = Image.open(img_path).convert(\"RGB\")\n        if self.transform:\n            img = self.transform(img)\n        return img, label\n\n# (A) compute approximate per-channel mean/std on a small random subset:\nimport numpy as np\n\nnsamp = 2000\nrng = np.random.default_rng(seed=42)\nindices = rng.choice(len(labels_df), size=nsamp, replace=False)\n\nmeans = []\nstds = []\nfor i in indices:\n    pid = labels_df.loc[i, \"id\"]\n    arr = np.array(Image.open(os.path.join(TRAIN_DIR, pid + \".tif\")).convert(\"RGB\")).astype(np.float32) / 255.0\n    means.append(arr.mean(axis=(0,1)))\n    stds.append(arr.std(axis=(0,1)))\nmeans = np.vstack(means)\nstds  = np.vstack(stds)\nglobal_mean = means.mean(axis=0).tolist()\nglobal_std  = stds.mean(axis=0).tolist()\nprint(\"Global mean:\", global_mean)\nprint(\"Global std: \", global_std)\n\n# (B) Stratified Train/Val split\nfrom sklearn.model_selection import train_test_split\ntrain_df, val_df = train_test_split(labels_df, test_size=0.20, stratify=labels_df[\"label\"], random_state=42)\n\n# (C) Define transforms\ntrain_transforms = transforms.Compose([\n    transforms.Resize((96,96)),\n    transforms.RandomHorizontalFlip(),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=global_mean, std=global_std),\n])\nval_transforms = transforms.Compose([\n    transforms.Resize((96,96)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=global_mean, std=global_std),\n])\n\n# (D) Create Datasets\ntrain_ds = HistopathFolderDataset(train_df, TRAIN_DIR, transform=train_transforms)\nval_ds   = HistopathFolderDataset(val_df,   TRAIN_DIR, transform=val_transforms)\n\n# (E) DataLoaders\ntrain_loader = DataLoader(train_ds, batch_size=64, shuffle=True,  num_workers=2, pin_memory=True)\nval_loader   = DataLoader(val_ds,   batch_size=64, shuffle=False, num_workers=2, pin_memory=True)\n\nprint(\"Train batches per epoch:\", len(train_loader))\nprint(\"Val   batches per epoch:\", len(val_loader))\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─── Cell 7: Quick EDA / Patch Visualization (using .iloc and skipping missing files) ────────────────────────\n\nimport random\nimport matplotlib.pyplot as plt\nfrom PIL import Image\n\nplt.figure(figsize=(10,4))\n\nshown = 0\nattempts = 0\nmax_attempts = 30    # stop if we can’t find 6 valid files after 30 tries\n\nwhile shown < 6 and attempts < max_attempts:\n    attempts += 1\n    pos = random.randint(0, len(train_df) - 1)\n    # use iloc[] instead of loc[]\n    pid = train_df.iloc[pos][\"id\"]\n    lbl = train_df.iloc[pos][\"label\"]\n    img_path = os.path.join(TRAIN_DIR, pid + \".tif\")\n\n    if not os.path.isfile(img_path):\n        # file is missing—skip this one\n        continue\n\n    try:\n        img = Image.open(img_path).convert(\"RGB\")\n    except Exception:\n        # PIL couldn’t read it—skip and try again\n        continue\n\n    ax = plt.subplot(2, 3, shown + 1)\n    ax.imshow(img)\n    ax.set_title(f\"Label = {lbl}\")\n    ax.axis(\"off\")\n\n    shown += 1\n\nif shown < 6:\n    print(f\"Only {shown} valid patches were found after {attempts} attempts.\")\n\nplt.suptitle(\"Six Random Training Patches\", fontsize=16)\nplt.tight_layout(rect=[0,0,1,0.9])\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─── Cell 8: Define a Simple CNN Model ─────────────────────────────────────────\nimport torch.nn as nn\nimport torch.optim as optim\n\nclass TinyCNN(nn.Module):\n    def __init__(self):\n        super(TinyCNN, self).__init__()\n        self.features = nn.Sequential(\n            nn.Conv2d(3, 16, kernel_size=3, stride=1, padding=1),  # → 16×96×96\n            nn.ReLU(inplace=True),\n            nn.MaxPool2d(2,2),                                       # → 16×48×48\n\n            nn.Conv2d(16, 32, kernel_size=3, stride=1, padding=1),  # → 32×48×48\n            nn.ReLU(inplace=True),\n            nn.MaxPool2d(2,2),                                       # → 32×24×24\n        )\n        self.classifier = nn.Sequential(\n            nn.Flatten(),                                            # → 32×24×24 = 18432\n            nn.Linear(32*24*24, 64),\n            nn.ReLU(inplace=True),\n            nn.Dropout(0.5),\n            nn.Linear(64, 1)   # output raw logit\n        )\n\n    def forward(self, x):\n        x = self.features(x)\n        x = self.classifier(x)\n        return x\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = TinyCNN().to(device)\nprint(\"Model architecture:\", model)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# ─── Cell 9: Loss, Optimizer, and a Mini‐Batch AUC Helper (on small subsets) ─────\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = optim.Adam(model.parameters(), lr=1e-4)\n\nfrom sklearn.metrics import roc_auc_score\n\ndef compute_subset_auc(model, loader, device, subset_size=512):\n    \"\"\"\n    Runs inference on at most `subset_size` samples (not the full loader)\n    to get a quick AUC estimate.\n    \"\"\"\n    model.eval()\n    collected_logits = []\n    collected_labels = []\n    seen = 0\n\n    with torch.no_grad():\n        for images, labels in loader:\n            b = images.size(0)\n            if seen + b > subset_size:\n                take = subset_size - seen\n                images = images[:take]\n                labels = labels[:take]\n                b = take\n\n            images = images.to(device)\n            labels = labels.to(device).view(-1)\n            logits = model(images).squeeze(1).cpu().numpy()\n            collected_logits.append(logits)\n            collected_labels.append(labels.cpu().numpy())\n            seen += b\n\n            if seen >= subset_size:\n                break\n\n    all_logits = np.concatenate(collected_logits, axis=0)\n    all_labels = np.concatenate(collected_labels, axis=0)\n    probs = 1.0 / (1.0 + np.exp(-all_logits))\n    auc = roc_auc_score(all_labels, probs)\n    return auc\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─── Cell 10: Very‐Quick Training Loop (1 epoch, small subset for AUC) ───────────\nnum_epochs = 1    # run for just one epoch—super fast\nbest_val_auc = 0.0\n\nfor epoch in range(1, num_epochs+1):\n    print(f\"\\n→ Starting Epoch {epoch}/{num_epochs}\")\n    model.train()\n    running_loss = 0.0\n    for batch_idx, (images, labels) in enumerate(train_loader, start=1):\n        images = images.to(device)\n        labels = labels.to(device).view(-1,1)\n\n        optimizer.zero_grad()\n        outputs = model(images)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n\n        running_loss += loss.item() * images.size(0)\n\n        # Print a status every 200 batches\n        if batch_idx % 200 == 0:\n            print(f\"   Batch {batch_idx}/{len(train_loader)} →  loss {loss.item():.4f}\")\n\n        # If you only want to do a super‐small “sanity check”:\n        if batch_idx >= 500:      # only train on first 500 batches\n            break\n\n    epoch_loss = running_loss / (500 * images.size(0))\n    print(f\"→ Epoch {epoch} done. Avg. loss (on first 500 batches): {epoch_loss:.4f}\")\n\n    # Compute a quick train‐subset AUC\n    train_auc = compute_subset_auc(model, train_loader, device, subset_size=512)\n    # Compute a quick val‐subset AUC\n    val_auc   = compute_subset_auc(model, val_loader,   device, subset_size=512)\n\n    print(f\"→ Train‐subset AUC ≈ {train_auc:.4f},  Val‐subset AUC ≈ {val_auc:.4f}\")\n    best_val_auc = max(best_val_auc, val_auc)\n\nprint(f\"\\n===== Finished quick 1‐epoch pass. Best val‐subset AUC ≈ {best_val_auc:.4f} =====\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}