{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom PIL import Image  # Using PIL for image loading \nimport os\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport seaborn as sns\nfrom torchvision import transforms, models, transforms\nfrom torch.utils.data import Dataset, DataLoader\nfrom sklearn.model_selection import train_test_split  # For stratified sampling\nfrom sklearn.metrics import roc_auc_score, accuracy_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T07:24:36.831755Z","iopub.execute_input":"2025-12-02T07:24:36.832035Z","iopub.status.idle":"2025-12-02T07:24:40.613608Z","shell.execute_reply.started":"2025-12-02T07:24:36.832014Z","shell.execute_reply":"2025-12-02T07:24:40.612983Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Load the train_labels.csv\n#labels_path = 'train_labels.csv'\nlabels_path = r'/kaggle/input/histopathologic-cancer-detection/train_labels.csv'\ndf = pd.read_csv(labels_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T07:24:40.614926Z","iopub.execute_input":"2025-12-02T07:24:40.615402Z","iopub.status.idle":"2025-12-02T07:24:40.825838Z","shell.execute_reply.started":"2025-12-02T07:24:40.615384Z","shell.execute_reply":"2025-12-02T07:24:40.825242Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sample 5% of the data, stratified by label for balance\nsample_fraction = 0.99999 # full set of data or sample set e.g.,0.001\ndf_sample, _ = train_test_split(df, test_size=1 - sample_fraction, stratify=df['label'], random_state=42)\n\n# Print basic info on the sample\nprint(\"Full DataFrame Shape:\", df.shape)  # For reference: (220025, 2)\nprint(\"Sampled DataFrame Shape\", sample_fraction, \"%:\", df_sample.shape)  # ~ (11001, 2)\nprint(\"\\nHead of Sampled DataFrame:\")\nprint(df_sample.head())\nprint(\"\\nClass Distribution in Sample:\")\nprint(df_sample['label'].value_counts(normalize=True))  # Should mirror full ~0: 0.6, 1: 0.4","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T07:24:40.826536Z","iopub.execute_input":"2025-12-02T07:24:40.826727Z","iopub.status.idle":"2025-12-02T07:24:40.921311Z","shell.execute_reply.started":"2025-12-02T07:24:40.826712Z","shell.execute_reply":"2025-12-02T07:24:40.920552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_sample.head(15)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T07:24:40.922678Z","iopub.execute_input":"2025-12-02T07:24:40.922908Z","iopub.status.idle":"2025-12-02T07:24:40.931911Z","shell.execute_reply.started":"2025-12-02T07:24:40.92289Z","shell.execute_reply":"2025-12-02T07:24:40.931272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sample a few images for exploration from the subsample\n# To handle large dataset, we'll sample N images (e.g., 5) from each class in the subsample\nN = 5  # Number of samples per class for display; adjust as needed\ntrain_dir = r'/kaggle/input/histopathologic-cancer-detection/train/'\ndef file_exists(id):\n    img_path = os.path.join(train_dir, id + '.tif')\n    return os.path.exists(img_path)\n\nprint(f\"Original subsample size: {len(df_sample)}\")\ndf_sample = df_sample[df_sample['id'].apply(file_exists)]\nprint(f\"Filtered subsample size (valid files only): {len(df_sample)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T07:24:40.93257Z","iopub.execute_input":"2025-12-02T07:24:40.932758Z","iopub.status.idle":"2025-12-02T07:28:04.953666Z","shell.execute_reply.started":"2025-12-02T07:24:40.932743Z","shell.execute_reply":"2025-12-02T07:28:04.952879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sample IDs from each class in the subsample\ncancer_ids = df_sample[df_sample['label'] == 1]['id'].sample(N).values\nnon_cancer_ids = df_sample[df_sample['label'] == 0]['id'].sample(N).values\n\n# Function to load and display image\ndef load_and_show_image(img_id, label):\n    img_path = os.path.join(train_dir, img_id + '.tif')\n    if not os.path.exists(img_path):\n        print(f\"Image not found: {img_path}\")\n        return None\n    img = Image.open(img_path)\n    plt.figure(figsize=(3, 2))\n    plt.imshow(img)\n    plt.title(f\"Label: {label} (ID: {img_id})\")\n    plt.axis('off')\n    plt.show()\n    # Return numpy array for further analysis or to share sample\n    img_array = np.array(img)\n    print(f\"Image Shape: {img_array.shape}\")  # Expected: (96, 96, 3)\n    print(f\"Sample Pixel Values (first 5x5 of channel 0):\\n{img_array[:5, :5, 0]}\")  # Snippet to share\n    return img_array\n\n# Display samples\nprint(\"\\nCancer Samples (from\", sample_fraction, \"% subsample):\")\ncancer_samples = [load_and_show_image(cid, 1) for cid in cancer_ids if load_and_show_image(cid, 1) is not None]\n\nprint(\"\\nNon-Cancers (from\", sample_fraction, \"% subsample):\")\nnon_cancer_samples = [load_and_show_image(ncid, 0) for ncid in non_cancer_ids if load_and_show_image(ncid, 0) is not None]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T07:28:04.954787Z","iopub.execute_input":"2025-12-02T07:28:04.955052Z","iopub.status.idle":"2025-12-02T07:28:06.402206Z","shell.execute_reply.started":"2025-12-02T07:28:04.955022Z","shell.execute_reply":"2025-12-02T07:28:06.40129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compute mean and std for normalization (on a smaller subsample to speed up local testing)\n# Use sample images for quick estimate; use full 10% for now\n\n#STATS_SAMPLE_SIZE = 1000  # Adjust up on Kaggle; keeps local run fast\n#stats_df = df_sample.sample(STATS_SAMPLE_SIZE)  # From your existing filtered df_sample\n\nsample_ids = df_sample['id'].values\n# Custom Dataset for efficient loading with transforms (assuming this is defined; include if not)\nclass SampleDataset(Dataset):\n    def __init__(self, img_ids, img_dir, transform=None):\n        self.img_ids = img_ids\n        self.img_dir = img_dir\n        self.transform = transform or transforms.Compose([\n            transforms.ToTensor()  # Converts to tensor and normalizes to [0,1]\n        ])\n\n    def __len__(self):\n        return len(self.img_ids)\n\n    def __getitem__(self, idx):\n        img_path = os.path.join(self.img_dir, self.img_ids[idx] + '.tif')\n        try:\n            img = Image.open(img_path).convert('RGB')\n            return self.transform(img)\n        except Exception as e:\n            print(f\"Error loading {img_path}: {e}\")\n            return torch.zeros(3, 96, 96)  # Dummy tensor to skip bad files without crashing\n\n# Create DataLoader (batch_size=32; num_workers=0 to fix multiprocessing crash)\ndataset = SampleDataset(sample_ids, train_dir)\nloader = DataLoader(dataset, batch_size=32, shuffle=False, num_workers=0)  # Critical fix: num_workers=0\n\n# Compute mean and std with progress bar (if tqdm not installed, pip install tqdm or remove)\nfrom tqdm import tqdm  # Optional for progress\n\nmeans = []\nstds = []\nfor batch in tqdm(loader, desc=\"Computing stats\"):  # Shows progress\n    if batch.numel() == 0:  # Skip empty batches from errors\n        continue\n    means.append(batch.mean([0, 2, 3]).numpy())\n    stds.append(batch.std([0, 2, 3]).numpy())\n\nif means:  # Avoid error if all failed\n    mean = np.mean(means, axis=0)\n    std = np.mean(stds, axis=0)\n    print(\"\\nEstimated Mean (RGB) from subsample:\", mean)\n    print(\"Estimated Std (RGB) from subsample:\", std)\nelse:\n    print(\"No valid images loaded; check paths/files.\")\n\n# Optional: Histogram of pixel intensities (from displayed samples only, to keep quick)\nall_pixels = np.concatenate([img.reshape(-1, 3) for img in cancer_samples + non_cancer_samples if img is not None], axis=0)\nplt.figure(figsize=(10, 5))\nfor c in range(3):\n    plt.hist(all_pixels[:, c], bins=50, alpha=0.5, label=f'Channel {c}')\nplt.title('Pixel Intensity Histogram (Displayed Samples from 5% Subsample)')\nplt.xlabel('Intensity')\nplt.ylabel('Frequency')\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T07:30:36.257758Z","iopub.execute_input":"2025-12-02T07:30:36.258047Z","iopub.status.idle":"2025-12-02T07:47:26.390768Z","shell.execute_reply.started":"2025-12-02T07:30:36.258024Z","shell.execute_reply":"2025-12-02T07:47:26.390022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Inspect Data Structure and Cleaning\nprint(\"Full Dataset Summary:\")\nprint(df.info())  # No nulls expected\nprint(\"\\nCheck for Duplicates:\", df['id'].duplicated().sum())  # Should be 0\nprint(\"Class Balance (Full):\")\nprint(df['label'].value_counts(normalize=True))  # ~0: 0.595, 1: 0.405\n\n# Cleaning: No major issues; describe in markdown. If any nulls (unlikely), drop: df.dropna()\n\n# 2. Visualize Class Distribution\nplt.figure(figsize=(3, 2))\nsns.countplot(x='label', data=df)  # Use full df for accurate balance\nplt.title('Class Distribution (0: No Cancer, 1: Cancer)')\nplt.xlabel('Label')\nplt.ylabel('Count')\nplt.show()\n\n# Table for summary (use for report quality)\nclass_summary = df['label'].value_counts().reset_index()\nclass_summary.columns = ['Label', 'Count']\nclass_summary['Percentage'] = (class_summary['Count'] / len(df)) * 100\nprint(\"\\nClass Summary Table:\")\nprint(class_summary)\n\n# 3. Visualize More Sample Images (Grid for better overview)\ndef display_image_grid(ids, labels, title):\n    fig, axes = plt.subplots(2, 5, figsize=(15, 6))  # 10 samples\n    axes = axes.flatten()\n    for i, (img_id, label) in enumerate(zip(ids, labels)):\n        img_path = os.path.join(train_dir, img_id + '.tif')\n        if os.path.exists(img_path):\n            img = Image.open(img_path)\n            axes[i].imshow(img)\n            axes[i].set_title(f\"Label: {label}\")\n            axes[i].axis('off')\n    plt.suptitle(title)\n    plt.tight_layout()\n    plt.show()\n\n# Sample 10 per class from filtered subsample\ncancer_ids_eda = df_sample[df_sample['label'] == 1]['id'].sample(min(10, len(df_sample[df_sample['label'] == 1]))).values\nnon_cancer_ids_eda = df_sample[df_sample['label'] == 0]['id'].sample(min(10, len(df_sample[df_sample['label'] == 0]))).values\n\ndisplay_image_grid(cancer_ids_eda, [1]*len(cancer_ids_eda), \"Cancer Samples\")\ndisplay_image_grid(non_cancer_ids_eda, [0]*len(non_cancer_ids_eda), \"Non-Cancer Samples\")\n\n# 4. Pixel Intensity Histograms per Class (using larger sample for depth)\n# Reuse your mean/std computation loader, but split by class for comparison\ncancer_df = df_sample[df_sample['label'] == 1].sample(min(500, len(df_sample[df_sample['label'] == 1])))  # Subset for speed\nnon_cancer_df = df_sample[df_sample['label'] == 0].sample(min(500, len(df_sample[df_sample['label'] == 0])))\n\ndef compute_class_data(df_subset):\n    dataset = SampleDataset(df_subset['id'].values, train_dir)\n    loader = DataLoader(dataset, batch_size=32, shuffle=False, num_workers=0)\n    all_images = []\n    for batch in loader:\n        if batch.numel() > 0 and batch.shape[1] == 3:  # Ensure valid image batch\n            all_images.append(batch)\n    if not all_images:\n        return np.array([]), np.array([])\n    stacked_images = torch.cat(all_images, dim=0)  # (num_images, 3, H, W)\n    flattened_pixels = stacked_images.view(-1, 3).numpy()  # For histogram\n    return flattened_pixels, stacked_images.numpy()  # Return both for hist and avg\n\ncancer_pixels, cancer_images = compute_class_data(cancer_df)\nnon_cancer_pixels, non_cancer_images = compute_class_data(non_cancer_df)\n\n# Plot histograms per class\nfig, axes = plt.subplots(2, 1, figsize=(10, 8))\nfor c in range(3):\n    axes[0].hist(cancer_pixels[:, c], bins=50, alpha=0.5, label=f'Channel {c}', density=True)\naxes[0].set_title('Pixel Intensity Histogram - Cancer Class')\naxes[0].legend()\n\nfor c in range(3):\n    axes[1].hist(non_cancer_pixels[:, c], bins=50, alpha=0.5, label=f'Channel {c}', density=True)\naxes[1].set_title('Pixel Intensity Histogram - Non-Cancer Class')\naxes[1].legend()\nplt.tight_layout()\nplt.show()\n\n# 5. Additional Insights (e.g., Average Image per Class - optional for depth)\ndef average_image(images):\n    if len(images) == 0:\n        return None\n    return np.mean(images, axis=0).transpose(1, 2, 0)  # Mean over images (C,H,W) -> (H,W,C)\n\nif len(cancer_images) > 0:\n    avg_cancer = average_image(cancer_images)\n    if avg_cancer is not None:\n        plt.imshow(avg_cancer)\n        plt.title('Average Cancer Image')\n        plt.show()\n\nif len(non_cancer_images) > 0:\n    avg_non_cancer = average_image(non_cancer_images)\n    if avg_non_cancer is not None:\n        plt.imshow(avg_non_cancer)\n        plt.title('Average Non-Cancer Image')\n        plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T07:47:41.009998Z","iopub.execute_input":"2025-12-02T07:47:41.010596Z","iopub.status.idle":"2025-12-02T07:47:46.027994Z","shell.execute_reply.started":"2025-12-02T07:47:41.01057Z","shell.execute_reply":"2025-12-02T07:47:46.027284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom torch.utils.data import DataLoader, Subset\nfrom torchvision import transforms\n\n# Assuming df_sample (filtered subsample with 'id' and 'label'), train_dir from earlier\n# Split into train/val (80/20, stratified)\ntrain_df, val_df = train_test_split(df_sample, test_size=0.2, stratify=df_sample['label'], random_state=42)\n\n# Transforms: Normalization from your EDA mean/std, plus augmentation for train\nmean = [0.715, 0.653, 0.708]  # Update with your exact values if different\nstd = [0.240, 0.238, 0.221]\n\ntrain_transform = transforms.Compose([\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomVerticalFlip(),\n    transforms.RandomRotation(90),\n    transforms.ColorJitter(brightness=0.1, contrast=0.1, saturation=0.1, hue=0.1),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=mean, std=std)\n])\n\nval_transform = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize(mean=mean, std=std)\n])\n\n# Custom Dataset (update to return image and label)\nclass CancerDataset(Dataset):\n    def __init__(self, df, img_dir, transform=None):\n        self.df = df\n        self.img_dir = img_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img_path = os.path.join(self.img_dir, row['id'] + '.tif')\n        try:\n            img = Image.open(img_path).convert('RGB')\n            label = torch.tensor(row['label'], dtype=torch.float32)  # Binary for BCE loss\n            if self.transform:\n                img = self.transform(img)\n            return img, label\n        except Exception as e:\n            print(f\"Error loading {img_path}: {e}\")\n            return torch.zeros(3, 96, 96), torch.tensor(0.0)  # Dummy\n\n# Create datasets\ntrain_dataset = CancerDataset(train_df, train_dir, train_transform)\nval_dataset = CancerDataset(val_df, train_dir, val_transform)\n\n# DataLoaders (batch_size=32 for CPU; increase on Colab)\ntrain_loader = DataLoader(train_dataset, batch_size=32, shuffle=True, num_workers=0)\nval_loader = DataLoader(val_dataset, batch_size=32, shuffle=False, num_workers=0)\n\n# Quick check\nprint(f\"Train size: {len(train_dataset)}, Val size: {len(val_dataset)}\")\nfor imgs, labels in train_loader:\n    print(f\"Batch shape: {imgs.shape}, Labels: {labels[:5]}\")  # Example output\n    break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T07:47:46.029125Z","iopub.execute_input":"2025-12-02T07:47:46.029367Z","iopub.status.idle":"2025-12-02T07:47:46.227268Z","shell.execute_reply.started":"2025-12-02T07:47:46.029349Z","shell.execute_reply":"2025-12-02T07:47:46.22662Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Device (CPU for now; GPU on Kaggle for 10K sample)\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\n# Simple CNN\nclass SimpleCNN(nn.Module):\n    def __init__(self):\n        super(SimpleCNN, self).__init__()\n        self.conv1 = nn.Conv2d(3, 32, kernel_size=3, padding=1)\n        self.conv2 = nn.Conv2d(32, 64, kernel_size=3, padding=1)\n        self.conv3 = nn.Conv2d(64, 128, kernel_size=3, padding=1)\n        self.pool = nn.MaxPool2d(2, 2)\n        self.fc1 = nn.Linear(128 * 12 * 12, 512)  # After 3 pools: 96/8=12\n        self.fc2 = nn.Linear(512, 1)\n        self.dropout = nn.Dropout(0.5)\n        self.relu = nn.ReLU()\n\n    def forward(self, x):\n        x = self.pool(self.relu(self.conv1(x)))\n        x = self.pool(self.relu(self.conv2(x)))\n        x = self.pool(self.relu(self.conv3(x)))\n        x = x.view(x.size(0), -1)\n        x = self.dropout(self.relu(self.fc1(x)))\n        x = torch.sigmoid(self.fc2(x))\n        return x\n\n# ResNet Transfer\nclass ResNetModel(nn.Module):\n    def __init__(self):\n        super(ResNetModel, self).__init__()\n        self.resnet = models.resnet18(weights='IMAGENET1K_V1')\n        self.resnet.fc = nn.Linear(self.resnet.fc.in_features, 1)\n\n    def forward(self, x):\n        return torch.sigmoid(self.resnet(x))\n\n# Instantiate (example: simple CNN)\nmodel = SimpleCNN().to(device)\ncriterion = nn.BCELoss()  # Binary cross-entropy\noptimizer = optim.Adam(model.parameters(), lr=0.001)\n\nprint(model)  # Verify architecture","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T07:47:46.228003Z","iopub.execute_input":"2025-12-02T07:47:46.22825Z","iopub.status.idle":"2025-12-02T07:47:46.428163Z","shell.execute_reply.started":"2025-12-02T07:47:46.228227Z","shell.execute_reply":"2025-12-02T07:47:46.42751Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nfrom sklearn.metrics import roc_auc_score, accuracy_score\nimport numpy as np\nimport matplotlib.pyplot as plt\n\n# Training function\ndef train_model(model, train_loader, val_loader, criterion, optimizer, epochs=20, device=None):\n    if device is None:\n        device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n    else:\n        device = torch.device(device)\n   \n    model = model.to(device) # Ensure model is on the correct device\n    train_losses, val_losses = [], []\n    for epoch in range(epochs):\n        model.train()\n        train_loss = 0\n        for imgs, labels in train_loader:\n            imgs, labels = imgs.to(device), labels.to(device).unsqueeze(1).float() # Ensure float for BCELoss\n            optimizer.zero_grad()\n            outputs = model(imgs)\n            loss = criterion(outputs, labels)\n            loss.backward()\n            optimizer.step()\n            train_loss += loss.item()\n        train_losses.append(train_loss / len(train_loader))\n      \n        model.eval()\n        val_loss = 0\n        preds, true = [], []\n        with torch.no_grad():\n            for imgs, labels in val_loader:\n                imgs, labels = imgs.to(device), labels.to(device).unsqueeze(1).float()\n                outputs = model(imgs)\n                loss = criterion(outputs, labels)\n                val_loss += loss.item()\n                preds.extend(outputs.cpu().numpy().flatten())\n                true.extend(labels.cpu().numpy().flatten())\n        val_losses.append(val_loss / len(val_loader))\n      \n        auc = roc_auc_score(true, preds)\n        acc = accuracy_score(true, np.round(preds))\n        print(f\"Epoch {epoch+1}: Train Loss {train_losses[-1]:.4f}, Val Loss {val_losses[-1]:.4f}, AUC {auc:.4f}, Acc {acc:.4f}\")\n  \n    # Plot losses\n    plt.plot(train_losses, label='Train Loss')\n    plt.plot(val_losses, label='Val Loss')\n    plt.title('Loss Curves')\n    plt.legend()\n    plt.show()\n  \n    return auc, acc\n# Set device globally\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n# Example: Train SimpleCNN\nmodel = SimpleCNN().to(device)\ncriterion = nn.BCELoss()\noptimizer = optim.Adam(model.parameters(), lr=0.001)\nsimple_auc, simple_acc = train_model(model, train_loader, val_loader, criterion, optimizer, device=device)\n# Train ResNet (similarly)\nresnet = ResNetModel().to(device)\noptimizer = optim.Adam(resnet.parameters(), lr=0.0001) # Lower lr for pretrained\nresnet_auc, resnet_acc = train_model(resnet, train_loader, val_loader, criterion, optimizer, device=device)\n# Table for comparison\nimport pandas as pd\nresults = pd.DataFrame({\n    'Model': ['SimpleCNN', 'ResNet18'],\n    'AUC': [simple_auc, resnet_auc],\n    'Accuracy': [simple_acc, resnet_acc]\n})\nprint(results)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T07:47:46.429615Z","iopub.execute_input":"2025-12-02T07:47:46.429884Z","iopub.status.idle":"2025-12-02T15:46:51.158195Z","shell.execute_reply.started":"2025-12-02T07:47:46.429867Z","shell.execute_reply":"2025-12-02T15:46:51.15732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Test dir and transforms (normalize with your mean/std)\ntest_dir = r'/kaggle/input/histopathologic-cancer-detection/test/'\n\ntest_transform = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.715, 0.653, 0.708], std=[0.240, 0.238, 0.221])\n])\n\n# Sample fraction for efficiency \nsample_fraction = 1\n\n# Get all test IDs from folder (no labels.csv for test)\nall_test_ids = [f.split('.tif')[0] for f in os.listdir(test_dir) if f.endswith('.tif')]\n\n# Print basic info on full test set\nprint(\"Full Test Set Number of Images:\", len(all_test_ids))\nprint(\"\\nSample of Test IDs (first 5):\", all_test_ids[:5])\n\n# Sample a fraction (random, no stratification since unlabeled)\nnum_samples = int(len(all_test_ids) * sample_fraction)\nsampled_ids = np.random.choice(all_test_ids, size=num_samples, replace=False)  # Random sample without replacement\n\n# Print basic info on sample\nprint(\"\\nSampled Test Set Number of Images ({}%):\".format(sample_fraction * 100), len(sampled_ids))\nprint(\"Sample of Sampled Test IDs (first 5):\", sampled_ids[:5])\n\n# Filter for existing files (though os.listdir ensures they exist, for consistency/safety)\ndef file_exists(id):\n    img_path = os.path.join(test_dir, id + '.tif')\n    return os.path.exists(img_path)\n\nprint(f\"\\nOriginal sampled size: {len(sampled_ids)}\")\nfiltered_ids = [id for id in sampled_ids if file_exists(id)]\nprint(f\"Filtered sampled size (valid files only): {len(filtered_ids)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T15:46:51.159737Z","iopub.execute_input":"2025-12-02T15:46:51.160366Z","iopub.status.idle":"2025-12-02T15:49:04.858229Z","shell.execute_reply.started":"2025-12-02T15:46:51.160346Z","shell.execute_reply":"2025-12-02T15:49:04.857547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Test Dataset (no labels, using filtered sampled IDs)\nclass TestDataset(Dataset):\n    def __init__(self, img_ids, img_dir, transform=None):\n        self.img_ids = img_ids\n        self.img_dir = img_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.img_ids)\n\n    def __getitem__(self, idx):\n        img_id = self.img_ids[idx]\n        img_path = os.path.join(self.img_dir, img_id + '.tif')\n        try:\n            img = Image.open(img_path).convert('RGB')\n            if self.transform:\n                img = self.transform(img)\n            return img, img_id\n        except Exception as e:\n            print(f\"Error loading {img_path}: {e}\")\n            return torch.zeros(3, 96, 96), img_id  # Dummy image\n\n# Loader with filtered sampled IDs\ntest_dataset = TestDataset(filtered_ids, test_dir, test_transform)\ntest_loader = DataLoader(test_dataset, batch_size=32, shuffle=False, num_workers=0)\n\n# Inference (use trained resnet from earlier; assume it's loaded/trained)\nresnet.eval()\npreds = []\nids = []\nwith torch.no_grad():\n    for imgs, batch_ids in test_loader:\n        imgs = imgs.to(device)\n        outputs = resnet(imgs).cpu().numpy().flatten()\n        preds.extend(outputs)\n        ids.extend(batch_ids)\n\n# Submission (partial, based on sample)\nsubmission = pd.DataFrame({'id': ids, 'label': preds})\nsubmission.to_csv('submission_sample.csv', index=False)  # Save as sample to distinguish\nprint(f\"\\nGenerated partial submission for {len(ids)} samples:\")\nprint(submission.head())  # Verify","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T15:49:04.859006Z","iopub.execute_input":"2025-12-02T15:49:04.859231Z","iopub.status.idle":"2025-12-02T15:50:32.584371Z","shell.execute_reply.started":"2025-12-02T15:49:04.859213Z","shell.execute_reply":"2025-12-02T15:50:32.583707Z"}},"outputs":[],"execution_count":null}]}