{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"},{"sourceId":211959013,"sourceType":"kernelVersion"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom glob import glob\nfrom PIL import Image\nimport torch\nfrom torchvision import transforms\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch import nn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T02:31:24.679495Z","iopub.execute_input":"2024-12-09T02:31:24.679910Z","iopub.status.idle":"2024-12-09T02:31:24.684768Z","shell.execute_reply.started":"2024-12-09T02:31:24.679877Z","shell.execute_reply":"2024-12-09T02:31:24.683945Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Hyperparameters\nIMAGE_SIZE = 96\nBATCH_SIZE = 32\nINPUT_DIR = '/kaggle/input/histopathologic-cancer-detection/'\nMODEL_PATH = '/kaggle/input/as-week-7-baseline-cnn-training/unet_histopathologic.pth'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T02:31:24.686101Z","iopub.execute_input":"2024-12-09T02:31:24.686341Z","iopub.status.idle":"2024-12-09T02:31:24.700816Z","shell.execute_reply.started":"2024-12-09T02:31:24.686317Z","shell.execute_reply":"2024-12-09T02:31:24.700051Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Dataset class\nclass CancerDataset(Dataset):\n    def __init__(self, dataframe, transform=None):\n        self.dataframe = dataframe\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.dataframe)\n\n    def __getitem__(self, idx):\n        row = self.dataframe.iloc[idx]\n        image = Image.open(row['path']).convert(\"RGB\")\n        if self.transform:\n            image = self.transform(image)\n        return image, row['id']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T02:31:24.701608Z","iopub.execute_input":"2024-12-09T02:31:24.701847Z","iopub.status.idle":"2024-12-09T02:31:24.711783Z","shell.execute_reply.started":"2024-12-09T02:31:24.701824Z","shell.execute_reply":"2024-12-09T02:31:24.711118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Transforms\ntest_transforms = transforms.Compose([\n    transforms.Resize((IMAGE_SIZE, IMAGE_SIZE)),\n    transforms.ToTensor(),\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T02:31:24.713331Z","iopub.execute_input":"2024-12-09T02:31:24.713620Z","iopub.status.idle":"2024-12-09T02:31:24.722721Z","shell.execute_reply.started":"2024-12-09T02:31:24.713571Z","shell.execute_reply":"2024-12-09T02:31:24.721847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the test dataset\ntesting_dir = os.path.join(INPUT_DIR, 'test/')\ntest_files = glob(os.path.join(testing_dir, '*.tif'))\n\ntest_df = pd.DataFrame({'path': test_files})\ntest_df['id'] = test_df['path'].map(lambda x: x.split('/')[-1].split('.')[0])\n\ntest_dataset = CancerDataset(test_df, transform=test_transforms)\ntest_loader = DataLoader(test_dataset, batch_size=BATCH_SIZE, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T02:31:24.723719Z","iopub.execute_input":"2024-12-09T02:31:24.724038Z","iopub.status.idle":"2024-12-09T02:31:24.878237Z","shell.execute_reply.started":"2024-12-09T02:31:24.724002Z","shell.execute_reply":"2024-12-09T02:31:24.877639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define U-Net model (same architecture as used in training)\nclass UNet(nn.Module):\n    def __init__(self, in_channels=3, out_channels=1):\n        super(UNet, self).__init__()\n        self.encoder = nn.Sequential(\n            self.double_conv(in_channels, 64),\n            self.downsample(64, 128),\n            self.downsample(128, 256),\n            self.downsample(256, 512)\n        )\n        self.decoder = nn.Sequential(\n            self.upsample(512, 256),\n            self.upsample(256, 128),\n            self.upsample(128, 64),\n            nn.Conv2d(64, out_channels, kernel_size=1)\n        )\n        self.global_avg_pool = nn.AdaptiveAvgPool2d(1)  # Pool to 1x1\n        self.sigmoid = nn.Sigmoid()\n\n    def double_conv(self, in_channels, out_channels):\n        return nn.Sequential(\n            nn.Conv2d(in_channels, out_channels, kernel_size=3, padding=1),\n            nn.ReLU(inplace=True),\n            nn.Conv2d(out_channels, out_channels, kernel_size=3, padding=1),\n            nn.ReLU(inplace=True)\n        )\n\n    def downsample(self, in_channels, out_channels):\n        return nn.Sequential(\n            nn.MaxPool2d(2),\n            self.double_conv(in_channels, out_channels)\n        )\n\n    def upsample(self, in_channels, out_channels):\n        return nn.Sequential(\n            nn.ConvTranspose2d(in_channels, out_channels, kernel_size=2, stride=2),\n            self.double_conv(out_channels, out_channels)\n        )\n\n    def forward(self, x):\n        enc1 = self.encoder[0](x)\n        enc2 = self.encoder[1](enc1)\n        enc3 = self.encoder[2](enc2)\n        enc4 = self.encoder[3](enc3)\n\n        dec3 = self.decoder[0](enc4)\n        dec2 = self.decoder[1](dec3)\n        dec1 = self.decoder[2](dec2)\n        output = self.decoder[3](dec1)\n\n        # Global average pooling to reduce spatial dimensions\n        output = self.global_avg_pool(output)  # Output size: [batch_size, 1, 1, 1]\n        return self.sigmoid(output).squeeze(dim=-1).squeeze(dim=-1)  # Final size: [batch_size, 1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T02:31:24.879079Z","iopub.execute_input":"2024-12-09T02:31:24.879292Z","iopub.status.idle":"2024-12-09T02:31:24.888461Z","shell.execute_reply.started":"2024-12-09T02:31:24.879270Z","shell.execute_reply":"2024-12-09T02:31:24.887761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the trained model\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nmodel = UNet(in_channels=3, out_channels=1).to(device)\nmodel.load_state_dict(torch.load(MODEL_PATH))\nmodel.eval()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T02:31:24.889728Z","iopub.execute_input":"2024-12-09T02:31:24.890641Z","iopub.status.idle":"2024-12-09T02:31:25.372039Z","shell.execute_reply.started":"2024-12-09T02:31:24.890579Z","shell.execute_reply":"2024-12-09T02:31:25.371061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generate predictions for test data\npredictions = []\nids = []\n\nwith torch.no_grad():\n    for images, image_ids in test_loader:\n        images = images.to(device)\n        outputs = model(images).squeeze()  # Output size: [batch_size]\n        batch_preds = (outputs > 0.5).float().cpu().numpy()  # Convert to binary predictions\n        predictions.extend(batch_preds)\n        ids.extend(image_ids)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T02:31:25.374056Z","iopub.execute_input":"2024-12-09T02:31:25.374323Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create and save the submission file\nsubmission_df = pd.DataFrame({\"id\": ids, \"label\": predictions})\nsubmission_df['label'] = submission_df['label'].astype(int)  # Convert to integer for Kaggle submission\nsubmission_df.to_csv(\"submission.csv\", index=False)\nprint(\"Submission file saved as 'submission.csv'.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}