{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":117682,"databundleVersionId":14443416,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Install required packages\n!pip install imagecodecs\n\nimport os\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport torch.nn.functional as F\nimport tifffile as tiff\nimport pandas as pd\nfrom tqdm import tqdm\nimport cv2\n\n# GPU setup - ab GPU use karenge\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(f\"Using device: {device}\")\n\n# Dataset Class\nclass ScrollDataset(Dataset):\n    def __init__(self, image_dir, label_dir, csv_file, max_depth=32, target_size=256):\n        self.image_dir = image_dir\n        self.label_dir = label_dir\n        self.df = pd.read_csv(csv_file)\n        self.max_depth = max_depth\n        self.target_size = target_size\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def load_tiff(self, path):\n        try:\n            return tiff.imread(path)\n        except:\n            import imagecodecs\n            return tiff.imread(path)\n    \n    def resize_volume(self, volume, target_depth, target_height, target_width):\n        \"\"\"Resize 3D volume to target dimensions\"\"\"\n        # Use interpolation for resizing\n        resized = np.zeros((target_depth, target_height, target_width), dtype=volume.dtype)\n        \n        depth_ratio = volume.shape[0] / target_depth\n        height_ratio = volume.shape[1] / target_height\n        width_ratio = volume.shape[2] / target_width\n        \n        for d in range(target_depth):\n            orig_d = min(int(d * depth_ratio), volume.shape[0] - 1)\n            slice_2d = volume[orig_d]\n            # Resize 2D slice\n            resized_slice = cv2.resize(slice_2d, (target_width, target_height), \n                                     interpolation=cv2.INTER_AREA)\n            resized[d] = resized_slice\n            \n        return resized\n    \n    def __getitem__(self, idx):\n        image_id = str(self.df.iloc[idx]['id'])\n        image_path = os.path.join(self.image_dir, f\"{image_id}.tif\")\n        label_path = os.path.join(self.label_dir, f\"{image_id}.tif\")\n        \n        volume = self.load_tiff(image_path)\n        mask = self.load_tiff(label_path)\n        \n        # Handle 2D images\n        if len(volume.shape) == 2:\n            volume = volume[np.newaxis, :, :]\n            mask = mask[np.newaxis, :, :]\n        \n        # Resize to consistent dimensions\n        if volume.shape != (self.max_depth, self.target_size, self.target_size):\n            volume = self.resize_volume(volume, self.max_depth, self.target_size, self.target_size)\n            mask = self.resize_volume(mask, self.max_depth, self.target_size, self.target_size)\n        \n        # Normalize volume\n        volume = volume.astype(np.float32)\n        volume = (volume - volume.mean()) / (volume.std() + 1e-8)\n        \n        # Normalize mask to 0-1 range\n        mask = mask.astype(np.float32)\n        if mask.max() > 1.0:\n            mask = mask / 255.0\n        mask = np.clip(mask, 0.0, 1.0)\n        \n        # Convert to torch\n        volume = torch.from_numpy(volume).float().unsqueeze(0)  # [1, D, H, W]\n        mask = torch.from_numpy(mask).float().unsqueeze(0)     # [1, D, H, W]\n        \n        return volume, mask\n\n# Improved 3D UNet Model\nclass UNet3D(nn.Module):\n    def __init__(self, in_channels=1, out_channels=1, features=[16, 32, 64]):\n        super().__init__()\n        \n        # Encoder\n        self.enc1 = self._block(in_channels, features[0])\n        self.pool1 = nn.MaxPool3d(2, 2)\n        \n        self.enc2 = self._block(features[0], features[1])\n        self.pool2 = nn.MaxPool3d(2, 2)\n        \n        self.enc3 = self._block(features[1], features[2])\n        self.pool3 = nn.MaxPool3d(2, 2)\n        \n        # Bottleneck\n        self.bottleneck = self._block(features[2], features[2] * 2)\n        \n        # Decoder\n        self.upconv3 = nn.ConvTranspose3d(features[2] * 2, features[2], 2, 2)\n        self.dec3 = self._block(features[2] * 2, features[2])\n        \n        self.upconv2 = nn.ConvTranspose3d(features[2], features[1], 2, 2)\n        self.dec2 = self._block(features[1] * 2, features[1])\n        \n        self.upconv1 = nn.ConvTranspose3d(features[1], features[0], 2, 2)\n        self.dec1 = self._block(features[0] * 2, features[0])\n        \n        self.final = nn.Conv3d(features[0], out_channels, 1)\n        \n    def _block(self, in_channels, out_channels):\n        return nn.Sequential(\n            nn.Conv3d(in_channels, out_channels, 3, padding=1, bias=False),\n            nn.BatchNorm3d(out_channels),\n            nn.ReLU(inplace=True),\n            nn.Conv3d(out_channels, out_channels, 3, padding=1, bias=False),\n            nn.BatchNorm3d(out_channels),\n            nn.ReLU(inplace=True)\n        )\n    \n    def forward(self, x):\n        # Encoder\n        e1 = self.enc1(x)\n        e2 = self.enc2(self.pool1(e1))\n        e3 = self.enc3(self.pool2(e2))\n        \n        # Bottleneck\n        b = self.bottleneck(self.pool3(e3))\n        \n        # Decoder\n        d3 = self.upconv3(b)\n        d3 = torch.cat((d3, e3), dim=1)\n        d3 = self.dec3(d3)\n        \n        d2 = self.upconv2(d3)\n        d2 = torch.cat((d2, e2), dim=1)\n        d2 = self.dec2(d2)\n        \n        d1 = self.upconv1(d2)\n        d1 = torch.cat((d1, e1), dim=1)\n        d1 = self.dec1(d1)\n        \n        return torch.sigmoid(self.final(d1))\n\n# Initialize model\nmodel = UNet3D(in_channels=1, out_channels=1, features=[16, 32, 64])\nmodel = model.to(device)\n\nprint(f\"Model parameters: {sum(p.numel() for p in model.parameters()):,}\")\n\n# Loss function\ndef dice_loss(pred, target, smooth=1.):\n    pred = pred.contiguous()\n    target = target.contiguous()\n    \n    intersection = (pred * target).sum()\n    dice = (2. * intersection + smooth) / (pred.sum() + target.sum() + smooth)\n    return 1 - dice\n\ndef combined_loss(pred, target):\n    bce = F.binary_cross_entropy(pred, target)\n    dice = dice_loss(pred, target)\n    return bce + dice\n\noptimizer = optim.Adam(model.parameters(), lr=1e-4, weight_decay=1e-5)\nscheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, 'min', patience=2, factor=0.5)\n\n# DataLoader\ntrain_csv = '/kaggle/input/vesuvius-challenge-surface-detection/train.csv'\ntrain_image_dir = '/kaggle/input/vesuvius-challenge-surface-detection/train_images'\ntrain_label_dir = '/kaggle/input/vesuvius-challenge-surface-detection/train_labels'\n\ndataset = ScrollDataset(train_image_dir, train_label_dir, train_csv, max_depth=32, target_size=256)\ntrain_loader = DataLoader(dataset, batch_size=1, shuffle=True, num_workers=2, pin_memory=True)\n\nprint(f\"Training on {len(dataset)} samples\")\n\n# Training loop\ndef train_model():\n    model.train()\n    best_loss = float('inf')\n    \n    for epoch in range(10):\n        epoch_loss = 0.0\n        progress_bar = tqdm(train_loader, desc=f'Epoch {epoch+1}/10')\n        \n        for data, target in progress_bar:\n            data, target = data.to(device), target.to(device)\n            \n            optimizer.zero_grad()\n            output = model(data)\n            loss = combined_loss(output, target)\n            \n            loss.backward()\n            torch.nn.utils.clip_grad_norm_(model.parameters(), max_norm=1.0)\n            optimizer.step()\n            \n            epoch_loss += loss.item()\n            progress_bar.set_postfix({'Loss': f'{loss.item():.4f}'})\n        \n        avg_loss = epoch_loss / len(train_loader)\n        scheduler.step(avg_loss)\n        \n        print(f'Epoch {epoch+1}, Average Loss: {avg_loss:.4f}, LR: {optimizer.param_groups[0][\"lr\"]:.2e}')\n        \n        # Save best model\n        if avg_loss < best_loss:\n            best_loss = avg_loss\n            torch.save({\n                'epoch': epoch,\n                'model_state_dict': model.state_dict(),\n                'optimizer_state_dict': optimizer.state_dict(),\n                'loss': best_loss,\n            }, '/kaggle/working/best_model.pth')\n            print(f'Best model saved with loss: {best_loss:.4f}')\n\n# Start training\nprint(\"Starting training on GPU...\")\ntrain_model()\n\n# Save final model\ntorch.save({\n    'model_state_dict': model.state_dict(),\n    'model_config': {\n        'in_channels': 1,\n        'out_channels': 1, \n        'features': [16, 32, 64]\n    }\n}, '/kaggle/working/final_model.pth')\n\nprint(\"Training completed!\")\nprint(\"Models saved:\")\nprint(\"- /kaggle/working/best_model.pth\")\nprint(\"- /kaggle/working/final_model.pth\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-13T20:39:29.607806Z","iopub.execute_input":"2025-11-13T20:39:29.608118Z","iopub.status.idle":"2025-11-13T21:54:45.218867Z","shell.execute_reply.started":"2025-11-13T20:39:29.608064Z","shell.execute_reply":"2025-11-13T21:54:45.217895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}