{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":117682,"databundleVersionId":15062069,"sourceType":"competition"},{"sourceId":14351778,"sourceType":"datasetVersion","datasetId":9163946}],"dockerImageVersionId":31234,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import sys\n\nDATASET_NAME = \"imagecodecs\"\n\n!{sys.executable} -m pip install --no-index --find-links /kaggle/input/{DATASET_NAME} imagecodecs -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-06T10:11:21.756979Z","iopub.execute_input":"2026-01-06T10:11:21.757278Z","iopub.status.idle":"2026-01-06T10:11:25.067416Z","shell.execute_reply.started":"2026-01-06T10:11:21.757253Z","shell.execute_reply":"2026-01-06T10:11:25.066577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom PIL import Image, ImageSequence\nfrom pathlib import Path\nimport zipfile\nfrom io import BytesIO\nimport tifffile as tiff\nfrom tqdm.auto import tqdm\nimport glob\n\n# --- 0. AUTOMATIC CONFIGURATION (Fixes FileNotFoundError) ---\nCOMP_DIR = Path(\"/kaggle/input/vesuvius-challenge-surface-detection\")\nTRAIN_IMG_DIR = COMP_DIR / \"train_images\"\n\n# Dynamically find actual file IDs in the folder\nif TRAIN_IMG_DIR.exists():\n    # Get all .tif files\n    all_tifs = list(TRAIN_IMG_DIR.glob(\"*.tif\"))\n    # Extract IDs (filenames without extension)\n    valid_ids = [f.stem for f in all_tifs]\n    print(f\"Found {len(valid_ids)} training volumes. Example IDs: {valid_ids[:5]}\")\n    \n    # Select first 2 for training (Demo) and next 1 for validation\n    # Warning: If dataset is small, adjust slicing\n    train_ids = valid_ids[:2] if len(valid_ids) >= 2 else valid_ids\n    valid_ids = valid_ids[2:3] if len(valid_ids) >= 3 else []\nelse:\n    print(\"Warning: Train directory not found (Committing?). Using empty lists.\")\n    train_ids = []\n    valid_ids = []\n\nCONFIG = {\n    \"train_chunks\": train_ids, \n    \"valid_chunks\": valid_ids,\n    \"batch_size\": 4,  # Reduced to 4 to prevent OOM on standard GPU\n    \"lr\": 1e-3,\n    \"epochs\": 10,      # Keep at 1 for fast submission check\n    \"device\": \"cuda\" if torch.cuda.is_available() else \"cpu\",\n    \"slice_depth\": 3, \n    \"comp_dir\": COMP_DIR,\n}","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-01-06T10:11:25.069498Z","iopub.execute_input":"2026-01-06T10:11:25.070253Z","iopub.status.idle":"2026-01-06T10:11:25.083797Z","shell.execute_reply.started":"2026-01-06T10:11:25.070220Z","shell.execute_reply":"2026-01-06T10:11:25.083138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 1. MODEL: LIGHTWEIGHT 2D U-NET ---\nclass SimpleUNet(nn.Module):\n    def __init__(self):\n        super().__init__()\n        def double_conv(in_c, out_c):\n            return nn.Sequential(\n                nn.Conv2d(in_c, out_c, 3, padding=1), nn.BatchNorm2d(out_c), nn.ReLU(inplace=True),\n                nn.Conv2d(out_c, out_c, 3, padding=1), nn.BatchNorm2d(out_c), nn.ReLU(inplace=True)\n            )\n        self.d1 = double_conv(CONFIG[\"slice_depth\"], 32)\n        self.pool = nn.MaxPool2d(2)\n        self.d2 = double_conv(32, 64)\n        self.d3 = double_conv(64, 128)\n        \n        self.up1 = nn.ConvTranspose2d(128, 64, 2, stride=2)\n        self.u1 = double_conv(128, 64)\n        self.up2 = nn.ConvTranspose2d(64, 32, 2, stride=2)\n        self.u2 = double_conv(64, 32)\n        self.out = nn.Conv2d(32, 1, 1)\n\n    def forward(self, x):\n        c1 = self.d1(x)\n        p1 = self.pool(c1)\n        c2 = self.d2(p1)\n        p2 = self.pool(c2)\n        c3 = self.d3(p2) \n        \n        u1 = self.up1(c3)\n        u1 = torch.cat([u1, c2], dim=1)\n        u1 = self.u1(u1)\n        \n        u2 = self.up2(u1)\n        u2 = torch.cat([u2, c1], dim=1)\n        u2 = self.u2(u2)\n        \n        return self.out(u2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-06T10:11:25.084589Z","iopub.execute_input":"2026-01-06T10:11:25.084821Z","iopub.status.idle":"2026-01-06T10:11:25.096556Z","shell.execute_reply.started":"2026-01-06T10:11:25.084791Z","shell.execute_reply":"2026-01-06T10:11:25.095799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 2. DATASET ---\nclass VesuviusSliceDataset(Dataset):\n    def __init__(self, chunk_ids, mode='train'):\n        self.mode = mode\n        self.images = []\n        self.labels = []\n        \n        if len(chunk_ids) == 0: return\n\n        print(f\"Loading {mode} chunks: {chunk_ids}...\")\n        for cid in chunk_ids:\n            img_path = CONFIG[\"comp_dir\"] / f\"train_images/{cid}.tif\"\n            lbl_path = CONFIG[\"comp_dir\"] / f\"train_labels/{cid}.tif\"\n            \n            # Load and Normalize\n            try:\n                vol = tiff.imread(img_path).astype(np.float16) / 65535.0\n            except Exception as e:\n                print(f\"Skipping bad file {cid}: {e}\")\n                continue\n\n            if mode == 'train':\n                if not lbl_path.exists(): continue\n                mask = tiff.imread(lbl_path).astype(np.uint8)\n                \n                # Filter empty slices to speed up training\n                # Only train on slices that contain the scroll (Label 1)\n                # We stride by 2 to reduce memory usage\n                for z in range(0, vol.shape[0] - CONFIG[\"slice_depth\"], 2):\n                    mid_slice = z + CONFIG[\"slice_depth\"]//2\n                    if np.any(mask[mid_slice] == 1): \n                        self.images.append(vol[z : z+CONFIG[\"slice_depth\"]])\n                        self.labels.append(mask[mid_slice])\n            else:\n                # Validation/Inference logic (Dense sliding window)\n                 for z in range(0, vol.shape[0] - CONFIG[\"slice_depth\"], 10): \n                    self.images.append(vol[z : z+CONFIG[\"slice_depth\"]])\n                    # Load label if available (for validation)\n                    if lbl_path.exists():\n                        self.labels.append(tiff.imread(lbl_path)[z + CONFIG[\"slice_depth\"]//2])\n\n    def __len__(self):\n        return len(self.images)\n\n    def __getitem__(self, idx):\n        img = self.images[idx] \n        # Resize to 256x256 for demo speed\n        img_tensor = torch.from_numpy(img).float()\n        img_tensor = torch.nn.functional.interpolate(img_tensor.unsqueeze(0), size=(256, 256), mode='bilinear').squeeze(0)\n        \n        if self.mode == 'train' and idx < len(self.labels):\n            lbl = self.labels[idx]\n            lbl_tensor = torch.from_numpy(lbl).long()\n            lbl_tensor = torch.nn.functional.interpolate(lbl_tensor.unsqueeze(0).unsqueeze(0).float(), size=(256, 256), mode='nearest').squeeze()\n            return img_tensor, lbl_tensor\n        \n        return img_tensor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-06T10:11:25.098126Z","iopub.execute_input":"2026-01-06T10:11:25.098378Z","iopub.status.idle":"2026-01-06T10:11:25.119498Z","shell.execute_reply.started":"2026-01-06T10:11:25.098351Z","shell.execute_reply":"2026-01-06T10:11:25.118865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 3. TRAIN ---\ndef train():\n    model = SimpleUNet().to(CONFIG[\"device\"])\n    \n    # Skip training if no IDs found (e.g. Inference only run)\n    if not CONFIG[\"train_chunks\"]:\n        print(\"No training data found. Returning initialized model.\")\n        return model\n\n    ds = VesuviusSliceDataset(CONFIG[\"train_chunks\"], mode='train')\n    # Safety check if dataset is empty after filtering\n    if len(ds) == 0:\n        print(\"Dataset empty after filtering. Returning model.\")\n        return model\n\n    dl = DataLoader(ds, batch_size=CONFIG[\"batch_size\"], shuffle=True, num_workers=0) # workers=0 is safer\n    \n    optimizer = optim.Adam(model.parameters(), lr=CONFIG[\"lr\"])\n    criterion = nn.BCEWithLogitsLoss()\n    \n    print(\"Starting Training...\")\n    model.train()\n    for epoch in range(CONFIG[\"epochs\"]):\n        pbar = tqdm(dl, desc=f\"Epoch {epoch}\")\n        for imgs, masks in pbar:\n            imgs = imgs.to(CONFIG[\"device\"])\n            masks = (masks == 1).float().to(CONFIG[\"device\"])\n            \n            optimizer.zero_grad()\n            preds = model(imgs).squeeze(1)\n            loss = criterion(preds, masks)\n            loss.backward()\n            optimizer.step()\n            pbar.set_postfix(loss=loss.item())\n            \n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-06T10:11:25.120365Z","iopub.execute_input":"2026-01-06T10:11:25.120597Z","iopub.status.idle":"2026-01-06T10:11:25.140932Z","shell.execute_reply.started":"2026-01-06T10:11:25.120576Z","shell.execute_reply":"2026-01-06T10:11:25.140285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 4. INFERENCE ---\ndef inference(model):\n    test_csv = CONFIG[\"comp_dir\"] / \"test.csv\"\n    test_img_dir = CONFIG[\"comp_dir\"] / \"test_images\"\n    submission_path = Path(\"/kaggle/working/submission.zip\")\n    \n    # Get Test IDs\n    if test_csv.exists():\n        try:\n            df = pd.read_csv(test_csv)\n            chunks = df['id'].astype(str).tolist()\n        except:\n            chunks = []\n    else:\n        chunks = []\n        \n    # Fallback: If chunks empty, grab from folder (Code or Commit mode)\n    if not chunks and test_img_dir.exists():\n        chunks = [f.stem for f in test_img_dir.glob(\"*.tif\")]\n\n    print(f\"Running Inference on {len(chunks)} volumes: {chunks}\")\n    model.eval()\n    \n    with zipfile.ZipFile(submission_path, \"w\", compression=zipfile.ZIP_DEFLATED) as zf:\n        for cid in chunks:\n            img_path = test_img_dir / f\"{cid}.tif\"\n            if not img_path.exists(): \n                print(f\"File missing: {img_path}\")\n                continue\n            \n            print(f\"Processing {cid}...\")\n            try:\n                vol = tiff.imread(img_path).astype(np.float32) / 65535.0\n                D, H, W = vol.shape\n                \n                pred_vol = np.zeros((D, H, W), dtype=np.uint8)\n                \n                # Slice-by-slice inference\n                for z in range(CONFIG[\"slice_depth\"]//2, D - CONFIG[\"slice_depth\"]//2):\n                    input_slice = vol[z - CONFIG[\"slice_depth\"]//2 : z + CONFIG[\"slice_depth\"]//2 + 1]\n                    \n                    tensor = torch.from_numpy(input_slice).unsqueeze(0).float()\n                    tensor = torch.nn.functional.interpolate(tensor, size=(256, 256), mode='bilinear')\n                    tensor = tensor.to(CONFIG[\"device\"])\n                    \n                    with torch.no_grad():\n                        out = model(tensor)\n                        out = torch.sigmoid(out)\n                    \n                    out = torch.nn.functional.interpolate(out, size=(H, W), mode='bilinear')\n                    mask = (out.squeeze().cpu().numpy() > 0.5).astype(np.uint8)\n                    \n                    pred_vol[z] = mask\n\n                # Save\n                img_buffer = BytesIO()\n                pages = [Image.fromarray(s * 255) for s in pred_vol]\n                pages[0].save(img_buffer, format=\"TIFF\", save_all=True, append_images=pages[1:])\n                zf.writestr(f\"{cid}.tif\", img_buffer.getvalue())\n                \n            except Exception as e:\n                print(f\"Error on {cid}: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-06T10:11:25.141868Z","iopub.execute_input":"2026-01-06T10:11:25.142174Z","iopub.status.idle":"2026-01-06T10:11:25.164197Z","shell.execute_reply.started":"2026-01-06T10:11:25.142144Z","shell.execute_reply":"2026-01-06T10:11:25.163622Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- EXECUTE ---\nif __name__ == \"__main__\":\n    trained_model = train()\n    inference(trained_model)\n    print(\"Done! Output saved to submission.zip\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-06T10:11:25.165057Z","iopub.execute_input":"2026-01-06T10:11:25.165343Z","iopub.status.idle":"2026-01-06T10:12:00.083288Z","shell.execute_reply.started":"2026-01-06T10:11:25.165313Z","shell.execute_reply":"2026-01-06T10:12:00.082509Z"}},"outputs":[],"execution_count":null}]}