{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":117682,"databundleVersionId":15062069,"sourceType":"competition"}],"dockerImageVersionId":31234,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-25T01:25:41.827672Z","iopub.execute_input":"2025-12-25T01:25:41.827878Z","iopub.status.idle":"2025-12-25T01:25:45.084454Z","shell.execute_reply.started":"2025-12-25T01:25:41.827857Z","shell.execute_reply":"2025-12-25T01:25:45.083686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nimport zipfile\n\nDEVICE = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n\nprint(\"Torch:\", torch.__version__)\nprint(\"NumPy:\", np.__version__)\nprint(\"CUDA available:\", torch.cuda.is_available())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T01:25:45.085475Z","iopub.execute_input":"2025-12-25T01:25:45.085864Z","iopub.status.idle":"2025-12-25T01:25:48.941441Z","shell.execute_reply.started":"2025-12-25T01:25:45.085839Z","shell.execute_reply":"2025-12-25T01:25:48.940792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATASET_DIR = \"/kaggle/input/vesuvius-challenge-surface-detection\"\n\nTRAIN_IMG_DIR = os.path.join(DATASET_DIR, \"train_images\")\nTRAIN_MASK_DIR = os.path.join(DATASET_DIR, \"train_labels\")\nTEST_IMG_DIR  = os.path.join(DATASET_DIR, \"test_images\")\n\nprint(\"Train images:\", len(os.listdir(TRAIN_IMG_DIR)))\nprint(\"Train masks :\", len(os.listdir(TRAIN_MASK_DIR)))\nprint(\"Test images :\", os.listdir(TEST_IMG_DIR))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T01:25:48.942347Z","iopub.execute_input":"2025-12-25T01:25:48.942683Z","iopub.status.idle":"2025-12-25T01:25:48.949583Z","shell.execute_reply.started":"2025-12-25T01:25:48.942659Z","shell.execute_reply":"2025-12-25T01:25:48.948822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ScrollDataset(Dataset):\n    def __init__(self, image_dir, mask_dir=None, size=256):\n        self.image_dir = image_dir\n        self.mask_dir = mask_dir\n        self.size = size\n        self.images = sorted(os.listdir(image_dir))\n\n    def __len__(self):\n        return len(self.images)\n\n    def __getitem__(self, idx):\n        name = self.images[idx]\n\n        img = cv2.imread(\n            os.path.join(self.image_dir, name),\n            cv2.IMREAD_GRAYSCALE\n        )\n        img = cv2.resize(img, (self.size, self.size))\n        img = img.astype(\"float32\") / 255.0\n        img = torch.from_numpy(img).unsqueeze(0)\n\n        if self.mask_dir is None:\n            return img, name\n\n        mask = cv2.imread(\n            os.path.join(self.mask_dir, name),\n            cv2.IMREAD_GRAYSCALE\n        )\n        mask = cv2.resize(mask, (self.size, self.size), interpolation=cv2.INTER_NEAREST)\n        mask = (mask > 0).astype(\"float32\")\n        mask = torch.from_numpy(mask).unsqueeze(0)\n\n        return img, mask\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T01:25:48.950608Z","iopub.execute_input":"2025-12-25T01:25:48.950866Z","iopub.status.idle":"2025-12-25T01:25:48.959984Z","shell.execute_reply.started":"2025-12-25T01:25:48.950838Z","shell.execute_reply":"2025-12-25T01:25:48.959324Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class UNet(nn.Module):\n    def __init__(self):\n        super().__init__()\n\n        self.enc1 = nn.Sequential(\n            nn.Conv2d(1, 32, 3, padding=1),\n            nn.ReLU(),\n            nn.Conv2d(32, 32, 3, padding=1),\n            nn.ReLU(),\n        )\n        self.pool1 = nn.MaxPool2d(2)\n\n        self.enc2 = nn.Sequential(\n            nn.Conv2d(32, 64, 3, padding=1),\n            nn.ReLU(),\n            nn.Conv2d(64, 64, 3, padding=1),\n            nn.ReLU(),\n        )\n        self.pool2 = nn.MaxPool2d(2)\n\n        self.bottleneck = nn.Sequential(\n            nn.Conv2d(64, 128, 3, padding=1),\n            nn.ReLU(),\n        )\n\n        self.up2 = nn.ConvTranspose2d(128, 64, 2, stride=2)\n        self.dec2 = nn.Sequential(\n            nn.Conv2d(128, 64, 3, padding=1),\n            nn.ReLU(),\n        )\n\n        self.up1 = nn.ConvTranspose2d(64, 32, 2, stride=2)\n        self.dec1 = nn.Sequential(\n            nn.Conv2d(64, 32, 3, padding=1),\n            nn.ReLU(),\n        )\n\n        self.out = nn.Conv2d(32, 1, 1)\n\n    def forward(self, x):\n        e1 = self.enc1(x)\n        e2 = self.enc2(self.pool1(e1))\n        b  = self.bottleneck(self.pool2(e2))\n\n        d2 = self.up2(b)\n        d2 = self.dec2(torch.cat([d2, e2], dim=1))\n\n        d1 = self.up1(d2)\n        d1 = self.dec1(torch.cat([d1, e1], dim=1))\n\n        return self.out(d1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T01:25:48.960803Z","iopub.execute_input":"2025-12-25T01:25:48.961052Z","iopub.status.idle":"2025-12-25T01:25:48.971780Z","shell.execute_reply.started":"2025-12-25T01:25:48.961030Z","shell.execute_reply":"2025-12-25T01:25:48.971156Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def dice_loss(pred, target, eps=1e-6):\n    pred = torch.sigmoid(pred)\n    inter = (pred * target).sum()\n    union = pred.sum() + target.sum()\n    return 1 - (2 * inter + eps) / (union + eps)\n\nmodel = UNet().to(DEVICE)\noptimizer = torch.optim.Adam(model.parameters(), lr=1e-3)\n\ntrain_ds = ScrollDataset(TRAIN_IMG_DIR, TRAIN_MASK_DIR, size=256)\ntrain_loader = DataLoader(train_ds, batch_size=2, shuffle=True, num_workers=0)\n\nfor epoch in range(2):\n    model.train()\n    loss_sum = 0\n\n    for x, y in train_loader:\n        x, y = x.to(DEVICE), y.to(DEVICE)\n        optimizer.zero_grad()\n        loss = dice_loss(model(x), y)\n        loss.backward()\n        optimizer.step()\n        loss_sum += loss.item()\n\n    print(f\"Epoch {epoch+1} | Loss {loss_sum/len(train_loader):.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T01:25:48.972473Z","iopub.execute_input":"2025-12-25T01:25:48.972761Z","iopub.status.idle":"2025-12-25T01:35:30.595009Z","shell.execute_reply.started":"2025-12-25T01:25:48.972725Z","shell.execute_reply":"2025-12-25T01:35:30.594273Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def remove_small_components(mask, min_size=500):\n    mask = np.ascontiguousarray(mask.astype(np.uint8))\n    num_labels, labels, stats, _ = cv2.connectedComponentsWithStats(mask, connectivity=8)\n    out = np.zeros_like(mask)\n\n    for i in range(1, num_labels):\n        if stats[i, cv2.CC_STAT_AREA] >= min_size:\n            out[labels == i] = 1\n    return out\n\n\ndef post_process(pred, thresh=0.7):\n    if isinstance(pred, torch.Tensor):\n        pred = pred.cpu().numpy()\n\n    mask = (pred > thresh).astype(np.uint8)\n    mask = remove_small_components(mask)\n\n    kernel = np.ones((3, 3), np.uint8)\n    mask = cv2.morphologyEx(mask, cv2.MORPH_CLOSE, kernel)\n\n    return mask\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T01:35:30.595936Z","iopub.execute_input":"2025-12-25T01:35:30.596367Z","iopub.status.idle":"2025-12-25T01:35:30.602273Z","shell.execute_reply.started":"2025-12-25T01:35:30.596340Z","shell.execute_reply":"2025-12-25T01:35:30.601558Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.eval()\nos.makedirs(\"preds\", exist_ok=True)\n\nwith torch.no_grad():\n    for name in os.listdir(TEST_IMG_DIR):\n        if not name.endswith(\".tif\"):\n            continue\n\n        img = cv2.imread(\n            os.path.join(TEST_IMG_DIR, name),\n            cv2.IMREAD_GRAYSCALE\n        )\n        img = cv2.resize(img, (256, 256))\n        img = img.astype(\"float32\") / 255.0\n\n        x = torch.from_numpy(img).unsqueeze(0).unsqueeze(0).to(DEVICE)\n        raw = torch.sigmoid(model(x))[0, 0]\n        mask = post_process(raw)\n\n        cv2.imwrite(f\"preds/{name.replace('.tif','.png')}\", mask * 255)\n\nprint(\"Predictions:\", os.listdir(\"preds\"))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T01:35:30.603229Z","iopub.execute_input":"2025-12-25T01:35:30.603505Z","iopub.status.idle":"2025-12-25T01:35:31.307058Z","shell.execute_reply.started":"2025-12-25T01:35:30.603463Z","shell.execute_reply":"2025-12-25T01:35:31.306275Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with zipfile.ZipFile(\"/kaggle/working/submission.zip\", \"w\") as z:\n    for f in os.listdir(\"preds\"):\n        z.write(os.path.join(\"preds\", f), f)\n\nprint(\"submission.zip created\")\nprint(\"Files:\", os.listdir(\"/kaggle/working\"))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T01:35:31.309025Z","iopub.execute_input":"2025-12-25T01:35:31.309538Z","iopub.status.idle":"2025-12-25T01:35:31.314371Z","shell.execute_reply.started":"2025-12-25T01:35:31.309506Z","shell.execute_reply":"2025-12-25T01:35:31.313845Z"}},"outputs":[],"execution_count":null}]}