{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":117682,"databundleVersionId":15062069,"sourceType":"competition"}],"dockerImageVersionId":31260,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"e189136d","cell_type":"markdown","source":"# Vesuvius Challenge - Improved v3 (Unsupervised, Host-Inspired)\n\n**Goal**: Improve over baseline (0.424) using unsupervised, host-inspired techniques.\n\n**Key Ideas Applied**:\n- Z-score normalization (host baseline uses ZScoreNormalization)\n- Edge-aware segmentation (gradient magnitude)\n- Optional sheetness filter (Frangi-inspired, unsupervised)\n- Conservative thresholds to avoid over-segmentation\n\nThis notebook stays unsupervised and produces a Kaggle-ready `submission.zip`.","metadata":{}},{"id":"7f10c359","cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom tqdm import tqdm\nfrom PIL import Image, ImageSequence\nimport tifffile as tiff\nfrom scipy import ndimage\nfrom scipy.ndimage import gaussian_filter\nimport os\nimport shutil\nimport zipfile","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T04:12:23.385941Z","iopub.execute_input":"2026-02-13T04:12:23.386326Z","iopub.status.idle":"2026-02-13T04:12:23.391038Z","shell.execute_reply.started":"2026-02-13T04:12:23.386285Z","shell.execute_reply":"2026-02-13T04:12:23.390251Z"}},"outputs":[],"execution_count":null},{"id":"1d07da06","cell_type":"code","source":"# Check if running on Kaggle\nIS_KAGGLE = 'KAGGLE_DATA_PROXY_URL' in os.environ\nprint(f\"Running on Kaggle: {IS_KAGGLE}\")\n\nif IS_KAGGLE:\n    DATA_PATH = Path('/kaggle/input/vesuvius-challenge-surface-detection')\n    OUTPUT_PATH = Path('/kaggle/working')\nelse:\n    DATA_PATH = Path('../vesuvius-challenge-surface-detection')\n    OUTPUT_PATH = Path('../output')\n\nOUTPUT_PATH.mkdir(parents=True, exist_ok=True)\nprint(f\"Data path: {DATA_PATH}\")\nprint(f\"Output path: {OUTPUT_PATH}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T04:12:23.392302Z","iopub.execute_input":"2026-02-13T04:12:23.392671Z","iopub.status.idle":"2026-02-13T04:12:23.409564Z","shell.execute_reply.started":"2026-02-13T04:12:23.392648Z","shell.execute_reply":"2026-02-13T04:12:23.408887Z"}},"outputs":[],"execution_count":null},{"id":"3e4ae4aa","cell_type":"code","source":"def read_volume(path: Path):\n    \"\"\"Read 3D volume using PIL\"\"\"\n    im = Image.open(str(path))\n    frames = [np.array(f) for f in ImageSequence.Iterator(im)]\n    vol = np.stack(frames, axis=0)\n    if vol.ndim == 2:\n        vol = vol[None, ...]\n    return vol\n\n# Load test data\ntest_df = pd.read_csv(DATA_PATH / 'test.csv')\ntest_ids = test_df['id'].astype(str).tolist()\n\nprint(f\"\\nLoading {len(test_ids)} test volumes...\")\ntest_volumes = {}\nfor tid in tqdm(test_ids):\n    img_path = DATA_PATH / 'test_images' / f\"{tid}.tif\"\n    if img_path.exists():\n        test_volumes[tid] = read_volume(img_path)\n\nprint(f\"✓ Loaded {len(test_volumes)} test volumes\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T04:12:23.410808Z","iopub.execute_input":"2026-02-13T04:12:23.411149Z","iopub.status.idle":"2026-02-13T04:12:25.668354Z","shell.execute_reply.started":"2026-02-13T04:12:23.411114Z","shell.execute_reply":"2026-02-13T04:12:25.667539Z"}},"outputs":[],"execution_count":null},{"id":"8995e7b8","cell_type":"markdown","source":"## Preprocessing (Host-Inspired)\n\nHost baseline uses **Z-score normalization**. We apply that per-volume.","metadata":{}},{"id":"7fd8e613","cell_type":"code","source":"def zscore_normalize(volume):\n    vol = volume.astype(np.float32)\n    mean = vol.mean()\n    std = vol.std() + 1e-6\n    return (vol - mean) / std\n\ndef preprocess_volume(volume, blur_sigma=0.5):\n    vol = zscore_normalize(volume)\n    if blur_sigma and blur_sigma > 0:\n        vol = gaussian_filter(vol, sigma=blur_sigma)\n    return vol","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T04:12:25.669395Z","iopub.execute_input":"2026-02-13T04:12:25.670082Z","iopub.status.idle":"2026-02-13T04:12:25.675272Z","shell.execute_reply.started":"2026-02-13T04:12:25.670051Z","shell.execute_reply":"2026-02-13T04:12:25.674369Z"}},"outputs":[],"execution_count":null},{"id":"50f0629a","cell_type":"markdown","source":"## Optional Sheetness Filter (Frangi-Inspired, Unsupervised)\n\nThis is a lightweight, unsupervised surface-enhancement filter inspired by Frangi.\nIt operates on the Hessian eigenvalues and favors sheet-like structures.","metadata":{}},{"id":"4f31794c","cell_type":"code","source":"def hessian_elements(volume, sigma=1.0):\n    v = volume.astype(np.float32)\n    vxx = gaussian_filter(v, sigma=sigma, order=(0, 0, 2))\n    vyy = gaussian_filter(v, sigma=sigma, order=(0, 2, 0))\n    vzz = gaussian_filter(v, sigma=sigma, order=(2, 0, 0))\n    vxy = gaussian_filter(v, sigma=sigma, order=(0, 1, 1))\n    vxz = gaussian_filter(v, sigma=sigma, order=(1, 0, 1))\n    vyz = gaussian_filter(v, sigma=sigma, order=(1, 1, 0))\n    return vxx, vyy, vzz, vxy, vxz, vyz\n\ndef sheetness_filter(volume, sigma=1.0, alpha=0.5, beta=0.5, gamma=15.0):\n    vxx, vyy, vzz, vxy, vxz, vyz = hessian_elements(volume, sigma=sigma)\n\n    shape = volume.shape\n    sheetness = np.zeros(shape, dtype=np.float32)\n\n    # Process slice-wise for memory safety\n    for z in range(shape[0]):\n        H = np.zeros((shape[1], shape[2], 3, 3), dtype=np.float32)\n        H[:, :, 0, 0] = vzz[z]\n        H[:, :, 1, 1] = vyy[z]\n        H[:, :, 2, 2] = vxx[z]\n        H[:, :, 0, 1] = H[:, :, 1, 0] = vyz[z]\n        H[:, :, 0, 2] = H[:, :, 2, 0] = vxz[z]\n        H[:, :, 1, 2] = H[:, :, 2, 1] = vxy[z]\n\n        eigvals = np.linalg.eigvalsh(H)\n        l1 = np.abs(eigvals[:, :, 0])\n        l2 = np.abs(eigvals[:, :, 1])\n        l3 = np.abs(eigvals[:, :, 2])\n\n        ra = l2 / (l3 + 1e-6)\n        rb = l1 / (np.sqrt(l2 * l3) + 1e-6)\n        s = np.sqrt(l1**2 + l2**2 + l3**2)\n\n        sheet = np.exp(-(ra**2) / (2 * alpha**2)) * np.exp(-(rb**2) / (2 * beta**2)) * (1 - np.exp(-(s**2) / (2 * gamma**2)))\n        sheetness[z] = sheet\n\n    # Normalize to [0, 1]\n    sheetness = (sheetness - sheetness.min()) / (sheetness.max() - sheetness.min() + 1e-6)\n    return sheetness","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T04:12:25.677536Z","iopub.execute_input":"2026-02-13T04:12:25.677860Z","iopub.status.idle":"2026-02-13T04:12:25.690447Z","shell.execute_reply.started":"2026-02-13T04:12:25.677809Z","shell.execute_reply":"2026-02-13T04:12:25.689534Z"}},"outputs":[],"execution_count":null},{"id":"ea6c9811","cell_type":"markdown","source":"## Unsupervised Segmentation (Host-Inspired)\n\nWe combine intensity thresholding + gradient edges + optional sheetness.\nAll thresholds are percentile-based to remain robust.","metadata":{}},{"id":"50c34415","cell_type":"code","source":"def segment_surface_unsupervised_v3(\n    volume,\n    intensity_percentile=65,\n    grad_percentile=70,\n    use_sheetness=False,\n    sheetness_percentile=75,\n    blur_sigma=0.5,\n    sheet_sigma=1.0,\n    use_2p5d=True,\n    kernel_size=3\n):\n    vol = preprocess_volume(volume, blur_sigma=blur_sigma)\n\n    # Intensity mask\n    t_int = np.percentile(vol, intensity_percentile)\n    mask = vol > t_int\n\n    # Gradient magnitude mask\n    gz = np.abs(np.diff(vol, axis=0, prepend=vol[0:1]))\n    gy = np.abs(np.diff(vol, axis=1, prepend=vol[:, 0:1]))\n    gx = np.abs(np.diff(vol, axis=2, prepend=vol[:, :, 0:1]))\n    grad = np.sqrt(gz**2 + gy**2 + gx**2)\n    t_grad = np.percentile(grad, grad_percentile)\n    mask = mask & (grad > t_grad)\n\n    # Optional sheetness filter (Frangi-inspired)\n    if use_sheetness:\n        sheet = sheetness_filter(vol, sigma=sheet_sigma)\n        t_sheet = np.percentile(sheet, sheetness_percentile)\n        mask = mask & (sheet > t_sheet)\n\n    # 2.5D voting\n    if use_2p5d:\n        refined = mask.copy()\n        for z in range(vol.shape[0]):\n            z_prev = max(0, z - 1)\n            z_next = min(vol.shape[0] - 1, z + 1)\n            votes = mask[z_prev].astype(int) + mask[z].astype(int) + mask[z_next].astype(int)\n            refined[z] = votes >= 2\n        mask = refined\n\n    # Morphology cleanup\n    kernel = np.ones((kernel_size, kernel_size, kernel_size))\n    mask = ndimage.binary_closing(mask, structure=kernel)\n    mask = ndimage.binary_opening(mask, structure=kernel)\n\n    return mask.astype(np.uint8)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T04:12:25.691531Z","iopub.execute_input":"2026-02-13T04:12:25.692111Z","iopub.status.idle":"2026-02-13T04:12:25.706221Z","shell.execute_reply.started":"2026-02-13T04:12:25.692065Z","shell.execute_reply":"2026-02-13T04:12:25.705532Z"}},"outputs":[],"execution_count":null},{"id":"cdd6828f","cell_type":"markdown","source":"## Quick Test on One Volume","metadata":{}},{"id":"f666e169","cell_type":"code","source":"test_id = list(test_volumes.keys())[0]\ntest_vol = test_volumes[test_id]\n\nmask = segment_surface_unsupervised_v3(\n    test_vol,\n    intensity_percentile=65,\n    grad_percentile=70,\n    use_sheetness=False\n)\n\ncoverage = mask.mean() * 100\nprint(f\"Test volume: {test_id}\")\nprint(f\"Shape: {mask.shape}\")\nprint(f\"Coverage: {coverage:.2f}%\")\nprint(f\"Unique values: {np.unique(mask)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T04:12:25.707249Z","iopub.execute_input":"2026-02-13T04:12:25.707692Z","iopub.status.idle":"2026-02-13T04:12:33.468771Z","shell.execute_reply.started":"2026-02-13T04:12:25.707660Z","shell.execute_reply":"2026-02-13T04:12:33.468031Z"}},"outputs":[],"execution_count":null},{"id":"f1e113cb","cell_type":"markdown","source":"## Generate Predictions","metadata":{}},{"id":"9fe5b09a","cell_type":"code","source":"INTENSITY_PCT = 65\nGRAD_PCT = 70\nUSE_SHEETNESS = False\nSHEET_PCT = 75\n\npredictions = {}\nfor tid, vol in tqdm(test_volumes.items(), desc=\"Volumes\"):\n    mask = segment_surface_unsupervised_v3(\n        vol,\n        intensity_percentile=INTENSITY_PCT,\n        grad_percentile=GRAD_PCT,\n        use_sheetness=USE_SHEETNESS,\n        sheetness_percentile=SHEET_PCT\n    )\n    predictions[tid] = mask\n    print(f\"{tid}: coverage={mask.mean() * 100:.2f}%\")\n\nprint(f\"\\nGenerated {len(predictions)} predictions\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T04:12:33.469815Z","iopub.execute_input":"2026-02-13T04:12:33.470191Z","iopub.status.idle":"2026-02-13T04:12:40.401303Z","shell.execute_reply.started":"2026-02-13T04:12:33.470150Z","shell.execute_reply":"2026-02-13T04:12:40.400563Z"}},"outputs":[],"execution_count":null},{"id":"5cc01fa2","cell_type":"markdown","source":"## Save Submission (Kaggle Format)","metadata":{}},{"id":"94e007f6","cell_type":"code","source":"submission_dir = OUTPUT_PATH / 'submission'\nif submission_dir.exists():\n    shutil.rmtree(submission_dir)\nsubmission_dir.mkdir(parents=True, exist_ok=True)\n\nprint(f\"Saving to {submission_dir}...\")\nfor tid, mask in predictions.items():\n    out_path = submission_dir / f\"{tid}.tif\"\n    tiff.imwrite(str(out_path), mask.astype(np.uint8), compression=None)\n\nprint(f\"✓ Saved {len(predictions)} TIFFs\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T04:12:40.402385Z","iopub.execute_input":"2026-02-13T04:12:40.402896Z","iopub.status.idle":"2026-02-13T04:12:40.436242Z","shell.execute_reply.started":"2026-02-13T04:12:40.402854Z","shell.execute_reply":"2026-02-13T04:12:40.435105Z"}},"outputs":[],"execution_count":null},{"id":"5a5db8d8","cell_type":"code","source":"zip_path = OUTPUT_PATH / 'submission.zip'\nif zip_path.exists():\n    zip_path.unlink()\n\nwith zipfile.ZipFile(str(zip_path), 'w', zipfile.ZIP_DEFLATED) as zf:\n    for tid in predictions.keys():\n        fpath = submission_dir / f\"{tid}.tif\"\n        zf.write(str(fpath), arcname=f\"{tid}.tif\")\n\nprint(f\"Created submission: {zip_path}\")\nprint(f\"Size: {zip_path.stat().st_size / (1024 * 1024):.2f} MB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-13T04:12:40.437279Z","iopub.execute_input":"2026-02-13T04:12:40.437543Z","iopub.status.idle":"2026-02-13T04:12:41.024957Z","shell.execute_reply.started":"2026-02-13T04:12:40.437519Z","shell.execute_reply":"2026-02-13T04:12:41.024237Z"}},"outputs":[],"execution_count":null},{"id":"24a81344","cell_type":"markdown","source":"## Notes\n\nIf coverage is too low/high, adjust:\n- `INTENSITY_PCT` (60-70)\n- `GRAD_PCT` (70-85)\n- `SHEET_PCT` (75-90)\n- `USE_SHEETNESS` (False to simplify)","metadata":{}}]}