{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":117682,"databundleVersionId":14443416,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Vesuvius Challenge 2025: Multi-Otsu Thresholding\n## Classical CV - Single Method Pipeline\n\n**Method:** Multi-Level Otsu Thresholding\n- Multiple threshold levels (3 classes)\n- Better for complex histograms\n- Enhanced segmentation","metadata":{}},{"cell_type":"markdown","source":"## Part 1: Imports","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\nfrom PIL import Image\nfrom io import BytesIO\nimport zipfile\nimport warnings\nwarnings.filterwarnings('ignore')\n\nfrom skimage.filters import threshold_multiotsu\nfrom skimage.morphology import remove_small_objects, binary_opening, binary_closing\nfrom skimage.measure import label\nfrom scipy import ndimage\n\nprint('✓ Imports successful')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T07:35:21.544634Z","iopub.execute_input":"2025-11-16T07:35:21.545431Z","iopub.status.idle":"2025-11-16T07:35:23.089333Z","shell.execute_reply.started":"2025-11-16T07:35:21.545399Z","shell.execute_reply":"2025-11-16T07:35:23.088505Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Part 2: TIFF I/O","metadata":{}},{"cell_type":"code","source":"def load_3d_tiff(path):\n    img = Image.open(path)\n    slices = []\n    try:\n        for i in range(10000):\n            img.seek(i)\n            slices.append(np.array(img))\n    except EOFError:\n        pass\n    return np.stack(slices, axis=0)\n\ndef save_3d_tiff_to_zip(volume, zip_file, filename):\n    if volume.dtype != np.uint8:\n        volume = (np.clip(volume, 0, 1) * 255).astype(np.uint8)\n    slices = [Image.fromarray(v) for v in volume]\n    buffer = BytesIO()\n    slices[0].save(buffer, format='TIFF', save_all=True, append_images=slices[1:])\n    zip_file.writestr(filename, buffer.getvalue())\n\nprint('✓ TIFF I/O ready')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T07:35:23.090490Z","iopub.execute_input":"2025-11-16T07:35:23.090835Z","iopub.status.idle":"2025-11-16T07:35:23.097588Z","shell.execute_reply.started":"2025-11-16T07:35:23.090817Z","shell.execute_reply":"2025-11-16T07:35:23.096443Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Part 3: Preprocessing","metadata":{}},{"cell_type":"code","source":"def preprocess_volume(volume):\n    volume = volume.astype(np.float32)\n    p1, p99 = np.percentile(volume, [1, 99])\n    volume = np.clip(volume, p1, p99)\n    median = np.median(volume)\n    q1, q3 = np.percentile(volume, [25, 75])\n    iqr = q3 - q1\n    if iqr > 0:\n        volume = (volume - median) / iqr\n    else:\n        volume = (volume - volume.mean()) / (volume.std() + 1e-8)\n    volume = np.clip(volume, -5, 5)\n    return volume\n\ndef post_process_prediction(pred, min_size=500):\n    binary = (pred > 0.5).astype(np.uint8)\n    footprint = np.ones((1, 3, 3))\n    binary = binary_closing(binary, footprint=footprint)\n    binary = binary_opening(binary, footprint=footprint)\n    labeled = label(binary)\n    binary = remove_small_objects(labeled, min_size=min_size)\n    binary = ndimage.binary_fill_holes(binary).astype(np.uint8)\n    return binary.astype(np.float32)\n\nprint('✓ Preprocessing ready')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T07:35:23.098523Z","iopub.execute_input":"2025-11-16T07:35:23.098875Z","iopub.status.idle":"2025-11-16T07:35:23.126639Z","shell.execute_reply.started":"2025-11-16T07:35:23.098850Z","shell.execute_reply":"2025-11-16T07:35:23.125867Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Part 4: Multi-Otsu Thresholding","metadata":{}},{"cell_type":"code","source":"def multiotsu_segmentation(volume, min_size=5000):\n    \"\"\"Multi-level Otsu thresholding\"\"\"\n    try:\n        thresholds = threshold_multiotsu(volume, classes=3)\n        mask = (volume > thresholds[1]).astype(np.float32)\n    except:\n        mask = (volume > volume.mean()).astype(np.float32)\n    labeled = label(mask)\n    return remove_small_objects(labeled, min_size=min_size).astype(np.float32)\n\nprint('✓ Multi-Otsu method ready')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T07:35:23.128302Z","iopub.execute_input":"2025-11-16T07:35:23.128527Z","iopub.status.idle":"2025-11-16T07:35:23.137540Z","shell.execute_reply.started":"2025-11-16T07:35:23.128502Z","shell.execute_reply":"2025-11-16T07:35:23.136772Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Part 5: Load Test Data","metadata":{}},{"cell_type":"code","source":"ROOT = Path('/kaggle/input/vesuvius-challenge-surface-detection')\ntest_df = pd.read_csv(ROOT / 'test.csv')\ntest_imgs = ROOT / 'test_images'\nprint(f'✓ Test data: {len(test_df)} volumes')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T07:35:23.138329Z","iopub.execute_input":"2025-11-16T07:35:23.138602Z","iopub.status.idle":"2025-11-16T07:35:23.167157Z","shell.execute_reply.started":"2025-11-16T07:35:23.138576Z","shell.execute_reply":"2025-11-16T07:35:23.166430Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Part 6: Inference & Submission","metadata":{}},{"cell_type":"code","source":"print(f'\\n📊 Processing {len(test_df)} volumes...\\n')\nsubmission_zip_path = Path('/kaggle/working/submission.zip')\n\nwith zipfile.ZipFile(submission_zip_path, 'w', compression=zipfile.ZIP_DEFLATED, compresslevel=9) as zf:\n    for idx, row in tqdm(test_df.iterrows(), total=len(test_df)):\n        image_id = row['id']\n        img_path = test_imgs / f'{image_id}.tif'\n        \n        if not img_path.exists():\n            continue\n        try:\n            volume = load_3d_tiff(img_path).astype(np.float32)\n            volume = preprocess_volume(volume)\n            volume = ((volume + 5) / 10 * 255).astype(np.uint8)\n            prob = multiotsu_segmentation(volume)\n            mask = post_process_prediction(prob)\n            save_3d_tiff_to_zip(mask, zf, f'{image_id}.tif')\n        except Exception as e:\n            continue\n\nprint(f'\\n✅ Submission: {submission_zip_path.stat().st_size / 1e6:.1f} MB')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T07:35:23.167907Z","iopub.execute_input":"2025-11-16T07:35:23.168170Z","iopub.status.idle":"2025-11-16T07:35:39.693801Z","shell.execute_reply.started":"2025-11-16T07:35:23.168151Z","shell.execute_reply":"2025-11-16T07:35:39.693009Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Part 7: Verify","metadata":{}},{"cell_type":"code","source":"print('\\n🔍 Verifying...\\n')\nwith zipfile.ZipFile(submission_zip_path, 'r') as zf:\n    file_list = zf.namelist()\n    print(f'✓ Files: {len(file_list)} / {len(test_df)}')\n    if len(file_list) == len(test_df):\n        print('✓ Perfect match!')\nprint('\\n✅ Ready to submit!')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T07:35:39.694574Z","iopub.execute_input":"2025-11-16T07:35:39.694784Z","iopub.status.idle":"2025-11-16T07:35:39.701076Z","shell.execute_reply.started":"2025-11-16T07:35:39.694757Z","shell.execute_reply":"2025-11-16T07:35:39.700391Z"}},"outputs":[],"execution_count":null}]}