{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":117682,"databundleVersionId":14443416,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Vesuvius Challenge 2025: Classical CV + TTA + Ensemble\n## Optimized Classical Methods with Test-Time Augmentation\n\n**Features:**\n- 6 Classical CV methods\n- Test-Time Augmentation (4 rotations)\n- Smart preprocessing (simple standardization)\n- Data augmentation during inference\n- Weighted ensemble blending","metadata":{}},{"cell_type":"markdown","source":"## Part 1: Imports","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\nfrom PIL import Image\nfrom io import BytesIO\nimport zipfile\nimport warnings\nwarnings.filterwarnings('ignore')\n\nfrom skimage.filters import threshold_otsu, threshold_multiotsu\nfrom skimage.morphology import remove_small_objects, binary_opening, binary_closing\nfrom skimage.measure import label\nfrom skimage.segmentation import watershed\nfrom scipy import ndimage\n\nprint('✓ Imports successful')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-17T17:45:19.248237Z","iopub.execute_input":"2025-11-17T17:45:19.248542Z","iopub.status.idle":"2025-11-17T17:45:22.159697Z","shell.execute_reply.started":"2025-11-17T17:45:19.248512Z","shell.execute_reply":"2025-11-17T17:45:22.158791Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Part 2: TIFF I/O","metadata":{}},{"cell_type":"code","source":"def load_3d_tiff(path):\n    img = Image.open(path)\n    slices = []\n    try:\n        for i in range(10000):\n            img.seek(i)\n            slices.append(np.array(img))\n    except EOFError:\n        pass\n    return np.stack(slices, axis=0)\n\ndef save_3d_tiff_to_zip(volume, zip_file, filename):\n    if volume.dtype != np.uint8:\n        volume = (np.clip(volume, 0, 1) * 255).astype(np.uint8)\n    slices = [Image.fromarray(v) for v in volume]\n    buffer = BytesIO()\n    slices[0].save(buffer, format='TIFF', save_all=True, append_images=slices[1:])\n    zip_file.writestr(filename, buffer.getvalue())\n\nprint('✓ TIFF I/O ready')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-17T17:45:22.161652Z","iopub.execute_input":"2025-11-17T17:45:22.162003Z","iopub.status.idle":"2025-11-17T17:45:22.169298Z","shell.execute_reply.started":"2025-11-17T17:45:22.161983Z","shell.execute_reply":"2025-11-17T17:45:22.168359Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Part 3: Preprocessing (Simple Standardization)","metadata":{}},{"cell_type":"code","source":"def preprocess_volume(volume):\n    \"\"\"Simple standardization - preserves contrast\"\"\"\n    volume = volume.astype(np.float32)\n    \n    # Simple standardization (NOT aggressive IQR)\n    mean = volume.mean()\n    std = volume.std() + 1e-8\n    volume = (volume - mean) / std\n    \n    # Clip to prevent extreme values\n    volume = np.clip(volume, -5, 5)\n    \n    return volume\n\ndef post_process_prediction(pred, min_size=500):\n    \"\"\"Post-processing with morphological ops\"\"\"\n    binary = (pred > 0.5).astype(np.uint8)\n    \n    footprint = np.ones((1, 3, 3))\n    binary = binary_closing(binary, footprint=footprint)\n    binary = binary_opening(binary, footprint=footprint)\n    \n    labeled = label(binary)\n    binary = remove_small_objects(labeled, min_size=min_size)\n    binary = ndimage.binary_fill_holes(binary).astype(np.uint8)\n    \n    return binary.astype(np.float32)\n\nprint('✓ Preprocessing ready')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-17T17:45:22.170711Z","iopub.execute_input":"2025-11-17T17:45:22.171295Z","iopub.status.idle":"2025-11-17T17:45:22.191911Z","shell.execute_reply.started":"2025-11-17T17:45:22.171274Z","shell.execute_reply":"2025-11-17T17:45:22.190897Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Part 4: Test-Time Augmentation (TTA)","metadata":{}},{"cell_type":"code","source":"def rotate_volume_90(volume, k):\n    \"\"\"Rotate volume by 90*k degrees\"\"\"\n    return np.rot90(volume, k=k, axes=(1, 2))\n\ndef apply_tta(volume, segmentation_func):\n    \"\"\"Test-Time Augmentation with 4 rotations\"\"\"\n    predictions = []\n    \n    # Original (0°)\n    pred0 = segmentation_func(volume)\n    predictions.append(pred0)\n    \n    # Rotate 90°\n    vol90 = rotate_volume_90(volume, k=1)\n    pred90 = segmentation_func(vol90)\n    pred90 = rotate_volume_90(pred90, k=-1)  # Rotate back\n    predictions.append(pred90)\n    \n    # Rotate 180°\n    vol180 = rotate_volume_90(volume, k=2)\n    pred180 = segmentation_func(vol180)\n    pred180 = rotate_volume_90(pred180, k=-2)  # Rotate back\n    predictions.append(pred180)\n    \n    # Rotate 270°\n    vol270 = rotate_volume_90(volume, k=3)\n    pred270 = segmentation_func(vol270)\n    pred270 = rotate_volume_90(pred270, k=-3)  # Rotate back\n    predictions.append(pred270)\n    \n    # Average all predictions\n    tta_pred = np.mean(predictions, axis=0)\n    \n    return tta_pred\n\nprint('✓ TTA ready')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-17T17:45:22.192911Z","iopub.execute_input":"2025-11-17T17:45:22.193183Z","iopub.status.idle":"2025-11-17T17:45:22.206717Z","shell.execute_reply.started":"2025-11-17T17:45:22.193156Z","shell.execute_reply":"2025-11-17T17:45:22.205681Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Part 5: Classical CV Methods","metadata":{}},{"cell_type":"code","source":"def otsu_segmentation(volume, min_size=5000):\n    try:\n        thresh = threshold_otsu(volume)\n        mask = (volume > thresh).astype(np.float32)\n    except:\n        mask = (volume > volume.mean()).astype(np.float32)\n    labeled = label(mask)\n    return remove_small_objects(labeled, min_size=min_size).astype(np.float32)\n\ndef multiotsu_segmentation(volume, min_size=5000):\n    try:\n        thresholds = threshold_multiotsu(volume, classes=3)\n        mask = (volume > thresholds[1]).astype(np.float32)\n    except:\n        return otsu_segmentation(volume, min_size)\n    labeled = label(mask)\n    return remove_small_objects(labeled, min_size=min_size).astype(np.float32)\n\ndef watershed_segmentation(volume, min_size=5000):\n    binary = (volume > np.percentile(volume, 60)).astype(np.uint8)\n    distance = ndimage.distance_transform_edt(binary)\n    local_maxima = ndimage.maximum_filter(distance, size=5) == distance\n    markers = label(local_maxima)\n    try:\n        segmented = watershed(-distance, markers, mask=binary)\n    except:\n        segmented = markers\n    segmented = remove_small_objects(segmented, min_size=min_size)\n    return (segmented > 0).astype(np.float32)\n\ndef percentile_segmentation(volume, percentile=60, min_size=5000):\n    thresh = np.percentile(volume, percentile)\n    mask = (volume > thresh).astype(np.float32)\n    labeled = label(mask)\n    return remove_small_objects(labeled, min_size=min_size).astype(np.float32)\n\ndef morphological_segmentation(volume, percentile=60, min_size=5000):\n    thresh = np.percentile(volume, percentile)\n    binary = (volume > thresh).astype(np.uint8)\n    footprint = np.ones((1, 3, 3))\n    binary = binary_opening(binary, footprint=footprint)\n    binary = binary_closing(binary, footprint=footprint)\n    labeled = label(binary)\n    return remove_small_objects(labeled, min_size=min_size).astype(np.float32)\n\ndef adaptive_segmentation(volume, block_size=32, min_size=5000):\n    result = np.zeros_like(volume, dtype=np.float32)\n    for z in range(volume.shape[0]):\n        slice_2d = volume[z]\n        local_mean = ndimage.uniform_filter(slice_2d.astype(float), size=block_size)\n        binary = (slice_2d > local_mean).astype(np.uint8)\n        result[z] = binary\n    labeled = label(result)\n    return remove_small_objects(labeled, min_size=min_size).astype(np.float32)\n\nprint('✓ All 6 classical methods ready')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-17T17:45:22.207657Z","iopub.execute_input":"2025-11-17T17:45:22.207970Z","iopub.status.idle":"2025-11-17T17:45:22.222588Z","shell.execute_reply.started":"2025-11-17T17:45:22.207941Z","shell.execute_reply":"2025-11-17T17:45:22.221659Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Part 6: Ensemble with TTA","metadata":{}},{"cell_type":"code","source":"def ensemble_predict_with_tta(volume):\n    \"\"\"Ensemble of 6 methods with TTA\"\"\"\n    weights = {\n        #'multiotsu': 0.25,\n        'watershed': 1.0,\n        #'morph': 0.25,\n        #'adaptive': 0.25,\n    }\n    \n    # Normalize to uint8 for classical methods\n    volume_norm = ((volume - volume.min()) / (volume.max() - volume.min() + 1e-8) * 255).astype(np.uint8)\n    \n    predictions = {}\n    \n    print('  Running ensemble with TTA (4 rotations per method)...')\n    \n    # 1. Otsu with TTA\n    #print('    [1/6] Otsu + TTA...')\n    #predictions['otsu'] = apply_tta(volume_norm, otsu_segmentation)\n    \n    # 2. Multi-Otsu with TTA\n    #print('    [2/6] Multi-Otsu + TTA...')\n    #predictions['multiotsu'] = apply_tta(volume_norm, multiotsu_segmentation)\n    \n    # 3. Watershed with TTA\n    print('    [3/6] Watershed + TTA...')\n    predictions['watershed'] = apply_tta(volume_norm, watershed_segmentation)\n    \n    # 4. Percentile with TTA\n    #print('    [4/6] Percentile + TTA...')\n    #predictions['percentile'] = apply_tta(volume_norm, percentile_segmentation)\n    \n    # 5. Morphological with TTA\n    #print('    [5/6] Morphological + TTA...')\n    #predictions['morph'] = apply_tta(volume_norm, morphological_segmentation)\n    \n    # 6. Adaptive with TTA\n    #print('    [6/6] Adaptive + TTA...')\n    #predictions['adaptive'] = apply_tta(volume_norm, adaptive_segmentation)\n    \n    # Blend with weights\n    print('  Blending predictions...')\n    ensemble_pred = np.zeros_like(volume, dtype=np.float32)\n    total_weight = sum(weights.values())\n    \n    for method, pred in predictions.items():\n        w = weights[method] / total_weight\n        ensemble_pred += w * pred\n    \n    return ensemble_pred\n\nprint('✓ Ensemble with TTA ready')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-17T17:45:22.223674Z","iopub.execute_input":"2025-11-17T17:45:22.224246Z","iopub.status.idle":"2025-11-17T17:45:22.242523Z","shell.execute_reply.started":"2025-11-17T17:45:22.224217Z","shell.execute_reply":"2025-11-17T17:45:22.241506Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Part 7: Load Test Data","metadata":{}},{"cell_type":"code","source":"ROOT = Path('/kaggle/input/vesuvius-challenge-surface-detection')\ntest_df = pd.read_csv(ROOT / 'test.csv')\ntest_imgs = ROOT / 'test_images'\nprint(f'✓ Test data: {len(test_df)} volumes')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-17T17:45:22.244958Z","iopub.execute_input":"2025-11-17T17:45:22.245434Z","iopub.status.idle":"2025-11-17T17:45:22.273407Z","shell.execute_reply.started":"2025-11-17T17:45:22.245400Z","shell.execute_reply":"2025-11-17T17:45:22.272620Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Part 8: Inference & Submission","metadata":{}},{"cell_type":"code","source":"print(f'\\n📊 Processing {len(test_df)} volumes with TTA + Ensemble...\\n')\nsubmission_zip_path = Path('/kaggle/working/submission.zip')\n\nwith zipfile.ZipFile(submission_zip_path, 'w', compression=zipfile.ZIP_DEFLATED, compresslevel=9) as zf:\n    for idx, row in tqdm(test_df.iterrows(), total=len(test_df), desc='Processing'):\n        image_id = row['id']\n        img_path = test_imgs / f'{image_id}.tif'\n        \n        if not img_path.exists():\n            continue\n        \n        try:\n            # Load volume\n            volume = load_3d_tiff(img_path).astype(np.float32)\n            \n            # Preprocess\n            volume = preprocess_volume(volume)\n            \n            # Normalize to 0-255 for classical methods\n            volume = ((volume + 5) / 10 * 255).astype(np.uint8)\n            \n            # Ensemble with TTA\n            prob = ensemble_predict_with_tta(volume)\n            \n            # Post-process\n            mask = post_process_prediction(prob)\n            \n            # Save\n            save_3d_tiff_to_zip(mask, zf, f'{image_id}.tif')\n            \n        except Exception as e:\n            print(f'  ❌ Error {image_id}: {str(e)[:50]}')\n            continue\n\nprint(f'\\n✅ Submission: {submission_zip_path.stat().st_size / 1e6:.1f} MB')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-17T17:45:22.274356Z","iopub.execute_input":"2025-11-17T17:45:22.274836Z","iopub.status.idle":"2025-11-17T17:47:06.150550Z","shell.execute_reply.started":"2025-11-17T17:45:22.274802Z","shell.execute_reply":"2025-11-17T17:47:06.149861Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Part 9: Verify","metadata":{}},{"cell_type":"code","source":"print('\\n🔍 Verifying submission...\\n')\n\nwith zipfile.ZipFile(submission_zip_path, 'r') as zf:\n    file_list = zf.namelist()\n    print(f'✓ Files: {len(file_list)} / {len(test_df)}')\n    \n    if len(file_list) == len(test_df):\n        print('✓ Perfect match!')\n    \n    print('\\nSample files:')\n    for fname in file_list[:3]:\n        info = zf.getinfo(fname)\n        print(f'  {fname}: {info.file_size:,} bytes')\n\nprint('\\n✅ Ready to submit!')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-17T17:47:06.151215Z","iopub.execute_input":"2025-11-17T17:47:06.151398Z","iopub.status.idle":"2025-11-17T17:47:06.158510Z","shell.execute_reply.started":"2025-11-17T17:47:06.151383Z","shell.execute_reply":"2025-11-17T17:47:06.157795Z"}},"outputs":[],"execution_count":null}]}