{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":117682,"databundleVersionId":15062069,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":290917305,"sourceType":"kernelVersion"},{"sourceId":732880,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":516822,"modelId":510647},{"sourceId":736297,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":561244,"modelId":573856}],"dockerImageVersionId":31193,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Training Notebooks\n\n- [Vesuvius Surface 3D Detection in Keras-JAX](https://www.kaggle.com/code/ipythonx/vesuvius-surface-3d-detection-in-jax)\n- [Vesuvius Surface 3D Detection in PyTorch](https://www.kaggle.com/code/ipythonx/vesuvius-surface-3d-detection-in-pytorch)\n- [Vesuvius Surface 3D Detection in PyTorch Lightning](https://www.kaggle.com/code/ipythonx/train-vesuvius-surface-3d-detection-in-lightning)\n- [[WIP] Vesuvius Surface 2.5D Detection](https://www.kaggle.com/code/ipythonx/wip-vesuvius-surface-2-5d-detection)\n\n**Note**\n1. The inference code below is adapted from the **Keras-JAX** version. The PyTorch and Lightning implementations follow the same workflow. Training was performed on a single Tesla T4 (16 GB VRAM) with extended epochs.\n2. Both the training and inference pipelines are implemented using [`medicai`](https://github.com/innat/medic-ai), a **Keras 3** based multi-backend medical ML library designed for 2D and 3D classification and segmentation tasks. However, please note, `medicai` project is still new and actively evolving.","metadata":{}},{"cell_type":"markdown","source":"# Inference","metadata":{}},{"cell_type":"code","source":"from IPython.display import clear_output\n\nvar=\"/kaggle/input/vsdetection-packages-offline-installer-only/whls\"\n!pip install \\\n  \"$var\"/keras_nightly-*.whl \\\n  \"$var\"/tifffile-*.whl \\\n  \"$var\"/imagecodecs-*.whl \\\n  \"$var\"/medicai-*.whl \\\n  --no-index \\\n  --find-links \"$var\"\n\nclear_output()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T15:49:17.618981Z","iopub.execute_input":"2026-01-30T15:49:17.619181Z","iopub.status.idle":"2026-01-30T15:49:25.294135Z","shell.execute_reply.started":"2026-01-30T15:49:17.619154Z","shell.execute_reply":"2026-01-30T15:49:25.29337Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nos.environ[\"KERAS_BACKEND\"] = \"jax\"\n\nimport keras\nfrom medicai.transforms import (\n    Compose,\n    ScaleIntensityRange,\n    NormalizeIntensity\n)\nfrom medicai.models import SegFormer, TransUNet, unetr_plus_plus\nfrom medicai.utils.inference import SlidingWindowInference\n\nimport numpy as np\nimport pandas as pd\nimport zipfile\nimport tifffile\nimport scipy.ndimage as ndi\nfrom skimage.morphology import remove_small_objects\nfrom matplotlib import pyplot as plt\n\nkeras.config.backend(), keras.version()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T15:49:25.29501Z","iopub.execute_input":"2026-01-30T15:49:25.295246Z","iopub.status.idle":"2026-01-30T15:49:43.357392Z","shell.execute_reply.started":"2026-01-30T15:49:25.295223Z","shell.execute_reply":"2026-01-30T15:49:43.356611Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Dataset**","metadata":{}},{"cell_type":"code","source":"root_dir = \"/kaggle/input/vesuvius-challenge-surface-detection\"\ntest_dir = f\"{root_dir}/test_images\"\noutput_dir = \"/kaggle/working/submission_masks\"\nzip_path = \"/kaggle/working/submission.zip\"\nos.makedirs(output_dir, exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T15:49:43.358256Z","iopub.execute_input":"2026-01-30T15:49:43.359054Z","iopub.status.idle":"2026-01-30T15:49:43.36268Z","shell.execute_reply.started":"2026-01-30T15:49:43.359032Z","shell.execute_reply":"2026-01-30T15:49:43.361948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df = pd.read_csv(f\"{root_dir}/test.csv\")\ntest_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T15:49:43.363398Z","iopub.execute_input":"2026-01-30T15:49:43.36364Z","iopub.status.idle":"2026-01-30T15:49:43.574329Z","shell.execute_reply.started":"2026-01-30T15:49:43.363623Z","shell.execute_reply":"2026-01-30T15:49:43.573543Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Transformation**","metadata":{}},{"cell_type":"code","source":"def val_transformation(image):\n    data = {\"image\": image}\n    pipeline = Compose([\n        NormalizeIntensity(\n            keys=[\"image\"], \n            nonzero=True,\n            channel_wise=False\n        ),\n    ])\n    result = pipeline(data)\n    return result[\"image\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T15:49:43.575172Z","iopub.execute_input":"2026-01-30T15:49:43.57573Z","iopub.status.idle":"2026-01-30T15:49:43.579749Z","shell.execute_reply.started":"2026-01-30T15:49:43.575696Z","shell.execute_reply":"2026-01-30T15:49:43.578917Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Model**","metadata":{}},{"cell_type":"code","source":"tta=1\nnum_classes=3\ninput_shape=(160, 160, 160)\nkaggle_model_path = \"/kaggle/input/vsd-model/keras/\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T15:49:43.580425Z","iopub.execute_input":"2026-01-30T15:49:43.580633Z","iopub.status.idle":"2026-01-30T15:49:43.593032Z","shell.execute_reply.started":"2026-01-30T15:49:43.580616Z","shell.execute_reply":"2026-01-30T15:49:43.592405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_model():\n    ## LB: 0.486\n    # model = SegFormer(\n    #     input_shape=(128, 128, 128, 1),\n    #     encoder_name='mit_b2',\n    #     classifier_activation='softmax',\n    #     num_classes=2,\n    # )\n    # model.load_weights(\n    #     \"/kaggle/input/vsd-model/keras/segformer.mit.b2/2/segformer.mit.b2.weights.h5\"\n    # )\n\n    ## LB: 0.5 \n    # model = TransUNet(\n    #     input_shape=(128, 128, 128, 1),\n    #     encoder_name='seresnext50',\n    #     classifier_activation='softmax',\n    #     num_classes=2,\n    # )\n    # model.load_weights(\n    #     f\"{kaggle_model_path}/transunet/2/transunet.seresnext50.128px.weights.h5\"\n    # )\n\n    # ## LB: 505\n    # model = TransUNet(\n    #     input_shape=(160, 160, 160, 1),\n    #     encoder_name='seresnext50',\n    #     classifier_activation='softmax',\n    #     num_classes=3,\n    # )\n    # model.load_weights(\n    #     f\"{kaggle_model_path}/transunet/2/transunet.seresnext50.160px.weights.h5\"\n    # )\n\n    # 0.545 (tta+pp)\n    # model = TransUNet(\n    #     input_shape=(160, 160, 160, 1),\n    #     encoder_name='seresnext50',\n    #     classifier_activation='softmax',\n    #     num_classes=3,\n    # )\n    # model.load_weights(\n    #     f\"{kaggle_model_path}/transunet/3/transunet.seresnext50.160px.comboloss.weights.h5\"\n    # )\n\n    enc = unetr_plus_plus.UNETRPlusPlusEncoder(\n        input_shape=(160,160,160,1),\n        input_tensor=None,\n        patch_size=(5,5,5),\n        filters=[32, 64, 128, 256],\n        spatial_reduced_tokens=[64, 64, 64, 64],\n        depths=[4,4,4,4],\n        num_heads=4,\n        transformer_dropout_rate=0.2,\n    )\n\n    model = unetr_plus_plus.UNETRPlusPlus(input_shape=(160,160,160,1),\n        encoder=enc,\n        num_classes=3,\n        feature_size=16,\n        norm_name='instance',\n        classifier_activation='softmax',\n        name=None)\n\n    model.load_weights(\n        \"/kaggle/input/unetr-pp-topo-loss-pretrained-d-074/keras/default/2/fine_tuning_epoch_100.weights.h5\"\n    )\n\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T15:49:43.595467Z","iopub.execute_input":"2026-01-30T15:49:43.595732Z","iopub.status.idle":"2026-01-30T15:49:43.606455Z","shell.execute_reply.started":"2026-01-30T15:49:43.595716Z","shell.execute_reply":"2026-01-30T15:49:43.605982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = get_model()\nmodel.count_params() / 1e6","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T15:49:43.607071Z","iopub.execute_input":"2026-01-30T15:49:43.607226Z","iopub.status.idle":"2026-01-30T15:49:57.548923Z","shell.execute_reply.started":"2026-01-30T15:49:43.607206Z","shell.execute_reply":"2026-01-30T15:49:57.548141Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.instance_describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T15:49:57.54988Z","iopub.execute_input":"2026-01-30T15:49:57.550385Z","iopub.status.idle":"2026-01-30T15:49:57.555307Z","shell.execute_reply.started":"2026-01-30T15:49:57.550362Z","shell.execute_reply":"2026-01-30T15:49:57.554484Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Sliding Window Inference**","metadata":{}},{"cell_type":"code","source":"swi = SlidingWindowInference(\n    model,\n    num_classes=3,\n    roi_size=input_shape,\n    sw_batch_size=1,\n    mode='gaussian',\n    overlap=0.5,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T15:49:57.556199Z","iopub.execute_input":"2026-01-30T15:49:57.557042Z","iopub.status.idle":"2026-01-30T15:49:57.584257Z","shell.execute_reply.started":"2026-01-30T15:49:57.557018Z","shell.execute_reply":"2026-01-30T15:49:57.583726Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_volume(path):\n    vol = tifffile.imread(path)\n    vol = vol.astype(np.float32)\n    vol = vol[None, ..., None]\n    return vol","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T15:49:57.585055Z","iopub.execute_input":"2026-01-30T15:49:57.585255Z","iopub.status.idle":"2026-01-30T15:49:57.599546Z","shell.execute_reply.started":"2026-01-30T15:49:57.58524Z","shell.execute_reply":"2026-01-30T15:49:57.598894Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Test Time Augmentation (TTA)**","metadata":{}},{"cell_type":"code","source":"def predict_with_tta(inputs, swi):\n    logits = []\n\n    # Original\n    logits.append(swi(inputs))\n\n    # Flips (spatial only)\n    for axis in [1, 2, 3]:\n        img_f = np.flip(inputs, axis=axis)\n        p = swi(img_f)\n        p = np.flip(p, axis=axis)\n        logits.append(p)\n\n    # Axial rotations (H, W)\n    for k in [1, 2, 3]:\n        img_r = np.rot90(inputs, k=k, axes=(2, 3))\n        p = swi(img_r)\n        p = np.rot90(p, k=-k, axes=(2, 3))\n        logits.append(p)\n\n    mean_logits = np.mean(logits, axis=0)\n    return mean_logits.argmax(-1).astype(np.uint8).squeeze()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T15:49:57.600322Z","iopub.execute_input":"2026-01-30T15:49:57.600604Z","iopub.status.idle":"2026-01-30T15:49:57.613657Z","shell.execute_reply.started":"2026-01-30T15:49:57.600581Z","shell.execute_reply":"2026-01-30T15:49:57.613052Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Post Processing**","metadata":{}},{"cell_type":"code","source":"# https://www.kaggle.com/code/choudharymanas/inference-baseline-transunet-lb-0-537\ndef build_anisotropic_struct(z_radius: int, xy_radius: int):\n    z, r = z_radius, xy_radius\n    if z == 0 and r == 0:\n        return None\n    if z == 0 and r > 0:\n        size = 2 * r + 1\n        struct = np.zeros((1, size, size), dtype=bool)\n        cy, cx = r, r\n        for dy in range(-r, r + 1):\n            for dx in range(-r, r + 1):\n                if dy * dy + dx * dx <= r * r:\n                    struct[0, cy + dy, cx + dx] = True\n        return struct\n    if z > 0 and r == 0:\n        struct = np.zeros((2 * z + 1, 1, 1), dtype=bool)\n        struct[:, 0, 0] = True\n        return struct\n    depth = 2 * z + 1\n    size = 2 * r + 1\n    struct = np.zeros((depth, size, size), dtype=bool)\n    cz, cy, cx = z, r, r\n    for dz in range(-z, z + 1):\n        for dy in range(-r, r + 1):\n            for dx in range(-r, r + 1):\n                if dy * dy + dx * dx <= r * r:\n                    struct[cz + dz, cy + dy, cx + dx] = True\n    return struct\n\ndef topo_postprocess(\n    probs,\n    T_low=0.90,\n    T_high=0.90,\n    z_radius=1,\n    xy_radius=0,\n    dust_min_size=100,\n):\n    # Step 1: 3D Hysteresis\n    strong = probs >= T_high\n    weak   = probs >= T_low\n\n    if not strong.any():\n        return np.zeros_like(probs, dtype=np.uint8)\n\n    struct_hyst = ndi.generate_binary_structure(3, 3)\n    mask = ndi.binary_propagation(\n        strong, mask=weak, structure=struct_hyst\n    )\n\n    if not mask.any():\n        return np.zeros_like(probs, dtype=np.uint8)\n\n    # Step 2: 3D Anisotropic Closing\n    if z_radius > 0 or xy_radius > 0:\n        struct_close = build_anisotropic_struct(z_radius, xy_radius)\n        if struct_close is not None:\n            mask = ndi.binary_closing(mask, structure=struct_close)\n\n    # Step 3: Dust Removal\n    if dust_min_size > 0:\n        mask = remove_small_objects(\n            mask.astype(bool), min_size=dust_min_size\n        )\n\n    return mask.astype(np.uint8)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T15:49:57.614359Z","iopub.execute_input":"2026-01-30T15:49:57.614623Z","iopub.status.idle":"2026-01-30T15:49:57.629032Z","shell.execute_reply.started":"2026-01-30T15:49:57.614601Z","shell.execute_reply":"2026-01-30T15:49:57.628392Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Prediction and Zip Submission**","metadata":{}},{"cell_type":"code","source":"def inference_pipelines(\n    volume,\n    T_low=0.3,\n    T_high=0.8,\n    z_radius=4,\n    xy_radius=4,\n    dust_min_size=200,\n):\n    probs = predict_with_tta(volume, swi)\n    final = topo_postprocess(\n        probs,\n        T_low=T_low,\n        T_high=T_high,\n        z_radius=z_radius,\n        xy_radius=xy_radius,\n        dust_min_size=dust_min_size,\n    )\n    return probs, final","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T15:49:57.62967Z","iopub.execute_input":"2026-01-30T15:49:57.629854Z","iopub.status.idle":"2026-01-30T15:49:57.647438Z","shell.execute_reply.started":"2026-01-30T15:49:57.629839Z","shell.execute_reply":"2026-01-30T15:49:57.646864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with zipfile.ZipFile(\n    zip_path, \"w\", compression=zipfile.ZIP_DEFLATED\n) as z:\n    for image_id in test_df[\"id\"]:\n        tif_path = f\"{test_dir}/{image_id}.tif\"\n        \n        volume = load_volume(tif_path)\n        volume = val_transformation(volume)\n        probs, output = inference_pipelines(volume) \n        \n        out_path = f\"{output_dir}/{image_id}.tif\"\n        tifffile.imwrite(out_path, output.astype(np.uint8))\n\n        z.write(out_path, arcname=f\"{image_id}.tif\")\n        os.remove(out_path)\n\nprint(\"Submission ZIP:\", zip_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T15:49:57.648252Z","iopub.execute_input":"2026-01-30T15:49:57.648456Z","iopub.status.idle":"2026-01-30T15:51:42.472428Z","shell.execute_reply.started":"2026-01-30T15:49:57.648442Z","shell.execute_reply":"2026-01-30T15:51:42.471535Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Sample View**","metadata":{}},{"cell_type":"code","source":"# def plot_sample(x, y, sample_idx=0, max_slices=16):\n#     img = np.squeeze(x[sample_idx])  # make (D, H, W)\n#     mask = np.squeeze(y[sample_idx])  # make (D, H, W)\n#     D = img.shape[0]\n\n#     # Decide which slices to plot\n#     step = max(1, D // max_slices)\n#     slices = range(0, D, step)\n\n#     n_slices = len(slices)\n#     fig, axes = plt.subplots(2, n_slices, figsize=(3*n_slices, 6))\n\n#     for i, s in enumerate(slices):\n#         axes[0, i].imshow(img[s], cmap='gray')\n#         axes[0, i].set_title(f\"Slice {s}\")\n#         axes[0, i].axis('off')\n\n#         axes[1, i].imshow(mask[s], cmap='gray')\n#         axes[1, i].set_title(f\"Mask {s}\")\n#         axes[1, i].axis('off')\n\n#     plt.suptitle(f\"Sample {sample_idx}\")\n#     plt.tight_layout()\n#     plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T15:51:42.476526Z","iopub.execute_input":"2026-01-30T15:51:42.477124Z","iopub.status.idle":"2026-01-30T15:51:42.48294Z","shell.execute_reply.started":"2026-01-30T15:51:42.477096Z","shell.execute_reply":"2026-01-30T15:51:42.482204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot_sample(\n#     volume.numpy(), final[None], sample_idx=0, max_slices=5\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T15:58:28.788659Z","iopub.execute_input":"2026-01-30T15:58:28.788957Z","iopub.status.idle":"2026-01-30T15:58:29.911007Z","shell.execute_reply.started":"2026-01-30T15:58:28.788919Z","shell.execute_reply":"2026-01-30T15:58:29.910126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}