{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":117682,"databundleVersionId":14443416,"sourceType":"competition"}],"dockerImageVersionId":31234,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install imagecodecs\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-20T06:29:45.985001Z","iopub.execute_input":"2025-12-20T06:29:45.985407Z","iopub.status.idle":"2025-12-20T06:29:53.267234Z","shell.execute_reply.started":"2025-12-20T06:29:45.985373Z","shell.execute_reply":"2025-12-20T06:29:53.265738Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# AIM\nThe aim of this code is extract 2d 64x64 Patches without any unlablled regions, can be tweeked to extract 3d patches. ","metadata":{}},{"cell_type":"code","source":"# NETracer 2D Patch Extraction Pipeline\n# ------------------------------------------------------------\n# This script prepares slice-wise 2D patches from 3D volumes\n# for NETracer-style training.\n# It strictly follows the user-specified instructions.\n# ------------------------------------------------------------\nimport imagecodecs\nimport os\nimport numpy as np\nimport tifffile as tiff\nfrom tqdm import tqdm\nfrom typing import List, Tuple\n\n# ============================================================\n# ------------------- Hyperparameters ------------------------\n# ============================================================\n\nRANDOM_SEED = 42              # fixed random state for reproducibility\nNUM_VOLUMES = 100             # number of volumes to sample\nPATCH_SIZE = 64               # 64x64 crops\nMIN_SLICES_PER_VOLUME = 20    # minimum slices to take from each volume\nMAX_PATCHES_PER_VOLUME = 300  # maximum patches per volume\nFOREGROUND_THRESHOLD = 10     # minimum number of foreground pixels (label==1)\n\n# ============================================================\n# ------------------- Utility Functions ----------------------\n# ============================================================\n\n\ndef list_tif_files(root: str) -> List[str]:\n    \"\"\"List .tif files in a directory, sorted.\"\"\"\n    files = [os.path.join(root, f) for f in os.listdir(root) if f.endswith('.tif')]\n    files.sort()\n    return files\n\n\ndef random_center_crop(h: int, w: int, crop: int, rng: np.random.Generator) -> Tuple[int, int]:\n    \"\"\"Return random top-left (y, x) for a crop.\"\"\"\n    y = rng.integers(0, h - crop + 1)\n    x = rng.integers(0, w - crop + 1)\n    return y, x\n\n\n# ============================================================\n# ------------------- Core Pipeline --------------------------\n# ============================================================\n\n\ndef extract_patches(\n    train_volume_dir: str,\n    train_label_dir: str,\n    output_dir: str,\n    num_volumes: int = NUM_VOLUMES,\n    min_slices_per_volume: int = MIN_SLICES_PER_VOLUME,\n    max_patches_per_volume: int = MAX_PATCHES_PER_VOLUME,\n    patch_size: int = PATCH_SIZE,\n):\n    \"\"\"\n    Pipeline that extracts 64x64 patches from 3D volumes for NETracer.\n\n    INPUTS:\n        train_volume_dir : path to volume .tif files\n        train_label_dir  : path to label .tif files (0,1,2)\n        output_dir       : root output directory\n    \"\"\"\n\n    os.makedirs(output_dir, exist_ok=True)\n    img_out = os.path.join(output_dir, 'images')\n    lbl_out = os.path.join(output_dir, 'labels')\n    os.makedirs(img_out, exist_ok=True)\n    os.makedirs(lbl_out, exist_ok=True)\n\n    rng = np.random.default_rng(RANDOM_SEED)\n\n    # --------------------------------------------------------\n    # 1. Prepare list of volumes (reproducible)\n    # --------------------------------------------------------\n    volume_paths = list_tif_files(train_volume_dir)\n    label_paths = list_tif_files(train_label_dir)\n\n    assert len(volume_paths) == len(label_paths), \"Volume/Label count mismatch\"\n\n    indices = np.arange(len(volume_paths))\n    rng.shuffle(indices)\n    selected_indices = indices[:num_volumes]\n\n    selected_volumes = [(volume_paths[i], label_paths[i]) for i in selected_indices]\n\n    # --------------------------------------------------------\n    # 2. Iterate over selected volumes\n    # --------------------------------------------------------\n    for vol_idx, (vol_path, lbl_path) in enumerate(tqdm(selected_volumes, desc='Volumes')):\n\n        vol_id = os.path.splitext(os.path.basename(vol_path))[0]\n\n        # ----------------------------------------------------\n        # 3. Load volume + label into RAM\n        # ----------------------------------------------------\n        volume = tiff.imread(vol_path)        # (Z, H, W)\n        label = tiff.imread(lbl_path)          # (Z, H, W), values in {0,1,2}\n\n        Z, H, W = volume.shape\n\n        used_slices = set()\n        patches_saved = 0\n        slices_used = 0\n\n        # ----------------------------------------------------\n        # 4–6. Random slice & patch sampling\n        # ----------------------------------------------------\n        pbar = tqdm(total=max_patches_per_volume, desc=f'Patches[{vol_id}]', leave=False)\n\n        while (\n            slices_used < min_slices_per_volume\n            and patches_saved < max_patches_per_volume\n        ):\n            # pick a random slice not used yet\n            z = rng.integers(0, Z)\n            if z in used_slices:\n                continue\n\n            img_slice = volume[z]\n            lbl_slice = label[z]\n\n            used_slices.add(z)\n            slices_used += 1\n\n            # try multiple crops per slice\n            for _ in range(50):\n                if patches_saved >= max_patches_per_volume:\n                    break\n\n                y, x = random_center_crop(H, W, patch_size, rng)\n\n                img_patch = img_slice[y:y+patch_size, x:x+patch_size]\n                lbl_patch = lbl_slice[y:y+patch_size, x:x+patch_size]\n\n                # ------------------------------------------------\n                # 5. Filtering conditions\n                # ------------------------------------------------\n                if np.any(lbl_patch == 2):\n                    continue\n\n                if np.sum(lbl_patch == 1) < FOREGROUND_THRESHOLD:\n                    continue\n\n                # ------------------------------------------------\n                # 7. Save patch (.npy)\n                # ------------------------------------------------\n                name = f\"{vol_id}_z{z}_p{patches_saved}.npy\"\n                np.save(os.path.join(img_out, name), img_patch.astype(np.float32))\n                np.save(os.path.join(lbl_out, name), lbl_patch.astype(np.uint8))\n\n                patches_saved += 1\n                pbar.update(1)\n\n            if slices_used >= min_slices_per_volume:\n                break\n\n        pbar.close()\n\n        # ----------------------------------------------------\n        # 8. Unload volume from RAM\n        # ----------------------------------------------------\n        del volume, label\n\n    print('Patch extraction completed.')\n\n\n# ============================================================\n# ------------------- Example Usage --------------------------\n# ============================================================\n\nif __name__ == '__main__':\n    extract_patches(\n        train_volume_dir='/kaggle/input/vesuvius-challenge-surface-detection/train_images',\n        train_label_dir='/kaggle/input/vesuvius-challenge-surface-detection/train_labels',\n        output_dir='/kaggle/working/',\n    )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T06:29:53.270325Z","iopub.execute_input":"2025-12-20T06:29:53.270675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"l = os.listdir(\"/kaggle/working/images\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T06:34:22.865667Z","iopub.execute_input":"2025-12-20T06:34:22.866031Z","iopub.status.idle":"2025-12-20T06:34:22.885767Z","shell.execute_reply.started":"2025-12-20T06:34:22.866003Z","shell.execute_reply":"2025-12-20T06:34:22.884774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(l)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T06:34:26.185409Z","iopub.execute_input":"2025-12-20T06:34:26.185787Z","iopub.status.idle":"2025-12-20T06:34:26.192587Z","shell.execute_reply.started":"2025-12-20T06:34:26.185756Z","shell.execute_reply":"2025-12-20T06:34:26.191567Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}