{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":117682,"databundleVersionId":14443416,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Speed up data loading\nAs I started working on an UNet baseline, I found the speed to be extremely slow. I realized the bottleneck is not the models but rather the I/O. It takes a very long time to read a tiff file. So it seemed a good idea to think about alternative formats to save the data in. `.npy` is always a great option in these scenarios. But first I wanted to compare a few file formats on the basis of loading speed and file size.\n\nThe files saved from this notebook in `train_images_npy` and `train_images_rle` can directly be used for a faster training experience. To load the `RLE` encoded labels, use the `load_rle` function from the notebook.\n\nIn actual training epoch, this data is giving about ~8-10x speedup.","metadata":{}},{"cell_type":"code","source":"!pip install -q imagecodecs zarr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T02:24:56.906200Z","iopub.execute_input":"2025-11-15T02:24:56.906810Z","iopub.status.idle":"2025-11-15T02:25:01.006575Z","shell.execute_reply.started":"2025-11-15T02:24:56.906783Z","shell.execute_reply":"2025-11-15T02:25:01.005132Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom pathlib import Path\nimport numpy as np\nimport tifffile as tiff\nfrom tqdm import tqdm\nimport zarr\nimport time\nfrom concurrent.futures import ProcessPoolExecutor\n\n\nDATA_DIR = Path(\"/kaggle/input/vesuvius-challenge-surface-detection\")\nIMG_DIR = DATA_DIR / \"train_images\"\nLBL_DIR = DATA_DIR / \"train_labels\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-15T02:25:01.008770Z","iopub.execute_input":"2025-11-15T02:25:01.009125Z","iopub.status.idle":"2025-11-15T02:25:01.015480Z","shell.execute_reply.started":"2025-11-15T02:25:01.009091Z","shell.execute_reply":"2025-11-15T02:25:01.014452Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -----------------------------\n# File I/O functions\n# -----------------------------\n\ndef save_npy(array, path):\n    np.save(path, array.astype(np.uint8))\n\ndef save_npz(array, path):\n    np.savez_compressed(path, array.astype(np.uint8))\n\ndef save_zarr(array, path):\n    z = zarr.open(str(path), mode='w', shape=array.shape, chunks=(64,64,64), dtype=array.dtype)\n    z[:] = array.astype(np.uint8)\n\ndef rle_encode_3d(mask):\n    \"\"\"\n    mask: 3D numpy array of int/uint8\n    Returns: dict with 'shape', 'vals', 'runs'\n    \"\"\"\n    flat = mask.ravel()\n\n    # Where values change\n    changes = np.where(np.diff(flat) != 0)[0] + 1\n\n    # Start positions of each run\n    starts = np.concatenate(([0], changes))\n\n    # Run lengths\n    ends = np.concatenate((changes, [len(flat)]))\n    runs = ends - starts\n\n    # Correct values at each run start\n    vals = flat[starts]\n\n    return {\n        \"shape\": mask.shape,\n        \"vals\": vals.astype(np.uint8),\n        \"runs\": runs.astype(np.uint32),\n    }\n\ndef save_rle(mask, path):\n    \"\"\"\n    Saves RLE to .npz\n    \"\"\"\n    rle = rle_encode_3d(mask)\n    np.savez_compressed(\n        path,\n        shape=np.array(rle[\"shape\"], dtype=np.int32),\n        vals=rle[\"vals\"],\n        runs=rle[\"runs\"],\n    )\n\ndef load_rle(path):\n    rle = np.load(path)\n    shape = tuple(rle[\"shape\"])\n    vals = rle[\"vals\"]\n    runs = rle[\"runs\"]\n    flat = np.repeat(vals, runs)\n    return flat.reshape(shape)\n\ndef load_array(path, fmt):\n    if fmt == \"tiff\":\n        return tiff.imread(path)\n    elif fmt == \"npy\":\n        return np.load(path)\n    elif fmt == \"npz\":\n        return np.load(path)[\"arr_0\"]\n    elif fmt == \"zarr\":\n        return zarr.open(str(path), mode='r')[:]\n    elif fmt == \"rle\":\n        return load_rle(path)\n    else:\n        raise ValueError(fmt)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T07:35:39.225533Z","iopub.execute_input":"2025-11-15T07:35:39.226799Z","iopub.status.idle":"2025-11-15T07:35:39.239034Z","shell.execute_reply.started":"2025-11-15T07:35:39.226760Z","shell.execute_reply":"2025-11-15T07:35:39.237733Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"formats = [\"tiff\", \"npy\", \"npz\", \"zarr\"]\n\nIMG_LIST = sorted([f for f in os.listdir(IMG_DIR) if f.endswith(\".tif\")])[:10]\nOUT_DIR = Path(\"/kaggle/working/format_test_images\")\nOUT_DIR.mkdir(exist_ok=True)\n\n# -----------------------------\n# Step 1: Save all formats\n# -----------------------------\nprint(f\"Saving files in different formats... {formats}\")\nfor i, f in enumerate(IMG_LIST):\n    tiff_path = IMG_DIR / f\n    array = tiff.imread(tiff_path)\n\n    save_npy(array, OUT_DIR / f\"{i}_npy.npy\")\n    save_npz(array, OUT_DIR / f\"{i}_npz.npz\")\n    save_zarr(array, OUT_DIR / f\"{i}_zarr.zarr\")\n\n# -----------------------------\n# Step 2: Compare sizes\n# -----------------------------\nfile_sizes = {}\n\nfor fmt in formats:\n    if fmt == \"tiff\":\n        total_size = sum((IMG_DIR / f).stat().st_size for f in IMG_LIST)\n    else:\n        total_size = sum(\n            (OUT_DIR / f).stat().st_size if (OUT_DIR / f).is_file() \n            else sum(p.stat().st_size for p in (OUT_DIR / f).rglob(\"*\")) \n            for f in os.listdir(OUT_DIR) if fmt in f\n        )\n    file_sizes[fmt] = total_size / 10e6\n\n# -----------------------------\n# Step 3: Load timing\n# -----------------------------\nprint(\"Calculating load times per format...\\n\")\ntimings = {fmt: [] for fmt in formats}\n\nfor fmt in formats:\n    if fmt == \"tiff\":\n        files = [IMG_DIR / f for f in IMG_LIST]\n    else:\n        files = [OUT_DIR / f for f in os.listdir(OUT_DIR) if fmt in f]\n        \n    for _ in range(5):\n        start = time.time()\n        for f in files:\n            arr = load_array(f, fmt)\n            assert isinstance(arr, np.ndarray)\n        timings[fmt].append(time.time() - start)\n\nload_times = {fmt: float(np.mean(timings[fmt])) for fmt in formats}\n\n# -----------------------------\n# Step 4: Display results in a table\n# -----------------------------\nimport pandas as pd\n\ndf = pd.DataFrame({\n    \"Format\": formats,\n    \"Avg File Size (MB)\": [file_sizes[f] for f in formats],\n    \"Avg Load Time (sec)\": [load_times[f] for f in formats],\n})\n\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T07:58:03.835713Z","iopub.execute_input":"2025-11-15T07:58:03.836273Z","iopub.status.idle":"2025-11-15T07:59:05.361018Z","shell.execute_reply.started":"2025-11-15T07:58:03.836228Z","shell.execute_reply":"2025-11-15T07:59:05.359759Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"So `.npy` is clearly the best choice for the image files. It's ~25x faster for loading. The size is also similar to the tiff file.","metadata":{}},{"cell_type":"code","source":"formats = [\"tiff\", \"npy\", \"zarr\", \"rle\"]\n\nLBL_LIST = sorted([f for f in os.listdir(LBL_DIR) if f.endswith(\".tif\")])[:10]\nOUT_DIR = Path(\"/kaggle/working/format_test_labels\")\nOUT_DIR.mkdir(exist_ok=True)\n\n# -----------------------------\n# Step 1: Save all formats\n# -----------------------------\nprint(f\"Saving files in different formats... {formats}\")\nfor i, f in enumerate(LBL_LIST):\n    tiff_path = LBL_DIR / f\n    array = tiff.imread(tiff_path)\n\n    save_npy(array, OUT_DIR / f\"{i}_npy.npy\")\n    save_npz(array, OUT_DIR / f\"{i}_npz.npz\")\n    save_zarr(array, OUT_DIR / f\"{i}_zarr.zarr\")\n    save_rle(array, OUT_DIR / f\"{i}_rle.npz\")\n\n# -----------------------------\n# Step 2: Compare sizes\n# -----------------------------\nfile_sizes = {}\n\nfor fmt in formats:\n    if fmt == \"tiff\":\n        total_size = sum((LBL_DIR / f).stat().st_size for f in LBL_LIST)\n    else:\n        total_size = sum(\n            (OUT_DIR / f).stat().st_size if (OUT_DIR / f).is_file() \n            else sum(p.stat().st_size for p in (OUT_DIR / f).rglob(\"*\")) \n            for f in os.listdir(OUT_DIR) if fmt in f\n        )\n    file_sizes[fmt] = total_size / 10e6\n\n# -----------------------------\n# Step 3: Load timing including original TIFFs\n# -----------------------------\nprint(\"Calculating load times per format...\\n\")\ntimings = {fmt: [] for fmt in formats}\n\nfor fmt in formats:\n    if fmt == \"tiff\":\n        files = [LBL_DIR / f for f in LBL_LIST]\n    else:\n        files = [OUT_DIR / f for f in os.listdir(OUT_DIR) if fmt in f]\n        \n    for _ in range(5):\n        start = time.time()\n        for f in files:\n            arr = load_array(f, fmt)\n        timings[fmt].append(time.time() - start)\n\nload_times = {fmt: float(np.mean(timings[fmt])) for fmt in formats}\n\n# -----------------------------\n# Step 4: Display results in a table\n# -----------------------------\nimport pandas as pd\n\ndf = pd.DataFrame({\n    \"Format\": formats,\n    \"Avg File Size (MB)\": [file_sizes[f] for f in formats],\n    \"Avg Load Time (sec)\": [load_times[f] for f in formats],\n})\n\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T08:38:09.594413Z","iopub.execute_input":"2025-11-15T08:38:09.594855Z","iopub.status.idle":"2025-11-15T08:38:39.362544Z","shell.execute_reply.started":"2025-11-15T08:38:09.594817Z","shell.execute_reply":"2025-11-15T08:38:39.361557Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"`.npy` is again the best here in terms of speed. But it is very inefficient in terms of memory. Most of the info contained in a label mask should not take up that much data. So I choose RLE (Run Length Encoding) format as it provides the best of both worlds.","metadata":{}},{"cell_type":"code","source":"!rm -rf /kaggle/working/*","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T08:41:04.617643Z","iopub.execute_input":"2025-11-15T08:41:04.619066Z","iopub.status.idle":"2025-11-15T08:41:04.940554Z","shell.execute_reply.started":"2025-11-15T08:41:04.619012Z","shell.execute_reply":"2025-11-15T08:41:04.938690Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Resize images and labels to 256x256x256 and save\nTo save a bit more space and shave off a few more miliseconds, we will downsample the images and masks to 256x256x256. It will have a negligible impact on quality but cut down the file size to half.","metadata":{}},{"cell_type":"code","source":"NPY_IMG_DIR = Path(\"/kaggle/working/train_images_npy\")\nNPY_LBL_DIR = Path(\"/kaggle/working/train_labels_rle\")\nNPY_IMG_DIR.mkdir(exist_ok=True)\nNPY_LBL_DIR.mkdir(exist_ok=True)\n\nimport numpy as np\nfrom pathlib import Path\nimport os\nfrom tqdm import tqdm\nfrom concurrent.futures import ProcessPoolExecutor\nimport tifffile as tiff\nfrom scipy.ndimage import zoom\n\n# -----------------------------\n# Single file processing\n# -----------------------------\ndef process_file(args, file_type=\"image\"):\n    tiff_path, npy_dir, target_shape = args\n    array = tiff.imread(tiff_path)\n    factors = [t / s for t, s in zip(target_shape, array.shape)]\n\n    if file_type == \"image\":\n        array_ds = zoom(array, factors, order=1).astype(np.uint8)\n        npy_path = NPY_IMG_DIR / (tiff_path.stem + \".npy\")\n        save_npy(array_ds, npy_path)\n    else:\n        array_ds = np.round(zoom(array, factors, order=1)).astype(np.uint8)\n        rle_path = Path(npy_dir) / (tiff_path.stem + \".npz\")\n        save_rle(array_ds, rle_path)\n\n# -----------------------------\n# Parallel processing\n# -----------------------------\ndef preprocess_and_save_tiffs_parallel(tiff_dir, npy_dir, file_type=\"image\", target_shape=(256,256,256), max_workers=4):\n    tiff_dir = Path(tiff_dir)\n    npy_dir = Path(npy_dir)\n    npy_dir.mkdir(exist_ok=True)\n\n    files = sorted([f for f in os.listdir(tiff_dir) if f.endswith(\".tif\")])\n    paths = [(tiff_dir / f, npy_dir, target_shape) for f in files]\n\n    with ProcessPoolExecutor(max_workers=max_workers) as executor:\n        list(tqdm(executor.map(process_file, paths, [file_type for _ in paths]), total=len(paths)))\n\n\npreprocess_and_save_tiffs_parallel(LBL_DIR, NPY_LBL_DIR, file_type=\"label\", target_shape=(256,256,256), max_workers=8)\npreprocess_and_save_tiffs_parallel(IMG_DIR, NPY_IMG_DIR, file_type=\"image\", target_shape=(256,256,256), max_workers=8)\n\nprint(\"Conversion completed!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T08:01:07.722880Z","iopub.execute_input":"2025-11-15T08:01:07.723274Z","iopub.status.idle":"2025-11-15T08:22:13.812845Z","shell.execute_reply.started":"2025-11-15T08:01:07.723249Z","shell.execute_reply":"2025-11-15T08:22:13.810374Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Comparing original and processed images and labels\nLet's verify some before and after results","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport tifffile as tiff\nimport numpy as np\nfrom pathlib import Path\n\ndef show_comparison(file_ids, orig_img_dir, orig_lbl_dir, proc_img_dir, proc_lbl_dir,\n                    axis=\"z\", slice_indices=None, max_files=5):\n    \"\"\"\n    Compare original and processed images/labels side by side.\n    \"\"\"\n    file_ids = file_ids[:max_files]\n\n    for fid in file_ids:\n        # Load original\n        orig_img = tiff.imread(Path(orig_img_dir) / f\"{fid}.tif\")\n        orig_lbl = tiff.imread(Path(orig_lbl_dir) / f\"{fid}.tif\")\n\n        # Load processed\n        proc_img = np.load(Path(proc_img_dir) / f\"{fid}.npy\")\n        proc_lbl = load_rle(Path(proc_lbl_dir) / f\"{fid}.npz\")\n\n        for idx in range(1):\n            def get_slice(arr):\n                if axis==\"z\":\n                    return arr[arr.shape[0]//2, ...]\n                elif axis==\"y\":\n                    return arr[:, arr.shape[1]//2, :]\n                elif axis==\"x\":\n                    return arr[:, :, arr.shape[2]//2]\n                else:\n                    raise ValueError(\"axis must be 'x', 'y', or 'z'\")\n\n            slices = {\n                \"orig_img\": get_slice(orig_img),\n                \"proc_img\": get_slice(proc_img),\n                \"orig_lbl\": get_slice(orig_lbl),\n                \"proc_lbl\": get_slice(proc_lbl)\n            }\n\n            # Plot side by side\n            plt.figure(figsize=(12,6))\n            plt.subplot(1,2,1)\n            plt.title(f\"{fid} Original Image (axis={axis}, slice={idx})\")\n            plt.imshow(slices[\"orig_img\"], cmap=\"gray\")\n            plt.axis(\"off\")\n\n            plt.subplot(1,2,2)\n            plt.title(f\"{fid} Processed Image (axis={axis}, slice={idx})\")\n            plt.imshow(slices[\"proc_img\"], cmap=\"gray\")\n            plt.axis(\"off\")\n            plt.tight_layout()\n            plt.show()\n\n            # Labels comparison\n            plt.figure(figsize=(12,6))\n            plt.subplot(1,2,1)\n            plt.title(f\"{fid} Original Label (axis={axis}, slice={idx})\")\n            plt.imshow(slices[\"orig_lbl\"], cmap=\"gray\")\n            plt.axis(\"off\")\n\n            plt.subplot(1,2,2)\n            plt.title(f\"{fid} Processed Label (axis={axis}, slice={idx})\")\n            plt.imshow(slices[\"proc_lbl\"], cmap=\"gray\")\n            plt.axis(\"off\")\n            plt.tight_layout()\n            plt.show()\n\n\nfile_ids = [\"1029212680\", \"1006462223\", \"11630450\", \"1083486419\"]\nshow_comparison(file_ids,\n                orig_img_dir=IMG_DIR,\n                orig_lbl_dir=LBL_DIR,\n                proc_img_dir=NPY_IMG_DIR,\n                proc_lbl_dir=NPY_LBL_DIR,\n                axis=\"z\",\n                slice_indices=None,\n                max_files=5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T07:52:40.365222Z","iopub.execute_input":"2025-11-15T07:52:40.365775Z","iopub.status.idle":"2025-11-15T07:52:43.193000Z","shell.execute_reply.started":"2025-11-15T07:52:40.365725Z","shell.execute_reply":"2025-11-15T07:52:43.192000Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}