{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Vesuvis Data Preparation","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"# Installs","metadata":{}},{"cell_type":"code","source":"!pip install nb_black -q","metadata":{"execution":{"iopub.status.busy":"2023-04-28T08:13:39.191056Z","iopub.execute_input":"2023-04-28T08:13:39.192213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%load_ext lab_black","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"from pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport PIL.Image as Image\nfrom tqdm.auto import tqdm","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Paths & Settings","metadata":{}},{"cell_type":"code","source":"KAGGLE_DIR = Path(\"/\") / \"kaggle\"\n\nINPUT_DIR = KAGGLE_DIR / \"input\"\n\nCOMPETITION_DATA_DIR = INPUT_DIR / \"vesuvius-challenge-ink-detection\"\n\nDOWNSAMPLING = 1.0\nNUM_Z_SLICES = 4","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare DataFrame","metadata":{}},{"cell_type":"code","source":"def create_df_from_mask_paths(stage, downsampling):\n    mask_paths = sorted(COMPETITION_DATA_DIR.glob(f\"{stage}/*/mask.png\"))\n\n    df = pd.DataFrame({\"mask_png\": mask_paths})\n\n    df[\"mask_png\"] = df[\"mask_png\"].astype(str)\n\n    df[\"stage\"] = df[\"mask_png\"].str.split(\"/\").str[-3]\n    df[\"fragment_id\"] = df[\"mask_png\"].str.split(\"/\").str[-2]\n\n    df[\"mask_npy\"] = df[\"mask_png\"].str.replace(\n        stage, f\"{stage}_{downsampling}\", regex=False\n    )\n    df[\"mask_npy\"] = df[\"mask_npy\"].str.replace(\"input\", \"working\", regex=False)\n    df[\"mask_npy\"] = df[\"mask_npy\"].str.replace(\"png\", \"npy\", regex=False)\n\n    if stage == \"train\":\n        df[\"label_png\"] = df[\"mask_png\"].str.replace(\"mask\", \"inklabels\", regex=False)\n        df[\"label_npy\"] = df[\"mask_npy\"].str.replace(\"mask\", \"inklabels\", regex=False)\n\n    df[\"volumes_dir\"] = df[\"mask_png\"].str.replace(\n        \"mask.png\", \"surface_volume\", regex=False\n    )\n    df[\"volume_npy\"] = df[\"mask_npy\"].str.replace(\"mask\", \"volume\", regex=False)\n\n    return df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = create_df_from_mask_paths(\"train\", DOWNSAMPLING)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"label_npy\"].values[0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Convert Data to NumPy\n\n## Based on https://www.kaggle.com/code/jpposma/vesuvius-challenge-ink-detection-tutorial","metadata":{}},{"cell_type":"code","source":"def load_image(path):\n    return Image.open(path)\n\n\ndef resize_image(image, downsampling):\n    size = int(image.size[0] * downsampling), int(image.size[1] * downsampling)\n    return image.resize(size)\n\n\ndef load_and_resize_image(path, downsampling):\n    image = load_image(path)\n    return resize_image(image, downsampling)\n\n\ndef load_label_npy(path, downsampling):\n    label = load_and_resize_image(path, downsampling)\n    return np.array(label) > 0\n\n\ndef load_mask_npy(path, downsampling):\n    mask = load_and_resize_image(path, downsampling).convert(\"1\")\n    return np.array(mask)\n\n\ndef load_z_slice_npy(path, downsampling):\n    z_slice = load_and_resize_image(path, downsampling)\n    return np.array(z_slice, dtype=np.float32) / 65535.0\n\n\ndef load_volume_npy(volumes_dir, num_z_slices, downsampling):\n    mid = 65 // 2\n    start = mid - num_z_slices // 2\n    end = mid + num_z_slices // 2\n\n    z_slices_paths = sorted(Path(volumes_dir).glob(\"*.tif\"))[start:end]\n\n    batch_size = num_z_slices // 4\n    paths_batches = [\n        z_slices_paths[i : i + batch_size]\n        for i in range(0, len(z_slices_paths), batch_size)\n    ]\n\n    volumes = []\n    for paths_batch in tqdm(\n        paths_batches, leave=False, desc=\"Processing batches\", position=1\n    ):\n        z_slices = [\n            load_z_slice_npy(path, downsampling)\n            for path in tqdm(\n                paths_batch, leave=False, desc=\"Processing paths\", position=2\n            )\n        ]\n        volumes.append(np.stack(z_slices, axis=0))\n        del z_slices\n\n        # break\n\n    volume = np.concatenate(volumes, axis=0)\n\n    return volume","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_data_as_npy(df, train=True):\n    for row in tqdm(\n        df.itertuples(), total=len(df), desc=\"Processing fragments\", position=0\n    ):\n        mask_npy = load_mask_npy(row.mask_png, DOWNSAMPLING)\n        volume_npy = load_volume_npy(row.volumes_dir, NUM_Z_SLICES, DOWNSAMPLING)\n\n        Path(row.mask_npy).parent.mkdir(exist_ok=True, parents=True)\n        np.save(row.mask_npy, mask_npy)\n        np.save(row.volume_npy, volume_npy)\n\n        if train:\n            label_npy = load_label_npy(row.label_png, DOWNSAMPLING)\n            np.save(row.label_npy, label_npy)\n\n        tqdm.write(f\"Created {row.volume_npy} with shape {volume_npy.shape}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_data_as_npy(train_df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fix paths","metadata":{}},{"cell_type":"code","source":"train_df[\"label_npy\"] = train_df[\"label_npy\"].str.replace(\n    \"working\", \"input/vesuvis-data-preparation\", regex=False\n)\ntrain_df[\"mask_npy\"] = train_df[\"mask_npy\"].str.replace(\n    \"working\", \"input/vesuvis-data-preparation\", regex=False\n)\ntrain_df[\"volume_npy\"] = train_df[\"volume_npy\"].str.replace(\n    \"working\", \"input/vesuvis-data-preparation\", regex=False\n)\n\ntrain_df.to_csv(f\"data_{DOWNSAMPLING}.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}