{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Data Prep Notebook\n## This notebook helps create dataset of required resolution for HubMAP-Organ-segmentation competition","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\n\nfrom skimage import io\nfrom skimage.transform import resize","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-14T18:23:50.479789Z","iopub.execute_input":"2022-09-14T18:23:50.481028Z","iopub.status.idle":"2022-09-14T18:23:51.612290Z","shell.execute_reply.started":"2022-09-14T18:23:50.480828Z","shell.execute_reply":"2022-09-14T18:23:51.610761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df=pd.read_csv(\"/kaggle/input/hubmap-organ-segmentation/train.csv\")\ntrain_df[\"img_path\"] = train_df[\"id\"].apply(lambda x: f\"/kaggle/input/hubmap-organ-segmentation/train_images/{str(x)}.tiff\")","metadata":{"execution":{"iopub.status.busy":"2022-09-14T18:23:51.614298Z","iopub.execute_input":"2022-09-14T18:23:51.614777Z","iopub.status.idle":"2022-09-14T18:23:52.004467Z","shell.execute_reply.started":"2022-09-14T18:23:51.614737Z","shell.execute_reply":"2022-09-14T18:23:52.003063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf \nfrom tensorflow import keras\nimport tensorflow.keras.backend as K\nfrom tensorflow.keras.layers import Input, Conv2D, MaxPooling2D, Activation, ReLU\nfrom tensorflow.keras.layers import BatchNormalization, Conv2DTranspose, Concatenate\nfrom tensorflow.keras.models import Model, Sequential, load_model\nfrom tensorflow.keras.optimizers import Adam","metadata":{"execution":{"iopub.status.busy":"2022-09-14T18:23:52.225871Z","iopub.execute_input":"2022-09-14T18:23:52.226333Z","iopub.status.idle":"2022-09-14T18:24:00.327729Z","shell.execute_reply.started":"2022-09-14T18:23:52.226294Z","shell.execute_reply":"2022-09-14T18:24:00.325988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Config\n## Mention Resolution in PATCH_SIZE","metadata":{}},{"cell_type":"code","source":"PATCH_SIZE = 512\nos.mkdir(\"patches\")\nos.mkdir(\"patches/train\")\nos.mkdir(\"patches/masks\")","metadata":{"execution":{"iopub.status.busy":"2022-09-14T18:26:56.386963Z","iopub.execute_input":"2022-09-14T18:26:56.387567Z","iopub.status.idle":"2022-09-14T18:26:56.395019Z","shell.execute_reply.started":"2022-09-14T18:26:56.387504Z","shell.execute_reply":"2022-09-14T18:26:56.393952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_patches(x):\n    return tf.image.extract_patches(\n        x,\n        sizes = (1, PATCH_SIZE ,PATCH_SIZE, 1),\n        strides = (1, PATCH_SIZE, PATCH_SIZE, 1),\n        rates = (1, 1, 1, 1),\n        padding=\"SAME\"\n    )\n\ndef get_mask(rle_string, img_shape):\n    rle = [int(i) for i in rle_string.split(' ')]\n    pairs = list(zip(rle[0::2],rle[1::2]))\n\n    p_loc = []\n\n    for start, length in pairs:\n        for p_pos in range(start, start + length):\n            p_loc.append((p_pos % img_shape[1], p_pos // img_shape[0]))\n  \n    return helper(p_loc, img_shape)\n\ndef helper(mask, img_shape):\n  \n    canvas = np.zeros(img_shape).T\n    canvas[tuple(zip(*mask))] = 1.0\n\n      # This is the Equivalent for loop of the above command for better understanding.\n      # for pos in range(len(p_loc)):\n      #   canvas[pos[0], pos[1]] = 1\n\n    return canvas","metadata":{"execution":{"iopub.status.busy":"2022-09-14T18:26:57.890092Z","iopub.execute_input":"2022-09-14T18:26:57.891173Z","iopub.status.idle":"2022-09-14T18:26:57.904706Z","shell.execute_reply.started":"2022-09-14T18:26:57.891116Z","shell.execute_reply":"2022-09-14T18:26:57.903159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Remove the break statement before running the notebook","metadata":{}},{"cell_type":"code","source":"for i,j in tqdm(train_df.iterrows()):\n    image = io.imread(j[\"img_path\"])\n    patches = extract_patches(np.expand_dims(image,0)).numpy()\n    n = patches.shape[1]\n    patches = np.reshape(patches,(n**2,PATCH_SIZE,PATCH_SIZE,3))\n    mask = get_mask(j[\"rle\"],image.shape[:2])\n    mask_patches = extract_patches(np.expand_dims(mask,axis=(0,3))).numpy()\n    mask_patches = np.reshape(mask_patches,(n**2,PATCH_SIZE,PATCH_SIZE))\n    for i in range(n**2):\n        io.imsave(f\"patches/train/{j['id']}_{i}.png\",patches[i],check_contrast=False)\n        io.imsave(f\"patches/masks/{j['id']}_{i}.png\",mask_patches[i],check_contrast=False)\n    break","metadata":{"execution":{"iopub.status.busy":"2022-09-14T18:27:08.730126Z","iopub.execute_input":"2022-09-14T18:27:08.730596Z","iopub.status.idle":"2022-09-14T18:27:34.873087Z","shell.execute_reply.started":"2022-09-14T18:27:08.730560Z","shell.execute_reply":"2022-09-14T18:27:34.871762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar czf patches.tar.gz /kaggle/working/patches/","metadata":{"execution":{"iopub.status.busy":"2022-09-14T18:27:34.875713Z","iopub.execute_input":"2022-09-14T18:27:34.877095Z","iopub.status.idle":"2022-09-14T18:27:36.637352Z","shell.execute_reply.started":"2022-09-14T18:27:34.877034Z","shell.execute_reply":"2022-09-14T18:27:36.635675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}