{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<br>\n\n<br><center><img src=\"https://storage.googleapis.com/kaggle-competitions/kaggle/34547/logos/header.png?t=2022-02-15-22-37-27\" width=100%></center>\n\n<h2 style=\"text-align: center; font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: underline; text-transform: none; letter-spacing: 2px; color: #BF2C58; background-color: #ffffff;\">🫁🫀 Create Tiled Dataset – Hacking the Human Body 🫀🫁</h2>\n<h5 style=\"text-align: center; font-family: Verdana; font-size: 12px; font-style: normal; font-weight: bold; text-decoration: None; text-transform: none; letter-spacing: 1px; color: black; background-color: #ffffff;\">CREATED BY: DARIEN SCHETTLER</h5>\n\n<br>\n\n---\n\n<br>\n\n<center><div class=\"alert alert-block alert-danger\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 18px;\">🛑 &nbsp; WARNING:</b><br><br><b>THIS IS A WORK IN PROGRESS</b><br>\n</div></center>\n\n\n<center><div class=\"alert alert-block alert-warning\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 18px;\">👏 &nbsp; IF YOU FORK THIS OR FIND THIS HELPFUL &nbsp; 👏</b><br><br><b style=\"font-size: 22px; color: darkorange\">PLEASE UPVOTE!</b><br><br>This was a lot of work for me and while it may seem silly, it makes me feel appreciated when others like my work. 😅\n</div></center>\n\n\n","metadata":{"papermill":{"duration":0.023091,"end_time":"2022-04-27T19:57:38.884056","exception":false,"start_time":"2022-04-27T19:57:38.860965","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<p id=\"toc\"></p>\n\n<br><br>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #BF2C58; background-color: #ffffff;\">TABLE OF CONTENTS</h1>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#imports\">0&nbsp;&nbsp;&nbsp;&nbsp;IMPORTS</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#setup\">1&nbsp;&nbsp;&nbsp;&nbsp;SETUP</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#helper_functions\">2&nbsp;&nbsp;&nbsp;&nbsp;HELPER FUNCTIONS</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#create_dataset\">3&nbsp;&nbsp;&nbsp;&nbsp;DATASET CREATION</a></h3>\n","metadata":{"papermill":{"duration":0.021671,"end_time":"2022-04-27T19:57:38.928105","exception":false,"start_time":"2022-04-27T19:57:38.906434","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"imports\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #BF2C58;\" id=\"imports\">0&nbsp;&nbsp;IMPORTS&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a></h1>","metadata":{"papermill":{"duration":0.021385,"end_time":"2022-04-27T19:57:38.971091","exception":false,"start_time":"2022-04-27T19:57:38.949706","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(\"\\n... IMPORTS STARTING ...\\n\")\n\nprint(\"\\n\\tVERSION INFORMATION\")\n# Machine Learning and Data Science Imports\nimport tensorflow as tf; print(f\"\\t\\t– TENSORFLOW VERSION: {tf.__version__}\");\nimport tensorflow_hub as tfhub; print(f\"\\t\\t– TENSORFLOW HUB VERSION: {tfhub.__version__}\");\nimport tensorflow_addons as tfa; print(f\"\\t\\t– TENSORFLOW ADDONS VERSION: {tfa.__version__}\");\nimport tensorflow_io as tfio; print(f\"\\t\\t– TENSORFLOW I/O VERSION: {tfio.__version__}\");\nimport pandas as pd; pd.options.mode.chained_assignment = None;\nimport numpy as np; print(f\"\\t\\t– NUMPY VERSION: {np.__version__}\");\nimport sklearn; print(f\"\\t\\t– SKLEARN VERSION: {sklearn.__version__}\");\nfrom sklearn.preprocessing import RobustScaler, PolynomialFeatures\nfrom pandarallel import pandarallel; pandarallel.initialize();\nfrom sklearn.model_selection import GroupKFold, StratifiedKFold\nfrom scipy.spatial import cKDTree\nimport tifffile as tiff\n\n# # RAPIDS\n# import cudf, cupy, cuml\n# from cuml.neighbors import NearestNeighbors\n# from cuml.manifold import TSNE, UMAP\n\n# Built In Imports\nfrom kaggle_datasets import KaggleDatasets\nfrom collections import Counter\nfrom datetime import datetime\nfrom glob import glob\nimport warnings\nimport requests\nimport hashlib\nimport imageio\nimport IPython\nimport sklearn\nimport urllib\nimport zipfile\nimport pickle\nimport random\nimport shutil\nimport string\nimport json\nimport math\nimport time\nimport gzip\nimport ast\nimport sys\nimport io\nimport os\nimport gc\nimport re\n\n# Visualization Imports\nfrom matplotlib.colors import ListedColormap\nfrom matplotlib.patches import Rectangle\nimport matplotlib.patches as patches\nimport plotly.graph_objects as go\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm; tqdm.pandas();\nimport plotly.express as px\nimport seaborn as sns\nfrom PIL import Image, ImageEnhance\nimport matplotlib; print(f\"\\t\\t– MATPLOTLIB VERSION: {matplotlib.__version__}\");\nfrom matplotlib import animation, rc; rc('animation', html='jshtml')\nimport plotly\nimport PIL\nimport cv2\n\nimport plotly.io as pio\nprint(pio.renderers)\n\ndef seed_it_all(seed=7):\n    \"\"\" Attempt to be Reproducible \"\"\"\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n\n    \nprint(\"\\n\\n... IMPORTS COMPLETE ...\\n\")","metadata":{"papermill":{"duration":9.381945,"end_time":"2022-04-27T19:57:48.376023","exception":false,"start_time":"2022-04-27T19:57:38.994078","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-01T19:50:27.483127Z","iopub.execute_input":"2022-07-01T19:50:27.483593Z","iopub.status.idle":"2022-07-01T19:50:41.919287Z","shell.execute_reply.started":"2022-07-01T19:50:27.483547Z","shell.execute_reply":"2022-07-01T19:50:41.917832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<a id=\"background_information\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #BF2C58; background-color: #ffffff;\" id=\"setup\">1&nbsp;&nbsp;SETUP&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a></h1>\n\n---\n","metadata":{"papermill":{"duration":0.023943,"end_time":"2022-04-27T19:57:48.424163","exception":false,"start_time":"2022-04-27T19:57:48.40022","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(\"\\n... DATA ACCESS SETUP STARTED ...\\n\")\n\n# Local path to training and validation images\nDATA_DIR = \"/kaggle/input/hubmap-organ-segmentation\"\nprint(f\"\\n... DATA DIRECTORY PATH IS:\\n\\t--> {DATA_DIR}\")\n\nprint(f\"\\n... IMMEDIATE CONTENTS OF DATA DIRECTORY IS:\")\nfor file in tf.io.gfile.glob(os.path.join(DATA_DIR, \"*\")): print(f\"\\t--> {file}\")\n\nprint(\"\\n\\n... DATA ACCESS SETUP COMPLETED ...\\n\")","metadata":{"papermill":{"duration":0.04107,"end_time":"2022-04-27T19:57:48.661145","exception":false,"start_time":"2022-04-27T19:57:48.620075","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-01T19:50:41.921522Z","iopub.execute_input":"2022-07-01T19:50:41.922145Z","iopub.status.idle":"2022-07-01T19:50:41.935487Z","shell.execute_reply.started":"2022-07-01T19:50:41.922094Z","shell.execute_reply":"2022-07-01T19:50:41.934158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\n... BASIC DATA SETUP STARTING ...\\n\\n\")\n\n# Open the training dataframe and display the initial dataframe\nTRAIN_IMAGES_DIR = os.path.join(DATA_DIR, \"train_images\")\nTRAIN_LABELS_DIR = os.path.join(DATA_DIR, \"train_annotations\")\nTRAIN_CSV = os.path.join(DATA_DIR, \"train.csv\")\ntrain_df = pd.read_csv(TRAIN_CSV)\n\n# Get all training images\nall_train_images = glob(os.path.join(TRAIN_IMAGES_DIR, \"*.tiff\"), recursive=True)\nall_train_labels = glob(os.path.join(TRAIN_LABELS_DIR, \"*.json\"), recursive=True)\n\nTEST_IMAGES_DIR = os.path.join(DATA_DIR, \"test_images\")\nTEST_CSV = os.path.join(DATA_DIR, \"test.csv\")\ntest_df = pd.read_csv(TEST_CSV)\n\nSS_CSV   = os.path.join(DATA_DIR, \"sample_submission.csv\")\nss_df = pd.read_csv(SS_CSV)\n\n# Get all testing images if there are any\nall_test_images = glob(os.path.join(TEST_IMAGES_DIR, \"*.tiff\"), recursive=True)\n\ndef rgb2hex(rgb_tuple):\n    r,g,b = rgb_tuple\n    def clamp(x): \n        return max(0, min(x, 255))\n    return \"#{0:02x}{1:02x}{2:02x}\".format(clamp(r), clamp(g), clamp(b))\nORGANS = ['kidney', 'largeintestine', 'lung', 'prostate', 'spleen']\n_COLOURS = [(230, 0, 73), (11, 180, 255), (80, 233, 145), (230, 216, 0), (155, 25, 245)]\nO2C_MAP = {_o:_c for _o,_c in zip(ORGANS, _COLOURS)}\nO2C_HEX_MAP = {_o:rgb2hex(_c) for _o,_c in O2C_MAP.items()}\n\nsex_2_int = {\"Male\":0, \"Female\":1}\nint_2_sex = {v:k for k,v in sex_2_int.items()}\nage_divisor = 100.0\n\ntrain_df[\"sex\"] = train_df[\"sex\"].map(sex_2_int)\ntrain_df[\"age\"] = train_df[\"age\"]/age_divisor\n\ntrain_img_map = {int(x[:-5].rsplit(\"/\", 1)[-1]):x for x in all_train_images}\ntrain_lbl_map = {int(x[:-5].rsplit(\"/\", 1)[-1]):x for x in all_train_labels}\ntest_img_map = {int(x[:-5].rsplit(\"/\", 1)[-1]):x for x in all_test_images}\n\ntrain_df.insert(3, \"img_path\", train_df[\"id\"].map(train_img_map))\ntrain_df.insert(4, \"lbl_path\", train_df[\"id\"].map(train_lbl_map))\ntest_df.insert(3, \"img_path\", test_df[\"id\"].map(test_img_map))\n\nprint(\"\\n... TRAINING DATAFRAME... \\n\")\ndisplay(train_df)\n\nprint(\"\\n\\n\\n... TEST DATAFRAME... \\n\")\ndisplay(test_df)\n\nprint(\"\\n\\n\\n... SUBMISSION DATAFRAME... \\n\")\ndisplay(ss_df)\n\nprint(\"\\n... BASIC DATA SETUP FINISHED ...\\n\\n\")","metadata":{"papermill":{"duration":4.583724,"end_time":"2022-04-27T19:57:53.431215","exception":false,"start_time":"2022-04-27T19:57:48.847491","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-01T19:50:41.937416Z","iopub.execute_input":"2022-07-01T19:50:41.938092Z","iopub.status.idle":"2022-07-01T19:50:42.507499Z","shell.execute_reply.started":"2022-07-01T19:50:41.938044Z","shell.execute_reply":"2022-07-01T19:50:42.506167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n\n<a id=\"helper_functions\"></a>\n\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #BF2C58; background-color: #ffffff;\" id=\"helper_functions\">\n    2&nbsp;&nbsp;HELPER FUNCTION & CLASSES&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a>\n</h1>\n\n---","metadata":{"papermill":{"duration":0.028617,"end_time":"2022-04-27T19:57:55.756004","exception":false,"start_time":"2022-04-27T19:57:55.727387","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# ref: https://www.kaggle.com/paulorzp/run-length-encode-and-decode\n# modified from: https://www.kaggle.com/inversion/run-length-decoding-quick-start\ndef rle_decode(mask_rle, shape, color=1):\n    \"\"\" TBD\n    \n    Args:\n        mask_rle (str): run-length as string formated (start length)\n        shape (tuple of ints): (height,width) of array to return \n    \n    Returns: \n        Mask (np.array)\n            - 1 indicating mask\n            - 0 indicating background\n\n    \"\"\"\n    # Split the string by space, then convert it into a integer array\n    s = np.array(mask_rle.split(), dtype=int)\n\n    # Every even value is the start, every odd value is the \"run\" length\n    starts = s[0::2] - 1\n    lengths = s[1::2]\n    ends = starts + lengths\n\n    # The image image is actually flattened since RLE is a 1D \"run\"\n    if len(shape)==3:\n        h, w, d = shape\n        img = np.zeros((h * w, d), dtype=np.float32)\n    else:\n        h, w = shape\n        img = np.zeros((h * w,), dtype=np.float32)\n\n    # The color here is actually just any integer you want!\n    for lo, hi in zip(starts, ends):\n        img[lo : hi] = color\n        \n    # Don't forget to change the image back to the original shape\n    return img.reshape(shape).T\n\n# https://www.kaggle.com/namgalielei/which-reshape-is-used-in-rle\ndef rle_decode_top_to_bot_first(mask_rle, shape):\n    \"\"\" TBD\n    \n    Args:\n        mask_rle (str): run-length as string formated (start length)\n        shape (tuple of ints): (height,width) of array to return \n    \n    Returns:\n        Mask (np.array)\n            - 1 indicating mask\n            - 0 indicating background\n\n    \"\"\"\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape((shape[1], shape[0]), order='F').T  # Reshape from top -> bottom first\n\n# ref.: https://www.kaggle.com/stainsby/fast-tested-rle\ndef rle_encode(img):\n    \"\"\" TBD\n    \n    Args:\n        img (np.array): \n            - 1 indicating mask\n            - 0 indicating background\n    \n    Returns: \n        run length as string formated\n    \"\"\"\n    pixels = img.T.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n\ndef flatten_l_o_l(nested_list):\n    \"\"\" Flatten a list of lists \"\"\"\n    return [item for sublist in nested_list for item in sublist]\n\ndef load_json_to_dict(json_path):\n    \"\"\" tbd \"\"\"\n    with open(json_path) as json_file:\n        data = json.load(json_file)\n    return data\n\ndef tf_decode_tiff(img_path, to_numpy=False, to_rgb=False):\n    img = tf.io.read_file(img_path)\n    img = tfio.experimental.image.decode_tiff(img)\n    \n    # Optionals\n    if to_rgb: img = tfio.experimental.color.rgba_to_rgb(img)\n    if to_numpy: img = img.numpy()\n        \n    return img\n        \ndef rle_decode_tf(mask_rle, shape):\n    \"\"\" TBD \"\"\"\n    \n    shape = tf.convert_to_tensor(shape, tf.int64)\n    size = tf.math.reduce_prod(shape)\n    \n    # Split string\n    s = tf.strings.split(mask_rle)\n    s = tf.strings.to_number(s, tf.int64)\n    \n    # Get starts and lengths\n    starts = s[::2] - 1\n    lens = s[1::2]\n    \n    # Make ones to be scattered\n    total_ones = tf.reduce_sum(lens)\n    ones = tf.ones([total_ones], tf.uint8)\n    \n    # Make scattering indices\n    r = tf.range(total_ones)\n    lens_cum = tf.math.cumsum(lens)\n    s = tf.searchsorted(lens_cum, r, 'right')\n    idx = r + tf.gather(starts - tf.pad(lens_cum[:-1], [(1, 0)]), s)\n    \n    # Scatter ones into flattened mask\n    mask_flat = tf.scatter_nd(tf.expand_dims(idx, 1), ones, [size])\n    \n    # Reshape into mask\n    return tf.transpose(tf.reshape(mask_flat, shape))\n\n\ndef _bytes_feature(value, is_list=False):\n    \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n    if isinstance(value, type(tf.constant(0))):\n        value = value.numpy() # BytesList won't unpack a string from an EagerTensor.\n    \n    if not is_list:\n        value = [value]\n    \n    return tf.train.Feature(bytes_list=tf.train.BytesList(value=value))\n\ndef _float_feature(value, is_list=False):\n    \"\"\"Returns a float_list from a float / double.\"\"\"\n        \n    if not is_list:\n        value = [value]\n        \n    return tf.train.Feature(float_list=tf.train.FloatList(value=value))\n\ndef _int64_feature(value, is_list=False):\n    \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n        \n    if not is_list:\n        value = [value]\n        \n    return tf.train.Feature(int64_list=tf.train.Int64List(value=value))\n\n    \ndef serialize_raw(image, mask, id_str):\n    \"\"\"\n    Creates a tf.Example message ready to be written to a file from 4 features.\n\n    Args:\n        image (TBD): TBD\n        mask (TBD): TBD\n        id_str (str): TBD\n    \n    Returns:\n        A tf.Example Message ready to be written to file\n    \"\"\"\n    \n    # Create a dictionary mapping the feature name to the \n    # tf.Example-compatible data type.\n    feature = {\n        \"image\"    : _bytes_feature(tf.io.encode_png(image), is_list=False),\n        \"mask\"     : _bytes_feature(tf.io.encode_png(mask),  is_list=False),\n        \"id_str\"   : _bytes_feature(id_str, is_list=False)\n    }\n\n    # Create a Features message using tf.train.Example.\n    example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n    return example_proto.SerializeToString()","metadata":{"papermill":{"duration":0.064826,"end_time":"2022-04-27T19:57:55.849577","exception":false,"start_time":"2022-04-27T19:57:55.784751","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-01T20:24:30.652596Z","iopub.execute_input":"2022-07-01T20:24:30.653077Z","iopub.status.idle":"2022-07-01T20:24:30.687272Z","shell.execute_reply.started":"2022-07-01T20:24:30.653042Z","shell.execute_reply":"2022-07-01T20:24:30.685488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n\n<a id=\"create_dataset\"></a>\n\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #BF2C58; background-color: #ffffff;\" id=\"create_dataset\">\n    3&nbsp;&nbsp;DATASET CREATION&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a>\n</h1>\n\n---\n\nThis notebook will allow you to create 3 different types of datasets. The 3 different types of datasets are:\n\n1. Tiled PNG Dataset [train|val]\n2. Tiled TFRecord Dataset [train|val]\n3. Stratified Splits For Tiled TFRecord Dataset [train|val]\n3. Stratified Splits For Tiled PNG Dataset [train|val]\n\n<br>\n\n**NOTE:**\n* I do not include stratified splits for the PNG dataset option because this is easy enough to create on your own. For TFRecords, the splitting has to be done PRIOR to the dataset creation and as such has to be handled here.","metadata":{"papermill":{"duration":0.029564,"end_time":"2022-04-27T19:57:55.908689","exception":false,"start_time":"2022-04-27T19:57:55.879125","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #BF2C58; background-color: #ffffff;\">3.1 DEFINE OPTIONS</h3>\n\n---\n","metadata":{"papermill":{"duration":0.029458,"end_time":"2022-04-27T19:57:55.967099","exception":false,"start_time":"2022-04-27T19:57:55.937641","status":"completed"},"tags":[]}},{"cell_type":"code","source":"STYLE = \"png\" # one of [\"png\"|\"tfrec\"]\nTILED_OUTPUT_DIR = \"/kaggle/working/tiled_dataset\"\nTILE_SIZE = 512\nSTRIDE = 384 # if STRIDE==TILE_SIZE there will be no overlap between tiles\nFULL_SIZE_RES = 0.4 # if 0.4 then no rescale on train data\nDROP_EMPTY_TILES = True\nUPSAMPLE=False # Whether or not to upsample under-represented classes\nMIX_ORGANS = True # If True, create a single dataset with all organs shuffled, otherwise, create indiv. organ datasets\n\norgan_dfs = {_organ:train_df[train_df.organ==_organ].reset_index(drop=True) for _organ in ORGANS}\nprint(\"ORGAN DISTRIBUTION\")\nfor _organ, _df in organ_dfs.items():\n    print(f\"{_organ:<15}: {len(_df)}\")","metadata":{"papermill":{"duration":0.397082,"end_time":"2022-04-27T19:57:56.393092","exception":false,"start_time":"2022-04-27T19:57:55.99601","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-01T19:50:42.557631Z","iopub.execute_input":"2022-07-01T19:50:42.558199Z","iopub.status.idle":"2022-07-01T19:50:42.618759Z","shell.execute_reply.started":"2022-07-01T19:50:42.558151Z","shell.execute_reply":"2022-07-01T19:50:42.617245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #BF2C58; background-color: #ffffff;\">3.2 CREATE TILED PNG DATASET</h3>\n\n---\n\nThis will either be the final version or an intermediate one if we are planning to leverage tfrecords.","metadata":{}},{"cell_type":"code","source":"def tile_example(img_path, mask_rle, ex_id, organ, \n                 do_save=False, save_dir=\"/kaggle/working/tiled_dataset\", \n                 tile_size=512, stride=None, discard_edge=True, edge_pad=(245, 245, 244),\n                 full_img_size=3000, new_size_res=0.4, original_size_res=0.4):\n    \"\"\"\n    \n    Args:\n        img_path (str):\n            A string pointing to the .tiff file to be tiled\n        mask_rle (str):\n            This is a run-length encoding specifying the location of the respective organ FTUs. \n            RLEs are generally used to encode the location of foreground objects in segmentation. \n                - Instead of outputting a mask image, you give a list of start pixels and how many pixels \n                  after each of those starts is included in the mask. \n                - The encoding rule is pretty simple: \n                    -- Where the mask IS --> 1\n                    -- Where the mask IS NOT --> 0\n        ex_id (int):\n            A unique identifier for a particular example\n        organ (str):\n            The respective organ for a given example indicating the FTU that will be masked by the RLE string.\n                - FTUs in the organs presented in this competition are:\n                    1. glomeruli in the kidney\n                    2. crypt in the large intestine\n                    3. alveolus in the lung\n                    4. glandular acinus in the prostate\n                    5. white pulp in the spleen\n        do_save (bool, optional):\n            Whether or not we want to save the tiles to disk or return them as a list of arrays\n        save_dir (str, optional):\n            A string pointing to the root directory to save the respective files IF the `do_save` argument is `True` \n        tile_size (int, optional):\n            An integer representing the square image size that the image will be tiled with\n                i.e. 3000x3000 image with 300x300 tiles (assume `stride=300`) will yield 100 tiles.\n        stride (int, optional):\n            An integer representing the stride between adjoining tiles. \n            When stride=tile_size there will be no overlap between adjoining tiles.\n                i.e. 3000x3000 image with 300x300 tiles (assume `stride=300`) will yield 100 tiles.\n            When stride<tile_size there will be some overlap between adjoining tiles\n                i.e. 3000x3000 image with 300x300 tiles (assume `stride=150`) will yield 324 tiles.\n            When stride>tile_size there will be some parts of the image missing between adjoining tiles\n                i.e. 3000x3000 image with 300x300 tiles (assume `stride=450`) will yield 36 tiles.\n        discard_edge (bool, optional):\n            A bool indicating whether or not to discard or pad edges that do not \"fill up\" a whole tile\n                - If `discard_edge` is `True` we simply discard incomplete tiles at the edges\n                - If `discard_edge` is `False` we pad incomplete tiles at the edges to make them complete\n        edge_pad (tuple of ints, optional):\n            The RGB value to use for padding the image (0 will be used for the mask)\n        full_img_size (int, optional):\n            An integer indicating the original full-scale pixel size of the image to be tiled\n        new_size_res (int, optional):\n            The resolution in um per pixel for the desired tiles (probably 0.4 to match HPA data)\n        original_size_res (int, optional):\n            The resolution in um per pixel for the original image (0.4 for HPA, other organ dependant res for HuBMAP)\n        \n    Returns:\n           ––– do_save=True –––\n        1. Returns a partial path that specifies the root str to access\n           the saved tile images and masks (saved as png) and saves all\n           of the tiled images to disk in the specified folder.\n           \n           ––– do_save=False –––\n        2. Returns the tiled image and mask as per the specified arguments\n    \"\"\"\n    # Create directory \n    if do_save and not os.path.isdir(save_dir): os.makedirs(save_dir, exist_ok=True)\n    \n    # Set stride to tile size if not provided\n    if stride is None: stride=tile_size\n    \n    # Open image and mask from file\n    img = tiff.imread(img_path)\n    mask = np.array(rle_decode_tf(mask_rle, (full_img_size,full_img_size)), dtype=np.uint8)\n    \n    # Update resolution and image size accordingly\n    if new_size_res!=original_size_res:\n        resize_to = full_img_size = full_img_size*(original_size_res/new_size_res)\n        img = cv2.resize(img, (resize_to, resize_to))\n        mask = cv2.resize(mask, (resize_to, resize_to), interpolation=cv2.INTER_NEAREST)\n    \n    # Loop over image\n    img_crops, mask_crops = [], []\n    for x in range(tile_size, full_img_size+stride, stride):\n        for y in range(tile_size, full_img_size+stride, stride):\n            x1, y1 = x-tile_size, y-tile_size\n            x2, y2 = min(x, full_img_size), min(y, full_img_size)\n            _img_crop = img[x1:x2, y1:y2]\n            _mask_crop = mask[x1:x2, y1:y2]\n            \n            if discard_edge and _img_crop.shape!=(tile_size, tile_size, 3): continue\n            \n            if do_save:\n                cv2.imwrite(os.path.join(save_dir, f\"image_{ex_id}_{organ}__s_{STRIDE}__t_{TILE_SIZE}__{x1}_{y1}.png\"), _img_crop[..., ::-1])\n                cv2.imwrite(os.path.join(save_dir, f\"mask_{ex_id}_{organ}__s_{STRIDE}__t_{TILE_SIZE}__{x1}_{y1}.png\"), _mask_crop)\n            else:\n                img_crops.append(_img_crop)\n                mask_crops.append(_mask_crop)\n    if not do_save:\n        return img_crops, mask_crops  \n    else:\n        return os.path.join(save_dir, f\"image_{ex_id}_{organ}__s_{STRIDE}__t_{TILE_SIZE}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-01T19:52:10.545212Z","iopub.execute_input":"2022-07-01T19:52:10.545655Z","iopub.status.idle":"2022-07-01T19:52:10.56621Z","shell.execute_reply.started":"2022-07-01T19:52:10.545618Z","shell.execute_reply":"2022-07-01T19:52:10.564957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**PLOT A DEMO TILING SO WE KNOW WHAT WE ARE GETTING**","metadata":{}},{"cell_type":"code","source":"DEMO_IDX = 10\nimg_crops, mask_crops = tile_example(train_df.iloc[DEMO_IDX].img_path, \n                                     train_df.iloc[DEMO_IDX].rle, \n                                     train_df.iloc[DEMO_IDX].id, \n                                     train_df.iloc[DEMO_IDX].organ,\n                                     tile_size=TILE_SIZE, stride=STRIDE)\n\nplt.figure(figsize=(20,20))\nplt.imshow(cv2.imread(train_df.iloc[DEMO_IDX].img_path)[..., ::-1])\nplt.axis(False)\nplt.show()\n\nplt.figure(figsize=(20,20))\nplt.imshow(rle_decode_tf(train_df.iloc[DEMO_IDX].rle, (train_df.iloc[DEMO_IDX].img_width,train_df.iloc[DEMO_IDX].img_height)))\nplt.axis(False)\nplt.show()\n\nplt.figure(figsize=(20,20))\nfor i, img_crop in tqdm(enumerate(img_crops)):\n    plt.subplot(int(len(img_crops)**0.5), int(len(img_crops)**0.5),i+1)\n    plt.imshow(img_crop)\n    plt.axis(False)\nplt.tight_layout()\nplt.show()\n\nplt.figure(figsize=(20,20))\nfor i, mask_crop in tqdm(enumerate(mask_crops)):\n    plt.subplot(int(len(mask_crops)**0.5), int(len(mask_crops)**0.5),i+1)\n    plt.imshow(mask_crop)\n    plt.axis(False)\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-01T19:52:16.047849Z","iopub.execute_input":"2022-07-01T19:52:16.048819Z","iopub.status.idle":"2022-07-01T19:52:28.575741Z","shell.execute_reply.started":"2022-07-01T19:52:16.048766Z","shell.execute_reply":"2022-07-01T19:52:28.574852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pandas_tile_fn(row):\n    \"\"\" \n    Create wrapper function to allow us to return root path and leverage\n    `.progress_apply|.parallel_apply` if we so wish\n    \n    This will take 5-30 minutes depending on:\n        - Settings related to tile-size and stride\n        - Computer hardware\n        - Whether you leverage multiprocessing\n    \"\"\"\n    return tile_example(row.img_path, row.rle, row.id, row.organ, tile_size=TILE_SIZE, stride=STRIDE, full_img_size=row.img_width, do_save=True, save_dir=TILED_OUTPUT_DIR)\n\ntrain_df[\"tile_image_root_path\"] = train_df.progress_apply(pandas_tile_fn, axis=1)\ntrain_df[\"tile_mask_root_path\"] = train_df[\"tile_image_root_path\"].str.replace(\"/image_\", \"/mask_\")","metadata":{"execution":{"iopub.status.busy":"2022-07-01T19:52:52.292271Z","iopub.execute_input":"2022-07-01T19:52:52.292683Z","iopub.status.idle":"2022-07-01T20:00:16.807722Z","shell.execute_reply.started":"2022-07-01T19:52:52.292647Z","shell.execute_reply":"2022-07-01T20:00:16.805732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_sorted_tiles(root_path):\n    return sorted(glob(root_path+\"*.png\"), key=lambda x: [int(_x) for _x in x[:-4].rsplit(\"_\", 2)[1:]])\ndemo_img_tiles_after_save  = get_sorted_tiles(train_df.iloc[DEMO_IDX].tile_image_root_path)\ndemo_mask_tiles_after_save = get_sorted_tiles(train_df.iloc[DEMO_IDX].tile_mask_root_path)\n\nplt.figure(figsize=(20,20))\nfor i, img_crop in tqdm(enumerate(demo_img_tiles_after_save)):\n    plt.subplot(int(len(demo_img_tiles_after_save)**0.5), int(len(demo_img_tiles_after_save)**0.5),i+1)\n    plt.imshow(cv2.imread(img_crop)[..., ::-1])\n    plt.axis(False)\nplt.tight_layout()\nplt.show()\n\nplt.figure(figsize=(20,20))\nfor i, mask_crop in tqdm(enumerate(demo_mask_tiles_after_save)):\n    plt.subplot(int(len(demo_mask_tiles_after_save)**0.5), int(len(demo_mask_tiles_after_save)**0.5),i+1)\n    plt.imshow(cv2.imread(mask_crop).astype(np.float32))\n    plt.axis(False)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-01T20:00:16.811588Z","iopub.execute_input":"2022-07-01T20:00:16.812095Z","iopub.status.idle":"2022-07-01T20:00:28.5409Z","shell.execute_reply.started":"2022-07-01T20:00:16.81204Z","shell.execute_reply":"2022-07-01T20:00:28.539681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #BF2C58; background-color: #ffffff;\">3.3 CREATE TILE DATAFRAME</h3>\n\n---\n\n","metadata":{}},{"cell_type":"code","source":"def sort_all_tiles(file_list):\n    \"\"\" sorts by id-->tile_h-->tile_w \"\"\"\n    return sorted(file_list, key=lambda x: [int(x.rsplit(\"_\", 11)[1]),]+[int(_x) for _x in x[:-4].rsplit(\"_\", 2)[1:]])\n\ndef get_meta(row):\n    \"\"\" Get information from file path and mask area \"\"\"\n    split_meta = row.img_path[:-4].rsplit(\"_\")[2:]\n    row[\"id\"] = int(split_meta[0])\n    row[\"organ\"] = split_meta[1]\n    row[\"stride\"] = int(split_meta[4])\n    row[\"tile_size\"] = int(split_meta[7])\n    row[\"tl_x\"] = int(split_meta[10])\n    row[\"tl_y\"] = int(split_meta[9])\n    row[\"mask_area\"] = int(cv2.imread(row.mask_path)[..., 0].sum())\n    return row\n\ntile_train_df = pd.DataFrame({\n    \"img_path\":sort_all_tiles(glob(os.path.join(TILED_OUTPUT_DIR, \"image*\"))),\n    \"mask_path\":sort_all_tiles(glob(os.path.join(TILED_OUTPUT_DIR, \"mask*\")))\n})\ntile_train_df = tile_train_df.progress_apply(get_meta, axis=1)\nif DROP_EMPTY_TILES:  tile_train_df=tile_train_df[tile_train_df.mask_area>0].reset_index(drop=True)\ndisplay(tile_train_df)\n\nfig = px.histogram(tile_train_df[tile_train_df.mask_area>0], \"mask_area\", color_discrete_map=O2C_HEX_MAP, color=\"organ\", title=\"<b>Mask Area</b>\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-01T20:14:17.445036Z","iopub.execute_input":"2022-07-01T20:14:17.445451Z","iopub.status.idle":"2022-07-01T20:14:17.598588Z","shell.execute_reply.started":"2022-07-01T20:14:17.445417Z","shell.execute_reply":"2022-07-01T20:14:17.597533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #BF2C58; background-color: #ffffff;\">3.4 ADD K-FOLD INFORMATION INTO DATAFRAME</h3>\n\n---\n","metadata":{}},{"cell_type":"code","source":"K_FOLDS = 8\ntile_train_df[\"fold\"] = 0\n\ngkf = GroupKFold(n_splits=K_FOLDS)\n\nfor i, (train_idxs, val_idxs) in enumerate(gkf.split(X=tile_train_df[\"img_path\"], \n                                            y=tile_train_df[\"organ\"], \n                                            groups=tile_train_df[\"id\"])):\n    print(f\"\\nFOLD #{i+1}\\n\\t\" \\\n          f\"--> {len(train_idxs)} TRAIN EXAMPLES\\n\\t\" \\\n          f\"--> {len(val_idxs)} VALIDATION EXAMPLES\\n\")\n    tile_train_df.loc[val_idxs, \"fold\"] = i+1\n    \ntile_train_df = tile_train_df.sample(len(tile_train_df)).reset_index(drop=True)\ntile_train_df.to_csv(\"./tile_train.csv\", index=False)\n\nprint(tile_train_df[\"fold\"].value_counts())\ntile_train_df","metadata":{"execution":{"iopub.status.busy":"2022-07-01T20:15:01.93275Z","iopub.execute_input":"2022-07-01T20:15:01.933164Z","iopub.status.idle":"2022-07-01T20:15:01.975197Z","shell.execute_reply.started":"2022-07-01T20:15:01.933129Z","shell.execute_reply":"2022-07-01T20:15:01.973975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <font color=\"red\">**WIP**</font> **...THE REST IS COMING**... <font color=\"red\">**WIP**</font>","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #BF2C58; background-color: #ffffff;\">3.5 TFRECORDS</h3>\n\n---\n","metadata":{}},{"cell_type":"code","source":"def write_tfrecords(ds, n_ex, n_ex_per_rec=240, serialize_fn=serialize_raw, out_dir=\"/kaggle/working/tfrecord_dataset\", ds_type=\"train\", fold=None):\n    \"\"\"  \"\"\"\n    n_recs = int(np.ceil(n_ex/n_ex_per_rec))\n    \n    # Make dataset iterable\n    ds = iter(ds)\n        \n    out_dir = os.path.join(out_dir, ds_type)\n    # Create folder\n    if not os.path.isdir(out_dir):\n        os.makedirs(out_dir, exist_ok=True)\n        \n    # Create tfrecords\n    for i in tqdm(range(n_recs), total=n_recs):\n        print(f\"\\n... Writing {ds_type.title()} TFRecord {i+1} of {n_recs} For Fold #{fold}...\\n\")\n        if fold is not None:\n            tfrec_path = os.path.join(out_dir, f\"{ds_type}_fold_{fold}__{(i+1):02}_{n_recs:02}.tfrec\")\n        else:\n            tfrec_path = os.path.join(out_dir, f\"{ds_type}__{(i+1):02}_{n_recs:02}.tfrec\")\n        \n        # This makes the tfrecord\n        with tf.io.TFRecordWriter(tfrec_path) as writer:\n            for ex in tqdm(range(n_ex_per_rec), total=n_ex_per_rec):\n                try:\n                    example = serialize_fn(*next(ds))\n                    writer.write(example)\n                except:\n                    break\n                    \ndef tf_load_img(path, channels=3):\n    \"\"\" Load an image with the correct shape using only TF\n    \n    Args:\n        path (tf.string): Path to the image to be loaded\n    \n    Returns:\n        3 channel tf.Constant image ready for training/inference\n    \n    \"\"\"\n    img_bytes = tf.io.read_file(path)\n    return tf.reshape(tf.image.decode_png(img_bytes, channels=channels, dtype=tf.uint8), (TILE_SIZE, TILE_SIZE, channels))\n\ndef create_ds(train_df, val_df):\n    \"\"\" TBD \"\"\"\n    train_ds = tf.data.Dataset.from_tensor_slices((train_df.img_path, train_df.mask_path))\n    train_ds = train_ds.map(lambda img_path,msk_path: (tf_load_img(img_path), tf_load_img(msk_path, channels=1), img_path), num_parallel_calls=tf.data.AUTOTUNE)\n    val_ds = tf.data.Dataset.from_tensor_slices((val_df.img_path, val_df.mask_path))\n    val_ds = val_ds.map(lambda img_path,msk_path: (tf_load_img(img_path), tf_load_img(msk_path, channels=1), img_path), num_parallel_calls=tf.data.AUTOTUNE)\n    return train_ds, val_ds\n\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(1,K_FOLDS+1):\n    sub_train_df = tile_train_df[tile_train_df.fold!=i].reset_index(drop=True)\n    sub_val_df   = tile_train_df[tile_train_df.fold==i].reset_index(drop=True)\n    \n    sub_train_ds, sub_val_ds = create_ds(sub_train_df, sub_val_df)\n    \n    write_tfrecords(sub_train_ds, len(sub_train_df), out_dir=\"/kaggle/working/tfrecord_dataset\", ds_type=\"train\", fold=i)\n    write_tfrecords(sub_val_ds, len(sub_val_df), out_dir=\"/kaggle/working/tfrecord_dataset\", ds_type=\"val\", fold=i)\n    \n    if i==3:\n        break","metadata":{"execution":{"iopub.status.busy":"2022-07-01T20:20:36.564705Z","iopub.execute_input":"2022-07-01T20:20:36.565189Z","iopub.status.idle":"2022-07-01T20:20:36.59302Z","shell.execute_reply.started":"2022-07-01T20:20:36.565151Z","shell.execute_reply":"2022-07-01T20:20:36.591992Z"},"trusted":true},"execution_count":null,"outputs":[]}]}