{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <div style=\"color:white;display:fill;border-radius:5px;background-color:#75B7BF;letter-spacing:0.1px;overflow:hidden\"><p style=\"padding:20px;color:white;overflow:hidden;margin:0;font-size:100%;text-align:center\">IMPORT</p></div>","metadata":{}},{"cell_type":"code","source":"print(\"\\n... IMPORTS STARTING ...\\n\")\n\nprint(\"\\n\\tVERSION INFORMATION\")\n# Machine Learning and Data Science Imports\nimport tensorflow as tf; print(f\"\\t\\t– TENSORFLOW VERSION: {tf.__version__}\");\nimport tensorflow_hub as tfhub; print(f\"\\t\\t– TENSORFLOW HUB VERSION: {tfhub.__version__}\");\nimport tensorflow_addons as tfa; print(f\"\\t\\t– TENSORFLOW ADDONS VERSION: {tfa.__version__}\");\nimport tensorflow_io as tfio; print(f\"\\t\\t– TENSORFLOW I/O VERSION: {tfio.__version__}\");\nimport pandas as pd; pd.options.mode.chained_assignment = None;\nimport numpy as np; print(f\"\\t\\t– NUMPY VERSION: {np.__version__}\");\nimport sklearn; print(f\"\\t\\t– SKLEARN VERSION: {sklearn.__version__}\");\nfrom sklearn.preprocessing import RobustScaler, PolynomialFeatures\nfrom pandarallel import pandarallel; pandarallel.initialize();\nfrom sklearn.model_selection import GroupKFold, StratifiedKFold\nfrom scipy.spatial import cKDTree\n\n# Built In Imports\nfrom kaggle_datasets import KaggleDatasets\nfrom collections import Counter\nfrom datetime import datetime\nfrom glob import glob\nimport warnings\nimport requests\nimport hashlib\nimport imageio\nimport IPython\nimport sklearn\nimport urllib\nimport zipfile\nimport pickle\nimport random\nimport shutil\nimport string\nimport json\nimport math\nimport time\nimport gzip\nimport ast\nimport sys\nimport io\nimport os\nimport gc\nimport re\n\n# Visualization Imports\nfrom matplotlib.colors import ListedColormap\nfrom matplotlib.patches import Rectangle\nimport matplotlib.patches as patches\nimport plotly.graph_objects as go\nimport plotly.express as px\nimport plotly.io as pio\nprint(pio.renderers)\n\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm; tqdm.pandas();\n\nimport seaborn as sns\nfrom PIL import Image, ImageEnhance\nimport matplotlib; print(f\"\\t\\t– MATPLOTLIB VERSION: {matplotlib.__version__}\");\nfrom matplotlib import animation, rc, gridspec \nrc('animation', html='jshtml')\nimport plotly\nimport cv2 as cv\n\n\n\ndef seed_it_all(seed=7):\n    \"\"\" Attempt to be Reproducible \"\"\"\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n\n    \nprint(\"\\n\\n... IMPORTS COMPLETE ...\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:21:29.547955Z","iopub.execute_input":"2022-07-05T07:21:29.548455Z","iopub.status.idle":"2022-07-05T07:21:44.316264Z","shell.execute_reply.started":"2022-07-05T07:21:29.548355Z","shell.execute_reply":"2022-07-05T07:21:44.314636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:white;display:fill;border-radius:5px;background-color:#75B7BF;letter-spacing:0.1px;overflow:hidden\"><p style=\"padding:20px;color:white;overflow:hidden;margin:0;font-size:100%;text-align:center\">BACKGROUND INFORMATION</p></div>\n****","metadata":{}},{"cell_type":"markdown","source":"## <div style=\"color:white;display:fill;border-radius:5px;background-color:#75B7BF;letter-spacing:0.1px;overflow:hidden\"><p style=\"padding:20px;color:white;overflow:hidden;margin:0;font-size:70%;text-align:left\">1. BASIC COMPETITION INFORMATION</p></div>\n****","metadata":{}},{"cell_type":"markdown","source":"<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">PRIMARY TASK DESCRIPTION</b>\n\nIn this competition, you’ll <mark><b>identify and segment functional tissue units</b></mark> **(FTUs)** across <mark><b>five</b></mark> human organs.\n* You'll build your model using a dataset of tissue section images, with the best submissions segmenting FTUs as accurately as possible.\n* The five human organs we are segmenting are:\n    * **Prostate**\n    * **Spleen**\n    * **Lung**\n    * **Kidney**\n    * **Large Intestine**\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">GENERAL EVALUATION INFORMATION</b>\n\nThis competition is evaluated on the mean <b><a href=\"https://radiopaedia.org/articles/dice-similarity-coefficient#:~:text=The%20Dice%20similarity%20coefficient%2C%20also,between%20two%20sets%20of%20data.\">Dice Coefficient</a></b>. \n\nThe <b><a href=\"https://radiopaedia.org/articles/dice-similarity-coefficient#:~:text=The%20Dice%20similarity%20coefficient%2C%20also,between%20two%20sets%20of%20data.\">Dice Coefficient</a></b> can be used to compare the pixel-wise agreement between a predicted segmentation and its corresponding ground truth. The formula is given by:\n\n$$\n\\frac{2 * |X \\cap Y|}{|X| + |Y|}\n$$\n\n* Where $X$ is the predicted set of pixels and $Y$ is the ground truth. \n* The Dice coefficient is defined to be $1$ when both $X$ and $Y$ are empty. \n* The leaderboard score is the mean of the <b><a href=\"https://radiopaedia.org/articles/dice-similarity-coefficient#:~:text=The%20Dice%20similarity%20coefficient%2C%20also,between%20two%20sets%20of%20data.\">Dice Coefficients</a></b> for each image in the test set.\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"## <div style=\"color:white;display:fill;border-radius:5px;background-color:#75B7BF;letter-spacing:0.1px;overflow:hidden\"><p style=\"padding:20px;color:white;overflow:hidden;margin:0;font-size:70%;text-align:left\">2. DATASET OVERVIEW</p></div>\n****","metadata":{}},{"cell_type":"markdown","source":"\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">GENERAL INFORMATION</b>\n\n<b><mark>The goal of this competition is to identify the locations of each functional tissue unit (FTU) in biopsy slides from several different organs.</mark></b>. \n\nThe training **<mark>annotations are provided as RLE-encoded masks</mark>**, and the images are in **<mark>16-bit</mark>**, **<mark>grayscale</mark>**, **<mark>PNG format</mark>**.\n\nThe underlying data includes <b>imagery from different sources</b> prepared with <b>different protocols</b> at a <b>variety of resolutions</b>, reflecting typical challenges for working with medical data.\n\nThis competition uses data from two different consortia, the <b><a href=\"https://www.nature.com/articles/s41556-021-00788-6\">Human Reference Atlas (HPA)</a></b> and <b><a href=\"https://hubmapconsortium.org/\">Human BioMolecular Atlas Program (HuBMAP)</a></b>. \n\nThe data is sourced as following:\n* The training dataset consists of data from public HPA data\n* the public test set is a combination of private HPA data **and HuBMAP data**\n* the private test set contains **only HuBMAP data**. \n\n<br>\n\n**Adapting models to function properly when presented with data that was prepared using a different protocol will be one of the core challenges of this competition. While this is expected to make the problem more difficult, developing models that generalize is a key goal of this endeavor.**\n\n<br>\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">FILE INFORMATION</b>\n\n**[`train|test.csv`]**\n\n<ul>\n<li><code>id</code> - The image ID.</li>\n<li><code>organ</code> - The organ that the biopsy sample was taken from.</li>\n<li><code>data_source</code> - Whether the image was provided by Hubamp or HPA.</li>\n<li><code>img_height</code> - The height of the image in pixels.</li>\n<li><code>img_width</code> - The width of the image in pixels.</li>\n<li><code>pixel_size</code> - The height/width of a single pixel from this image in micrometers. All HPA images have a pixel size of 0.4 µm. For Hubmap imagery the pixel size is 0.5 µm for kidney, 0.2290 µm for large intestine, 0.7562 µm for lung, 0.4945 µm for spleen, and 6.263 µm for prostate.</li>\n<li><code>tissue_thickness</code> - The thickness of the biopsy sample in micrometers. All HPA images have a thickness of 4 µm. The Hubmap samples have tissue slice thicknesses 10 µm for kidney, 8 µm for large intestine, 4 µm for spleen, 5 µm for lung, and 5 µm for prostate.</li>\n<li><code>rle</code> - The target column. A run length encoded copy of the annotations. Provided for the training set only.</li>\n<li><code>age</code> - The patient's age in years. Provided for the training set only.</li>\n<li><code>sex</code> - The sex of the patient. Provided for the training set only.</li>\n</ul>\n\n<br>\n\n**[`sample_submission.csv`]**\n\n<ul>\n<li><code>id</code> - The image ID.</li>\n<li><code>rle</code> - A run length encoded mask of the FTUs in the image.</li>\n</ul>\n\n<br>\n\n**[`train|test_images`]**\n \nThe images\n* Expect roughly **550 images in the hidden test set**. \n* All images used have **at least one FTU**.\n* All tissue data used in this competition is **from healthy donors that pathologists identified as pathologically unremarkable tissue**.\n* HPA details:\n    * All HPA images are **3000 x 3000 pixels** with a **tissue area within the image around 2500 x 2500 pixels**. \n    * HPA samples were stained with antibodies visualized with 3,3'-diaminobenzidine (DAB) and counterstained with hematoxylin. \n* HuBMAP details:\n    * The Hubmap images range in size from **4500x4500** down to **160x160 pixels**. \n    * HuBMAP images were prepared using Periodic acid-Schiff (PAS)/hematoxylin and eosin (H&E) stains. \n\n<br>\n\n**[`train_annotations`]**\n\nThe annotations \n* Provided in the format of **points that define the boundaries of the polygon masks of the FTUs**\n\n<br>\n","metadata":{}},{"cell_type":"markdown","source":"# <div style=\"color:white;display:fill;border-radius:5px;background-color:#75B7BF;letter-spacing:0.1px;overflow:hidden\"><p style=\"padding:20px;color:white;overflow:hidden;margin:0;font-size:100%;text-align:center\">SETUP</p></div>\n****","metadata":{}},{"cell_type":"code","source":"print(f\"\\n... ACCELERATOR SETUP STARTING ...\\n\")\n\n# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is set. On Kaggle this is always the case.\n    TPU = tf.distribute.cluster_resolver.TPUClusterResolver()  \nexcept ValueError:\n    TPU = None\n\nif TPU:\n    print(f\"\\n... RUNNING ON TPU - {TPU.master()}...\")\n    tf.config.experimental_connect_to_cluster(TPU)\n    tf.tpu.experimental.initialize_tpu_system(TPU)\n    strategy = tf.distribute.experimental.TPUStrategy(TPU)\nelse:\n    print(f\"\\n... RUNNING ON CPU/GPU ...\")\n    # Yield the default distribution strategy in Tensorflow\n    #   --> Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy() \n\n# What Is a Replica?\n#    --> A single Cloud TPU device consists of FOUR chips, each of which has TWO TPU cores. \n#    --> Therefore, for efficient utilization of Cloud TPU, a program should make use of each of the EIGHT (4x2) cores. \n#    --> Each replica is essentially a copy of the training graph that is run on each core and \n#        trains a mini-batch containing 1/8th of the overall batch size\nN_REPLICAS = strategy.num_replicas_in_sync\n    \nprint(f\"... # OF REPLICAS: {N_REPLICAS} ...\\n\")\n\nprint(f\"\\n... ACCELERATOR SETUP COMPLTED ...\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:21:44.318587Z","iopub.execute_input":"2022-07-05T07:21:44.319073Z","iopub.status.idle":"2022-07-05T07:21:44.339733Z","shell.execute_reply.started":"2022-07-05T07:21:44.319037Z","shell.execute_reply":"2022-07-05T07:21:44.338407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:white;display:fill;border-radius:5px;background-color:#75B7BF;letter-spacing:0.1px;overflow:hidden\"><p style=\"padding:20px;color:white;overflow:hidden;margin:0;font-size:100%;text-align:center\">CONFIGURATION and HANDING FUNCTIONS</p></div>\n****","metadata":{}},{"cell_type":"code","source":"class CFG:\n    ORGANS = ['kidney', 'largeintestine', 'lung', 'prostate', 'spleen']\n    _COLOURS = [(230, 0, 73), (11, 180, 255), (80, 233, 145), (230, 216, 0), (155, 25, 245)]\n    O2C_MAP = {_o:_c for _o,_c in zip(ORGANS, _COLOURS)}\n    O2C_HEX_MAP = {'kidney': '#e60049',\n                   'largeintestine': '#0bb4ff',\n                   'lung': '#50e991',\n                   'prostate': '#e6d800',\n                   'spleen': '#9b19f5'}\n    MAX_N_FTU = 50\n    N_EX = 10","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:21:44.341648Z","iopub.execute_input":"2022-07-05T07:21:44.342923Z","iopub.status.idle":"2022-07-05T07:21:44.376850Z","shell.execute_reply.started":"2022-07-05T07:21:44.342858Z","shell.execute_reply":"2022-07-05T07:21:44.375744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ref: https://www.kaggle.com/paulorzp/run-length-encode-and-decode\n# modified from: https://www.kaggle.com/inversion/run-length-decoding-quick-start\ndef rle_decode(mask_rle, shape, color=1):\n    \"\"\" TBD\n    \n    Args:\n        mask_rle (str): run-length as string formated (start length)\n        shape (tuple of ints): (height,width) of array to return \n    \n    Returns: \n        Mask (np.array)\n            - 1 indicating mask\n            - 0 indicating background\n\n    \"\"\"\n    # Split the string by space, then convert it into a integer array\n    s = np.array(mask_rle.split(), dtype=int)\n\n    # Every even value is the start, every odd value is the \"run\" length\n    starts = s[0::2] - 1\n    lengths = s[1::2]\n    ends = starts + lengths\n\n    # The image image is actually flattened since RLE is a 1D \"run\"\n    if len(shape)==3:\n        h, w, d = shape\n        img = np.zeros((h * w, d), dtype=np.float32)\n    else:\n        h, w = shape\n        img = np.zeros((h * w,), dtype=np.float32)\n\n    # The color here is actually just any integer you want!\n    for lo, hi in zip(starts, ends):\n        img[lo : hi] = color\n        \n    # Don't forget to change the image back to the original shape\n    return img.reshape(shape).T\n# Test\n# data = train_full.rle[0]\n# rle_decode(data, (3000,3000))","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:21:44.379831Z","iopub.execute_input":"2022-07-05T07:21:44.380205Z","iopub.status.idle":"2022-07-05T07:21:44.393859Z","shell.execute_reply.started":"2022-07-05T07:21:44.380171Z","shell.execute_reply":"2022-07-05T07:21:44.392413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/namgalielei/which-reshape-is-used-in-rle\ndef rle_decode_top_to_bot_first(mask_rle, shape):\n    \"\"\" TBD\n    \n    Args:\n        mask_rle (str): run-length as string formated (start length)\n        shape (tuple of ints): (height,width) of array to return \n    \n    Returns:\n        Mask (np.array)\n            - 1 indicating mask\n            - 0 indicating background\n\n    \"\"\"\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape((shape[1], shape[0]), order='F').T  # Reshape from top -> bottom first","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:21:44.395522Z","iopub.execute_input":"2022-07-05T07:21:44.395976Z","iopub.status.idle":"2022-07-05T07:21:44.413095Z","shell.execute_reply.started":"2022-07-05T07:21:44.395944Z","shell.execute_reply":"2022-07-05T07:21:44.411902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ref.: https://www.kaggle.com/stainsby/fast-tested-rle\ndef rle_encode(img):\n    \"\"\" TBD\n    \n    Args:\n        img (np.array): \n            - 1 indicating mask\n            - 0 indicating background\n    \n    Returns: \n        run length as string formated\n    \"\"\"\n    pixels = img.T.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:21:44.414927Z","iopub.execute_input":"2022-07-05T07:21:44.415277Z","iopub.status.idle":"2022-07-05T07:21:44.430410Z","shell.execute_reply.started":"2022-07-05T07:21:44.415247Z","shell.execute_reply":"2022-07-05T07:21:44.429136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ref.: https://www.kaggle.com/code/dschettler8845/eda-hubmap-hpa-organ-segmentation\ndef flatten_listOflist(nested_list):\n    \"\"\" Flatten a list of lists \"\"\"\n    return [item for sublist in nested_list for item in sublist]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:21:44.432197Z","iopub.execute_input":"2022-07-05T07:21:44.433301Z","iopub.status.idle":"2022-07-05T07:21:44.441624Z","shell.execute_reply.started":"2022-07-05T07:21:44.433250Z","shell.execute_reply":"2022-07-05T07:21:44.440464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ref.: https://www.kaggle.com/code/dschettler8845/eda-hubmap-hpa-organ-segmentation\ndef load_json_to_dict(json_path):\n    \"\"\" tbd \"\"\"\n    with open(json_path) as json_file:\n        data = json.load(json_file)\n    return data","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:21:44.443448Z","iopub.execute_input":"2022-07-05T07:21:44.444670Z","iopub.status.idle":"2022-07-05T07:21:44.454758Z","shell.execute_reply.started":"2022-07-05T07:21:44.444628Z","shell.execute_reply":"2022-07-05T07:21:44.453666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ref.: https://www.kaggle.com/code/dschettler8845/eda-hubmap-hpa-organ-segmentation\ndef tf_decode_tiff(img_path, to_numpy=False, to_rgb=False):\n    img = tf.io.read_file(img_path)\n    img = tfio.experimental.image.decode_tiff(img)\n    \n    # Optionals\n    if to_rgb: img = tfio.experimental.color.rgba_to_rgb(img)\n    if to_numpy: img = img.numpy()\n        \n    return img","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:21:44.456413Z","iopub.execute_input":"2022-07-05T07:21:44.456776Z","iopub.status.idle":"2022-07-05T07:21:44.468180Z","shell.execute_reply.started":"2022-07-05T07:21:44.456740Z","shell.execute_reply":"2022-07-05T07:21:44.466885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_overlay(img_path, \n                organ, \n                rle_str, \n                img_shape, \n                _alpha=0.999, \n                _beta=0.35, \n                _gamma=0):\n    _img = tf_decode_tiff(img_path, to_numpy=True, to_rgb=True).astype(np.float32)\n    _seg_rgb = (np.stack([rle_decode(rle_str, shape=img_shape, color=1),]*3, axis=-1)*CFG.O2C_MAP[organ]).astype(np.float32)\n    seg_overlay = cv.addWeighted(src1=_img, \n                                  alpha=_alpha, \n                                  src2=_seg_rgb, \n                                  beta=_beta, \n                                  gamma=_gamma)\n    return seg_overlay/255.","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:21:44.472377Z","iopub.execute_input":"2022-07-05T07:21:44.473124Z","iopub.status.idle":"2022-07-05T07:21:44.484748Z","shell.execute_reply.started":"2022-07-05T07:21:44.473081Z","shell.execute_reply":"2022-07-05T07:21:44.483633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def examine_id(df, \n               ex_id=None, \n               plot_overlay=True, \n               print_meta=True, \n               plot_original=False, \n               plot_segmentation=False, \n               _figsize=(10,10)):\n    \"\"\" Wrapper function to allow for easy visual exploration of a sample \"\"\"\n    if ex_id is None:\n        ex_id = df.id.sample(1).values[0]\n        print(f\"\\n... NO ID GIVEN... RANDOM ID CHOSEN: ID={ex_id} ...\")\n    \n    print(f\"\\n... ID ({ex_id}) EXPLORATION STARTED ...\\n\\n\")\n    demo_ex = df[df.id==ex_id].squeeze()\n\n    if print_meta:\n        print(f\"\\n... WITH id=`{ex_id}` (organ=`{demo_ex['organ']}`)  –  WE HAVE THE FOLLOWING DEMO EXAMPLE TO WORK FROM ... \\n\\n\")\n        display(demo_ex.to_frame())\n\n    if plot_original:\n        print(f\"\\n\\n... ORIGINAL IMAGE PLOT ...\\n\")\n        plt.figure(figsize=_figsize)\n        plt.imshow(tf_decode_tiff(demo_ex[\"img_path\"], to_numpy=True, to_rgb=True))\n        plt.title(f\"Original RGB Image For ID: {ex_id}\", fontweight=\"bold\")\n        plt.axis(False)\n        plt.show()\n\n    if plot_segmentation:\n        print(f\"\\n\\n... SEGMENTATION MASK PLOT ({demo_ex['organ']})...\\n\")\n        plt.figure(figsize=_figsize)\n        plt.imshow((np.stack([rle_decode(demo_ex.rle, shape=(demo_ex.img_width, demo_ex.img_height), color=1),]*3, axis=-1)*O2C_MAP[demo_ex.organ]).astype(np.float32))\n        plt.title(f\"Segmentation Mask ({demo_ex.organ})\", fontweight=\"bold\")\n        plt.axis(False)\n        plt.show()\n\n    if plot_overlay:\n        print(f\"\\n\\n... IMAGE WITH RGB SEGMENTATION MASK OVERLAY ({demo_ex['organ']}) ...\\n\")\n        seg_overlay = get_overlay(demo_ex.img_path, \n                                  demo_ex.organ, \n                                  demo_ex.rle, \n                                  img_shape=(demo_ex.img_width, demo_ex.img_height))\n\n        plt.figure(figsize=_figsize)\n        plt.imshow(seg_overlay)\n        plt.title(f\"Segmentation Overlay id=`{ex_id}` (organ=`{demo_ex.organ}`)\", fontweight=\"bold\")\n        handles = [Rectangle((0,0),1,1, color=(*[__c/255. for __c in _c], 0.5)) for _c in CFG._COLOURS]\n        labels = CFG.ORGANS\n        plt.legend(handles,labels)\n        plt.axis(False)\n        plt.show()\n\n    print(\"\\n\\n... SINGLE ID EXPLORATION FINISHED ...\\n\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:21:44.486068Z","iopub.execute_input":"2022-07-05T07:21:44.486589Z","iopub.status.idle":"2022-07-05T07:21:44.506689Z","shell.execute_reply.started":"2022-07-05T07:21:44.486552Z","shell.execute_reply":"2022-07-05T07:21:44.505204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_ftus(data, n_cols=4, height_per_row = 7):\n    \n    max_crops = 10*n_cols\n    data = data[:max_crops]\n    n_rows = int(np.ceil(len(data)/n_cols))\n\n    plt.figure(figsize=(20,n_rows*height_per_row))\n    for i, ftu_crop in enumerate(data):\n        plt.subplot(n_rows, n_cols, i+1)\n        plt.imshow(ftu_crop)\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:21:44.508136Z","iopub.execute_input":"2022-07-05T07:21:44.508505Z","iopub.status.idle":"2022-07-05T07:21:44.522911Z","shell.execute_reply.started":"2022-07-05T07:21:44.508467Z","shell.execute_reply":"2022-07-05T07:21:44.521580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def crop_ftu_polygons(img, cnts, pad=5):\n    ftu_crops = []\n    for cnt in cnts:\n        x1, y1, w, h = cv.boundingRect(np.array(cnt))\n        x2, y2 = x1 + w, y1 + h\n        x1, y1 = max(0, x1-pad), max(0, y1-pad)\n        ftu_crops.append(img[y1:y2, x1:x2])\n    return ftu_crops","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:21:44.526447Z","iopub.execute_input":"2022-07-05T07:21:44.527274Z","iopub.status.idle":"2022-07-05T07:21:44.535113Z","shell.execute_reply.started":"2022-07-05T07:21:44.527221Z","shell.execute_reply":"2022-07-05T07:21:44.533943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:white;display:fill;border-radius:5px;background-color:#75B7BF;letter-spacing:0.1px;overflow:hidden\"><p style=\"padding:20px;color:white;overflow:hidden;margin:0;font-size:100%;text-align:center\">LOADING DATASET</p></div>","metadata":{}},{"cell_type":"code","source":"print(\"\\n... DATA ACCESS SETUP STARTED ...\\n\")\n\nif TPU:\n    # Google Cloud Dataset path to training and validation images\n    DATA_DIR = KaggleDatasets().get_gcs_path('hubmap-organ-segmentation')\n    save_locally = tf.saved_model.SaveOptions(experimental_io_device='/job:localhost')\n    load_locally = tf.saved_model.LoadOptions(experimental_io_device='/job:localhost')\nelse:\n    # Local path to training and validation images\n    DATA_DIR = \"/kaggle/input/hubmap-organ-segmentation\"\n    save_locally = None\n    load_locally = None\n\nprint(f\"\\n... DATA DIRECTORY PATH IS:\\n\\t--> {DATA_DIR}\")\n\nprint(f\"\\n... IMMEDIATE CONTENTS OF DATA DIRECTORY IS:\")\nfor file in tf.io.gfile.glob(os.path.join(DATA_DIR, \"*\")): print(f\"\\t--> {file}\")\n\nprint(\"\\n\\n... DATA ACCESS SETUP COMPLETED ...\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:21:44.536501Z","iopub.execute_input":"2022-07-05T07:21:44.536840Z","iopub.status.idle":"2022-07-05T07:21:44.559741Z","shell.execute_reply.started":"2022-07-05T07:21:44.536809Z","shell.execute_reply":"2022-07-05T07:21:44.558453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## <div style=\"color:white;display:fill;border-radius:5px;background-color:#75B7BF;letter-spacing:0.1px;overflow:hidden\"><p style=\"padding:20px;color:white;overflow:hidden;margin:0;font-size:70%;text-align:left\">BASIC DATA DEFINITIONS & INITIALIZATIONS</p></div> ","metadata":{}},{"cell_type":"code","source":"print(\"\\n... BASIC DATA SETUP STARTING ...\\n\\n\")\nTRAIN_DIR = os.path.join(DATA_DIR, \"train_images\")\nTRAIN_ANNOTATIONS_DIR = os.path.join(DATA_DIR, \"train_annotations\")\nTRAIN_CSV = os.path.join(DATA_DIR, \"train.csv\")\ntrain_full = pd.read_csv(TRAIN_CSV)\n# Get all training images\nall_train_images = glob(os.path.join(TRAIN_DIR, \"**\", \"*.tiff\"), recursive=True)\nall_train_annotations = glob(os.path.join(TRAIN_ANNOTATIONS_DIR, \"*.json\"), recursive=True)\n\nprint(\"\\n... ORIGINAL TRAINING DATAFRAME... \\n\")\ndisplay(train_full[:5])\n\n# Get all testing images if there are any\nTEST_DIR = os.path.join(DATA_DIR, \"test_images\")\nall_test_images = glob(os.path.join(TEST_DIR, \"**\", \"*.tiff\"), recursive=True)\n\nTEST_CSV = os.path.join(DATA_DIR, \"test.csv\")\ntest_full = pd.read_csv(TEST_CSV)\n\nprint(\"\\n... ORIGINAL TESTING DATAFRAME... \\n\")\ndisplay(test_full)\n\nSS_CSV   = os.path.join(DATA_DIR, \"sample_submission.csv\")\nss_df = pd.read_csv(SS_CSV)\n\nprint(\"\\n\\n\\n... ORIGINAL SUBMISSION DATAFRAME... \\n\")\ndisplay(ss_df)\n\n# For debugging purposes when the test set hasn't been substituted we will know\nDEBUG=len(ss_df)==0\n\nif DEBUG:\n    TEST_DIR = TRAIN_DIR\n    all_test_images = all_train_images\n    ss_df = train_full.iloc[:10]\n    ss_df = ss_df[[\"id\", \"class\"]]\n    ss_df[\"predicted\"] = \"\"\n    \n    print(\"\\n\\n\\n... DEBUG SUBMISSION DATAFRAME... \\n\")\n    display(ss_df)\n\nprint(\"\\n... BASIC DATA SETUP FINISHED ...\\n\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:21:44.561396Z","iopub.execute_input":"2022-07-05T07:21:44.562390Z","iopub.status.idle":"2022-07-05T07:21:45.028941Z","shell.execute_reply.started":"2022-07-05T07:21:44.562327Z","shell.execute_reply":"2022-07-05T07:21:45.026498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## <div style=\"color:white;display:fill;border-radius:5px;background-color:#75B7BF;letter-spacing:0.1px;overflow:hidden\"><p style=\"padding:20px;color:white;overflow:hidden;margin:0;font-size:70%;text-align:left\">UPDATE DATAFRAMES WITH ACCESSIBLE EXTRA INFORMATION</p></div>  \n****","metadata":{}},{"cell_type":"code","source":"# Convert gender to Integer\nsex_2_int = {\"Male\":0, \"Female\":1}\ntrain_full.sex = train_full.sex.map(sex_2_int)\n# Change age to percentage\ntrain_full.age = train_full.age/100.0","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:21:45.030897Z","iopub.execute_input":"2022-07-05T07:21:45.031646Z","iopub.status.idle":"2022-07-05T07:21:45.048293Z","shell.execute_reply.started":"2022-07-05T07:21:45.031599Z","shell.execute_reply":"2022-07-05T07:21:45.046836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_img_map = {int(x[:-5].rsplit(\"/\", 1)[-1]):x for x in all_train_images}\ntrain_ann_map = {int(x[:-5].rsplit(\"/\", 1)[-1]):x for x in all_train_annotations}\ntest_img_map = {int(x[:-5].rsplit(\"/\", 1)[-1]):x for x in all_test_images}\n\ntrain_full.insert(3, \"img_path\", train_full.id.map(train_img_map))\ntrain_full.insert(4, \"ann_path\", train_full.id.map(train_ann_map))\ntest_full.insert(3, \"img_path\", test_full.id.map(test_img_map))","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:21:45.050172Z","iopub.execute_input":"2022-07-05T07:21:45.050546Z","iopub.status.idle":"2022-07-05T07:21:45.105387Z","shell.execute_reply.started":"2022-07-05T07:21:45.050514Z","shell.execute_reply":"2022-07-05T07:21:45.103118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_full.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:21:45.107446Z","iopub.execute_input":"2022-07-05T07:21:45.108327Z","iopub.status.idle":"2022-07-05T07:21:45.151127Z","shell.execute_reply.started":"2022-07-05T07:21:45.108280Z","shell.execute_reply":"2022-07-05T07:21:45.149577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:white;display:fill;border-radius:5px;background-color:#75B7BF;letter-spacing:0.1px;overflow:hidden\"><p style=\"padding:20px;color:white;overflow:hidden;margin:0;font-size:100%;text-align:center\">DATASET EXPLORATION</p></div>","metadata":{}},{"cell_type":"markdown","source":"## <div style=\"color:white;display:fill;border-radius:5px;background-color:#75B7BF;letter-spacing:0.1px;overflow:hidden\"><p style=\"padding:20px;color:white;overflow:hidden;margin:0;font-size:70%;text-align:left\">SIMPLE CASES</p></div>  \n****","metadata":{}},{"cell_type":"code","source":"for organ in CFG.ORGANS:\n    print(f'The organ is : {organ.upper()}')\n    examine_id(train_full[train_full.organ==organ])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:21:45.152887Z","iopub.execute_input":"2022-07-05T07:21:45.153644Z","iopub.status.idle":"2022-07-05T07:22:11.151727Z","shell.execute_reply.started":"2022-07-05T07:21:45.153598Z","shell.execute_reply":"2022-07-05T07:22:11.150226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## <div style=\"color:white;display:fill;border-radius:5px;background-color:#75B7BF;letter-spacing:0.1px;overflow:hidden\"><p style=\"padding:20px;color:white;overflow:hidden;margin:0;font-size:70%;text-align:left\">IMAGE SIZES</p></div>  \n****","metadata":{}},{"cell_type":"code","source":"_data = train_full.drop_duplicates(subset=[\"img_width\", \"img_height\"])\n_data.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:22:11.153827Z","iopub.execute_input":"2022-07-05T07:22:11.154847Z","iopub.status.idle":"2022-07-05T07:22:11.168593Z","shell.execute_reply.started":"2022-07-05T07:22:11.154797Z","shell.execute_reply":"2022-07-05T07:22:11.167426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**OBSERVATIONS**\n\n* Globally, we can see that most of the images are dominated by a single size $(3000 \\times 3000)$\n* All the other sizes (19) only have 1 or 2 occurences each.\n* All image sizes are square","metadata":{}},{"cell_type":"code","source":"_size = train_full.groupby([\"img_width\", \"img_height\"]).id.transform(\"count\").iloc[_data.index]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:22:11.170292Z","iopub.execute_input":"2022-07-05T07:22:11.170949Z","iopub.status.idle":"2022-07-05T07:22:11.206549Z","shell.execute_reply.started":"2022-07-05T07:22:11.170906Z","shell.execute_reply":"2022-07-05T07:22:11.205359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.scatter(data_frame=_data,\n                 x=\"img_width\",\n                 y=\"img_height\",\n                 size=_size,\n                 color=\"(\"+_data[\"img_width\"].astype(str)+\",\"+_data[\"img_height\"].astype(str)+\")\",\n                 title=\"<b>Bubble Chart Showing The Various Image Sizes</b>\", \n                 labels={\"color\":\"<b>Size Legend</b>\", \n                         \"size\":\"<b>Number Of Observations</b>\",\n                         \"img_height\":\"<b>Image Height (pixels)</b>\",\n                         \"img_width\":\"<b>Image Width (pixels)</b>\"},\n                 size_max=150 )\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:22:11.208017Z","iopub.execute_input":"2022-07-05T07:22:11.208770Z","iopub.status.idle":"2022-07-05T07:22:12.803217Z","shell.execute_reply.started":"2022-07-05T07:22:11.208729Z","shell.execute_reply":"2022-07-05T07:22:12.802142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## <div style=\"color:white;display:fill;border-radius:5px;background-color:#75B7BF;letter-spacing:0.1px;overflow:hidden\"><p style=\"padding:20px;color:white;overflow:hidden;margin:0;font-size:70%;text-align:left\">FREQUENCY OF ORGANS</p></div>  \n****","metadata":{}},{"cell_type":"code","source":"fig = px.histogram(train_full, \n                   \"organ\", \n                   color_discrete_map = CFG.O2C_HEX_MAP, \n                   color=\"organ\",\n                   title=\"<b>Number of Segmentation Masks Per Organ Type</b>\" )\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:22:12.804781Z","iopub.execute_input":"2022-07-05T07:22:12.805972Z","iopub.status.idle":"2022-07-05T07:22:12.928277Z","shell.execute_reply.started":"2022-07-05T07:22:12.805927Z","shell.execute_reply":"2022-07-05T07:22:12.926677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## <div style=\"color:white;display:fill;border-radius:5px;background-color:#75B7BF;letter-spacing:0.1px;overflow:hidden\"><p style=\"padding:20px;color:white;overflow:hidden;margin:0;font-size:70%;text-align:left\">AGES</p></div>  \n****","metadata":{}},{"cell_type":"code","source":"fig = px.histogram(train_full, \n                   train_full.age *100, \n                   color_discrete_map = CFG.O2C_HEX_MAP,\n                   color=\"organ\",\n                   title=\"<b>Number of Patients Per Gender(0:Male-1:Female)</b>\",\n                   barmode='group')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:22:12.929784Z","iopub.execute_input":"2022-07-05T07:22:12.930859Z","iopub.status.idle":"2022-07-05T07:22:13.019567Z","shell.execute_reply.started":"2022-07-05T07:22:12.930804Z","shell.execute_reply":"2022-07-05T07:22:13.018224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## <div style=\"color:white;display:fill;border-radius:5px;background-color:#75B7BF;letter-spacing:0.1px;overflow:hidden\"><p style=\"padding:20px;color:white;overflow:hidden;margin:0;font-size:70%;text-align:left\">GENDERS</p></div>  \n****","metadata":{}},{"cell_type":"code","source":"fig = px.histogram(train_full, \n                   train_full.sex , \n                   color_discrete_map = CFG.O2C_HEX_MAP,\n                   color=\"organ\",\n                   title=\"<b>Number of Patients Per Gender(0:Male-1:Female)</b>\",\n                   barmode='group')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:22:13.021922Z","iopub.execute_input":"2022-07-05T07:22:13.022434Z","iopub.status.idle":"2022-07-05T07:22:13.110864Z","shell.execute_reply.started":"2022-07-05T07:22:13.022387Z","shell.execute_reply":"2022-07-05T07:22:13.109689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## <div style=\"color:white;display:fill;border-radius:5px;background-color:#75B7BF;letter-spacing:0.1px;overflow:hidden\"><p style=\"padding:20px;color:white;overflow:hidden;margin:0;font-size:70%;text-align:left\">FTUS ACROSS DIFFERENT ORGANS</p></div>  \n****\n\nHere is a zoom sequence from the human body to the single-cell level for the kidney. \n* **Notice that FTU is one level higher than cells.**\n\n<center><img src=\"https://i.ibb.co/wRySj9z/results-5-1.png\"></center>\n\n<br>\n\nFTUs in the organs presented in this competition are: \n* **glomeruli** in the **kidney** – (a)\n* **crypt** in the **large intestine**  – (b)\n* **alveolus** in the **lung**  – (c)\n* **glandular acinus** in the **prostate**  – (d)\n* **white pulp** in the **spleen**  – (e)\n\n<center><img src=\"https://i.ibb.co/ZX0ZYYV/results-7-1.png\"></center>\n\n<br>","metadata":{}},{"cell_type":"code","source":"ftu_crop_maps = {}\nfor organ in CFG.ORGANS:\n    df = train_full[train_full.organ == organ].sample(CFG.N_EX)\n    imgs = [cv.imread(img_path) for img_path in df.img_path.values]\n    cnts = [load_json_to_dict(ann_path) for ann_path in df.ann_path.values]\n    ftu_crops = flatten_listOflist([crop_ftu_polygons(imgs, cnts, pad=0) for imgs, cnts in zip(imgs, cnts)])[:CFG.MAX_N_FTU]\n    ## Display the Ftus\n    print(f\"\\n\\n\\n... DISPLAYING {len(ftu_crops)} {organ} FTU CROPS ...\\n\")\n    plot_ftus(ftu_crops)\n    # Save data in a dict\n    ftu_crop_maps[organ] = ftu_crops","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:22:13.112205Z","iopub.execute_input":"2022-07-05T07:22:13.113185Z","iopub.status.idle":"2022-07-05T07:23:18.002091Z","shell.execute_reply.started":"2022-07-05T07:22:13.113148Z","shell.execute_reply":"2022-07-05T07:23:18.000750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## <div style=\"color:white;display:fill;border-radius:5px;background-color:#75B7BF;letter-spacing:0.1px;overflow:hidden\"><p style=\"padding:20px;color:white;overflow:hidden;margin:0;font-size:70%;text-align:left\">AVERAGE FTU SIZE BY ORGAN</p></div>  \n****","metadata":{}},{"cell_type":"code","source":"ftu_shape_map = {_organ:[c_map.shape[:-1] for c_map in crop_maps] for _organ, crop_maps in ftu_crop_maps.items()}\nftu_area_map = {_organ:[c_map.shape[0]*c_map.shape[1] for c_map in crop_maps] for _organ, crop_maps in ftu_crop_maps.items()}","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:23:18.003804Z","iopub.execute_input":"2022-07-05T07:23:18.005076Z","iopub.status.idle":"2022-07-05T07:23:18.014534Z","shell.execute_reply.started":"2022-07-05T07:23:18.005003Z","shell.execute_reply":"2022-07-05T07:23:18.012640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"organ_ftu_df = pd.DataFrame({\"organ\":CFG.ORGANS})\norgan_ftu_df[\"avg_area\"] = organ_ftu_df.organ.apply(lambda x: np.array(ftu_area_map[x]).mean())\n\norgan_ftu_df[\"avg_are_um\"] = organ_ftu_df.avg_area *0.4**2\norgan_ftu_df[\"avg_height\"] = organ_ftu_df.organ.apply(lambda x:np.array(ftu_shape_map[x][0]).mean())\norgan_ftu_df[\"avg_width\"] = organ_ftu_df.organ.apply(lambda x:np.array(ftu_shape_map[x][1]).mean())\norgan_ftu_df[\"avg_shape_µm\"] = (\"(\"+(organ_ftu_df.avg_height*0.4).round(2).astype(str)+\",\" \\\n                                +(organ_ftu_df.avg_width*0.4).round(2).astype(str)+\")\").apply(ast.literal_eval)\norgan_ftu_df","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:23:18.020056Z","iopub.execute_input":"2022-07-05T07:23:18.020623Z","iopub.status.idle":"2022-07-05T07:23:18.052817Z","shell.execute_reply.started":"2022-07-05T07:23:18.020587Z","shell.execute_reply":"2022-07-05T07:23:18.051919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## <div style=\"color:white;display:fill;border-radius:5px;background-color:#75B7BF;letter-spacing:0.1px;overflow:hidden\"><p style=\"padding:20px;color:white;overflow:hidden;margin:0;font-size:70%;text-align:left\">MASK AREA BY ORGAN</p></div>  \n****","metadata":{}},{"cell_type":"code","source":"train_full.head(1)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:23:18.054195Z","iopub.execute_input":"2022-07-05T07:23:18.054556Z","iopub.status.idle":"2022-07-05T07:23:18.070619Z","shell.execute_reply.started":"2022-07-05T07:23:18.054524Z","shell.execute_reply":"2022-07-05T07:23:18.069699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_full[\"mask_area_pct\"] = train_full.rle.apply(lambda x: rle_decode(x, (3000,3000)).sum()/(3000**2))\nfig = px.histogram(train_full, \"mask_area_pct\", \n                   color=\"organ\", \n                   title=\"<b>Mask Area by Organ</b>\", \n                   color_discrete_map=CFG.O2C_HEX_MAP, \n                   barmode=\"group\",\n                   nbins=10)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:23:18.072002Z","iopub.execute_input":"2022-07-05T07:23:18.072439Z","iopub.status.idle":"2022-07-05T07:23:23.714129Z","shell.execute_reply.started":"2022-07-05T07:23:18.072401Z","shell.execute_reply":"2022-07-05T07:23:23.712893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## <div style=\"color:white;display:fill;border-radius:5px;background-color:#75B7BF;letter-spacing:0.1px;overflow:hidden\"><p style=\"padding:20px;color:white;overflow:hidden;margin:0;font-size:70%;text-align:left\">RLE AND JSON FILE</p></div>  \n****\n1.  The RLE is the entire binary mask as a simple compressed string\n2.  The JSON contains the description of the masks as a collection (list) of polygons representing the respective masks.\n    *     A polygon is represented as a collection (list) of vertices.\n    * A vertex is a point represented by a pair of values –– the x,y position\n    * If you took these polygons and plotted them alongside the RLE decoded mask, they SHOULD be the same. I think it's simply another representation for us to use.\n","metadata":{}},{"cell_type":"code","source":"# Use first row as example --> ex_id=10044\nex_row = train_full[train_full.id==10044]\nprint(\"\\n... First – Plot the image and the RLE overlapped ontop ...\\n\")\nexamine_id(ex_row)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:23:23.715728Z","iopub.execute_input":"2022-07-05T07:23:23.716119Z","iopub.status.idle":"2022-07-05T07:23:27.733626Z","shell.execute_reply.started":"2022-07-05T07:23:23.716086Z","shell.execute_reply":"2022-07-05T07:23:27.732527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\n\\n\\n... Second – Load the .json file and display mask polygons ...\\n\")\nex_poly = load_json_to_dict(ex_row.ann_path.values[0])\nraw_img = np.array(cv.imread(ex_row.img_path[0]))\nimg = np.zeros_like(raw_img)\ncolor_space = np.linspace(235, 20, len(ex_poly), dtype=int)\nfor i, ex_p in enumerate(ex_poly):\n    img = cv.fillPoly(img, \n                      [np.expand_dims(np.array(ex_p, np.int32), axis=0)], \n                      color=(int(color_space[i]),245,245))\n    \nplt.figure(figsize=(10,10))\nplt.title(\"Mask Poly\", fontweight=\"bold\")\nplt.imshow(img)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:23:27.735656Z","iopub.execute_input":"2022-07-05T07:23:27.736452Z","iopub.status.idle":"2022-07-05T07:23:29.011189Z","shell.execute_reply.started":"2022-07-05T07:23:27.736407Z","shell.execute_reply":"2022-07-05T07:23:29.009895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\n\\n\\n Lastly – Compare mask polygons to rle decoded mask\")\n\nrle_mask = rle_decode(ex_row.rle.values[0], (ex_row.img_width.values[0], ex_row.img_height.values[0]))\njson_mask = np.where(img>0, 0, 1)[..., 0].astype(np.float32)\ndiff_mask = np.abs(rle_mask-json_mask).astype(np.uint8)\nplt.figure(figsize=(10,10))\nplt.title(f\"Number of Non-Equal Pixels = {(rle_mask!=json_mask).sum()}\", fontweight=\"bold\")\nplt.imshow(diff_mask, cmap=\"gray\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:23:29.012547Z","iopub.execute_input":"2022-07-05T07:23:29.012899Z","iopub.status.idle":"2022-07-05T07:23:30.608869Z","shell.execute_reply.started":"2022-07-05T07:23:29.012867Z","shell.execute_reply":"2022-07-05T07:23:30.607142Z"},"trusted":true},"execution_count":null,"outputs":[]}]}