{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pathlib\nimport json\nimport numpy as np\nimport pandas as pd\nimport tifffile\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"11622959-8e7e-474c-b8c9-c398b0d58258","_cell_guid":"a9beb333-21dc-417e-9643-ca3fcc30e49c","collapsed":false,"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-06T06:39:41.207600Z","iopub.execute_input":"2022-08-06T06:39:41.208118Z","iopub.status.idle":"2022-08-06T06:39:41.216259Z","shell.execute_reply.started":"2022-08-06T06:39:41.208077Z","shell.execute_reply":"2022-08-06T06:39:41.214571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## HuBMAP + HPA - Hacking the Human Body","metadata":{}},{"cell_type":"markdown","source":"## 1. Introduction\n\nPurpose of this notebook is visualizing all training set images along with their annotations and the single test set image by itself. Both run-length encoded masks and raw polygons are visualized. Yellow overlays are run-length encoded masks and red lines are raw polygons. 351 training set and 1 test set images are displayed with the given csv file order.\n\nImage metadata (image ID, organ, data source, sex, image dimensions, pixel size and tissue thickness) are displayed on super titles of the visualizations. Image ID is also printed just before the visualization so an image can be searched directly.","metadata":{}},{"cell_type":"code","source":"INTERNAL_DATASET = pathlib.Path('../input/hubmap-organ-segmentation')\n\ndf_train = pd.read_csv(INTERNAL_DATASET / 'train.csv')\ndf_test = pd.read_csv(INTERNAL_DATASET / 'test.csv')\n\ntrain_images = 'train_images/'\ntrain_annotations = 'train_annotations/'\ntest_images = 'test_images/'\n\ndf_train['image_filename'] = df_train['id'].apply(lambda x:  str(INTERNAL_DATASET) + '/' + train_images + str(x) + '.tiff')\ndf_train['polygon_filename'] = df_train['id'].apply(lambda x:  str(INTERNAL_DATASET) + '/' + train_annotations + str(x) + '.json')\ndf_test['image_filename'] = df_test['id'].apply(lambda x:  str(INTERNAL_DATASET) + '/' + test_images + str(x) + '.tiff')\ndf_test['age'] = np.nan\ndf_test['sex'] = np.nan\n\nprint(f'Training Set Shape: {df_train.shape} - Memory Usage: {df_train.memory_usage().sum() / 1024 ** 2:.2f} MB')\nprint(f'Test Set Shape: {df_test.shape} - Memory Usage: {df_test.memory_usage().sum() / 1024 ** 2:.2f} MB')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T06:39:44.286469Z","iopub.execute_input":"2022-08-06T06:39:44.287739Z","iopub.status.idle":"2022-08-06T06:39:44.493539Z","shell.execute_reply.started":"2022-08-06T06:39:44.287688Z","shell.execute_reply":"2022-08-06T06:39:44.491931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Annotations Utilities","metadata":{}},{"cell_type":"code","source":"def decode_rle_mask(rle_mask, shape):\n\n    \"\"\"\n    Decode run-length encoded segmentation mask string into 2d array\n\n    Parameters\n    ----------\n    rle_mask (str): Run-length encoded segmentation mask string\n    shape (tuple of shape (2)): Height and width of the mask\n\n    Returns\n    -------\n    mask (numpy.ndarray of shape (height, width)): Decoded 2d segmentation mask\n    \"\"\"\n\n    rle_mask = rle_mask.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (rle_mask[0:][::2], rle_mask[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n\n    mask = np.zeros((shape[0] * shape[1]), dtype=np.uint8)\n    for start, end in zip(starts, ends):\n        mask[start:end] = 1\n\n    mask = mask.reshape(shape[0], shape[1])\n    return mask\n\n\ndef encode_rle_mask(mask):\n\n    \"\"\"\n    Encode 2d array into run-length encoded segmentation mask string\n\n    Parameters\n    ----------\n    mask (numpy.ndarray of shape (height, width)): 2d segmentation mask\n\n    Returns\n    -------\n    rle_mask (str): Run-length encoded segmentation mask string\n    \"\"\"\n\n    mask = mask.T.flatten()\n    mask = np.concatenate([[0], mask, [0]])\n    runs = np.where(mask[1:] != mask[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n\n\ndef get_segmentation_polygons_from_json(annotation_directory, image_id):\n\n    \"\"\"\n    Retrieve polygons of given image_id from given directory\n\n    Parameters\n    ----------\n    annotation_directory (str): Directory of the annotations relative to root directory\n    image_id (str): ID of the image\n\n    Returns\n    -------\n    polygons (list of shape (n_polygons, n_points, 2)): Polygons\n    \"\"\"\n\n    with open(INTERNAL_DATASET / annotation_directory / f'{image_id}.json', mode='r') as f:\n        polygons = json.load(f)\n\n    return polygons\n","metadata":{"execution":{"iopub.status.busy":"2022-08-06T06:33:51.163982Z","iopub.execute_input":"2022-08-06T06:33:51.164457Z","iopub.status.idle":"2022-08-06T06:33:51.178578Z","shell.execute_reply.started":"2022-08-06T06:33:51.164421Z","shell.execute_reply":"2022-08-06T06:33:51.176840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Visualization Utilities","metadata":{}},{"cell_type":"code","source":"def visualize_annotations(image, rle_mask, polygons, metadata, path=None):\n\n    \"\"\"\n    Visualize image along with its annotations\n    \n    Parameters\n    ----------\n    image (path-like str or numpy.ndarray of shape (height, width, 3)): Image path relative to root/data or image array\n    rle_mask (str): Run-length encoded segmentation mask string\n    polygons (list of shape (n_polygons, n_points, 2)): Polygons\n    metadata (dict): Dictionary of metadata used in the visualization title\n    path (path-like str or None): Path of the output file or None (if path is None, plot is displayed with selected backend)\n    \"\"\"\n\n    if isinstance(image, pathlib.Path) or isinstance(image, str):\n        # Read image from the given path\n        image_path = image\n        image = tifffile.imread(str(image_path))\n\n    elif isinstance(image, np.ndarray):\n        title = ''\n\n    else:\n        # Raise TypeError if image argument is not an array-like object or a path-like string\n        raise TypeError('Image is not an array or path.')\n\n    if rle_mask is not None:\n        mask = decode_rle_mask(rle_mask=rle_mask, shape=image.shape[:2])\n        if metadata['data_source'] == 'HPA' or metadata['data_source'] == 'Hubmap':\n            mask = mask.T\n\n    if rle_mask is not None and polygons is not None:\n        \n        fig, axes = plt.subplots(figsize=(48, 20), ncols=3)\n\n        axes[0].imshow(image)\n        axes[1].imshow(image)\n\n        if rle_mask is not None:\n            axes[1].imshow(mask, alpha=0.5)\n            axes[2].imshow(mask)\n\n        if polygons is not None:\n            for polygon in polygons:\n                polygon = np.array(polygon)\n                axes[1].plot(polygon[:, 0], polygon[:, 1], linewidth=2, color='red', alpha=0.5)\n                axes[2].plot(polygon[:, 0], polygon[:, 1], linewidth=2, color='red')\n\n        for i in range(3):\n            axes[i].set_xlabel('')\n            axes[i].set_ylabel('')\n            axes[i].tick_params(axis='x', labelsize=15, pad=10)\n            axes[i].tick_params(axis='y', labelsize=15, pad=10)\n\n        axes[0].set_title('Image', size=25, pad=15)\n        axes[1].set_title('Image + Mask and Polygons', size=25, pad=15)\n        axes[2].set_title('Mask and Polygons', size=25, pad=15)\n    \n    else:\n        \n        fig, ax = plt.subplots(figsize=(16, 20))\n        ax.imshow(image)\n        ax.set_xlabel('')\n        ax.set_ylabel('')\n        ax.tick_params(axis='x', labelsize=15, pad=10)\n        ax.tick_params(axis='y', labelsize=15, pad=10)\n        ax.set_title('Image', size=25, pad=15)\n\n    fig.suptitle(\n        f'''\n        Image ID {metadata[\"id\"]} - {metadata[\"organ\"]} - {metadata[\"data_source\"]} - {metadata[\"age\"]} - {metadata[\"sex\"]}\n        Image Shape: {metadata[\"img_height\"]}x{metadata[\"img_width\"]} - Pixel Size: {metadata[\"pixel_size\"]}µm - Tissue Thickness: {metadata[\"tissue_thickness\"]}µm\n        ''',\n        fontsize=50\n    )\n\n    if path is None:\n        plt.show()\n    else:\n        plt.savefig(path)\n        plt.close(fig)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-06T06:45:24.853604Z","iopub.execute_input":"2022-08-06T06:45:24.854002Z","iopub.status.idle":"2022-08-06T06:45:24.873849Z","shell.execute_reply.started":"2022-08-06T06:45:24.853968Z","shell.execute_reply":"2022-08-06T06:45:24.872463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Training Set Images and Annotations","metadata":{}},{"cell_type":"code","source":"for idx, row in df_train.iterrows():\n    \n    image = tifffile.imread(row['image_filename'])\n    rle_mask = row['rle']\n    polygons = get_segmentation_polygons_from_json(annotation_directory='train_annotations', image_id=row['id'])\n    \n    print(f'Image ID {row[\"id\"]}')\n    \n    visualize_annotations(\n        image=image,\n        rle_mask=rle_mask,\n        polygons=polygons,\n        metadata=row.to_dict(),\n        path=None\n    )\n","metadata":{"execution":{"iopub.status.busy":"2022-08-06T06:45:26.865672Z","iopub.execute_input":"2022-08-06T06:45:26.866792Z","iopub.status.idle":"2022-08-06T06:45:32.543275Z","shell.execute_reply.started":"2022-08-06T06:45:26.866743Z","shell.execute_reply":"2022-08-06T06:45:32.541782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Test Set Image","metadata":{}},{"cell_type":"code","source":"for idx, row in df_test.iterrows():\n    \n    image = tifffile.imread(row['image_filename'])\n    rle_mask = None\n    polygons = None\n    \n    print(f'Image ID {row[\"id\"]}')\n    \n    visualize_annotations(\n        image=image,\n        rle_mask=rle_mask,\n        polygons=polygons,\n        metadata=row.to_dict(),\n        path=None\n    )\n","metadata":{"execution":{"iopub.status.busy":"2022-08-06T06:45:32.545093Z","iopub.execute_input":"2022-08-06T06:45:32.545423Z","iopub.status.idle":"2022-08-06T06:45:33.815510Z","shell.execute_reply.started":"2022-08-06T06:45:32.545394Z","shell.execute_reply":"2022-08-06T06:45:33.813877Z"},"trusted":true},"execution_count":null,"outputs":[]}]}