{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# EDA | HuBMAP - Hacking the Human Vasculature Competition","metadata":{}},{"cell_type":"markdown","source":"## Import Statements","metadata":{}},{"cell_type":"code","source":"# For data processing\nimport numpy as np\nimport pandas as pd\nimport json\nfrom pathlib import Path\nimport PIL\nimport skimage\nfrom shapely.geometry import LinearRing as ShapelyContour\nfrom shapely.geometry import Polygon as ShapelyPolygon","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:46:23.270041Z","iopub.execute_input":"2023-07-31T19:46:23.270425Z","iopub.status.idle":"2023-07-31T19:46:23.383485Z","shell.execute_reply.started":"2023-07-31T19:46:23.270382Z","shell.execute_reply":"2023-07-31T19:46:23.382138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For visualization\nimport plotly\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:46:23.385810Z","iopub.execute_input":"2023-07-31T19:46:23.386171Z","iopub.status.idle":"2023-07-31T19:46:24.192943Z","shell.execute_reply.started":"2023-07-31T19:46:23.386139Z","shell.execute_reply":"2023-07-31T19:46:24.191662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Setting Global Variables","metadata":{}},{"cell_type":"code","source":"# For reproducibility\nseed = 123\nnp.random.seed(seed)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:46:24.197334Z","iopub.execute_input":"2023-07-31T19:46:24.197723Z","iopub.status.idle":"2023-07-31T19:46:24.203042Z","shell.execute_reply.started":"2023-07-31T19:46:24.197689Z","shell.execute_reply":"2023-07-31T19:46:24.201851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_DIR = \"/kaggle/input/hubmap-hacking-the-human-vasculature/train\"\nTEST_DIR = \"/kaggle/input/hubmap-hacking-the-human-vasculature/test\"\nANNOT_PATH = \"/kaggle/input/hubmap-hacking-the-human-vasculature/polygons.jsonl\"\nWSI_TILE_CSV = \"/kaggle/input/hubmap-hacking-the-human-vasculature/tile_meta.csv\"\nIMG_SIZE = 512\nIMG_EXT = \"tif\"","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:46:24.204576Z","iopub.execute_input":"2023-07-31T19:46:24.205013Z","iopub.status.idle":"2023-07-31T19:46:24.217293Z","shell.execute_reply.started":"2023-07-31T19:46:24.204971Z","shell.execute_reply":"2023-07-31T19:46:24.216176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ids = [path.stem for path in Path(TRAIN_DIR).glob(f\"*.{IMG_EXT}\")]\nprint(f\"{len(train_ids)} Public Train Image Available\")","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:46:24.219913Z","iopub.execute_input":"2023-07-31T19:46:24.220341Z","iopub.status.idle":"2023-07-31T19:46:24.350628Z","shell.execute_reply.started":"2023-07-31T19:46:24.220312Z","shell.execute_reply":"2023-07-31T19:46:24.349502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hidden_test_ids = [path.stem for path in Path(TEST_DIR).glob(f\"*.{IMG_EXT}\")]\nprint(f\"{len(hidden_test_ids)} Public Test Image Available\")","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:46:24.352001Z","iopub.execute_input":"2023-07-31T19:46:24.353061Z","iopub.status.idle":"2023-07-31T19:46:24.361163Z","shell.execute_reply.started":"2023-07-31T19:46:24.353022Z","shell.execute_reply":"2023-07-31T19:46:24.360236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_names = [\"background\", \"blood_vessel\", \"glomerulus\", \"unsure\"]\nclass_color_dict = {\n    \"blood_vessel\": \"red\",\n    \"glomerulus\": \"green\",\n    \"unsure\": \"blue\",\n}","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:46:24.362545Z","iopub.execute_input":"2023-07-31T19:46:24.363162Z","iopub.status.idle":"2023-07-31T19:46:24.371948Z","shell.execute_reply.started":"2023-07-31T19:46:24.363131Z","shell.execute_reply":"2023-07-31T19:46:24.370639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wsi_df = pd.read_csv(WSI_TILE_CSV)\n# Tiles with Annotations\ndataset12_ids = wsi_df[wsi_df[\"dataset\"] != 3].id.values.tolist() \n# Tiles with Expert Reviewed Annotations\ndataset1_ids = wsi_df[wsi_df[\"dataset\"] == 1].id.values.tolist()\n# Tiles with Sparse Annotations\ndataset2_ids = wsi_df[wsi_df[\"dataset\"] == 2].id.values.tolist()\n# Tiles without Annotations\ndataset3_ids = wsi_df[wsi_df[\"dataset\"] == 3].id.values.tolist() ","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:46:24.373593Z","iopub.execute_input":"2023-07-31T19:46:24.373976Z","iopub.status.idle":"2023-07-31T19:46:24.419843Z","shell.execute_reply.started":"2023-07-31T19:46:24.373948Z","shell.execute_reply":"2023-07-31T19:46:24.418651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load and clean annotations","metadata":{}},{"cell_type":"code","source":"def load_annotations(annotations_path=ANNOT_PATH):\n    \n    annotations = {}\n    duplicates_ids = []\n\n    with open(annotations_path, \"r\") as f:\n        for line in f:\n            entry = json.loads(line)\n            identifier = entry[\"id\"]\n            annotations_list = entry[\"annotations\"]\n\n            revised_annotations = [] # [[label_str, polygon_list], ...]\n\n            for annotation in annotations_list:\n                label = annotation[\"type\"]\n                polygon = annotation[\"coordinates\"][0]\n\n                # Remove Duplicates\n                if not any(ShapelyContour(p).equals(ShapelyContour(polygon)) \\\n                           and l==label for l, p in revised_annotations):\n                    revised_annotations.append([label, polygon])\n\n            if len(annotations_list) != len(revised_annotations):\n                duplicates_ids.append(identifier)\n\n            annotations[identifier] = revised_annotations\n            \n    return annotations, duplicates_ids","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:46:24.423801Z","iopub.execute_input":"2023-07-31T19:46:24.424890Z","iopub.status.idle":"2023-07-31T19:46:24.434122Z","shell.execute_reply.started":"2023-07-31T19:46:24.424845Z","shell.execute_reply":"2023-07-31T19:46:24.432853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"annotations, duplicates_ids = load_annotations()\n\n# Sanity check\nassert len(annotations) == len(dataset12_ids)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:46:24.437813Z","iopub.execute_input":"2023-07-31T19:46:24.438952Z","iopub.status.idle":"2023-07-31T19:46:36.763398Z","shell.execute_reply.started":"2023-07-31T19:46:24.438906Z","shell.execute_reply":"2023-07-31T19:46:36.761866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_tile(identifier, annotations, color_dict,\n              properties_list = ['area', 'eccentricity']):\n    \n    path = Path(TRAIN_DIR) / f\"{identifier}.{IMG_EXT}\"\n    img = PIL.Image.open(path).convert(\"RGB\")\n    \n    fig = px.imshow(img, binary_string=True)\n    fig.update_traces(hoverinfo='skip')\n    fig.update_layout(width=IMG_SIZE, height=IMG_SIZE, \n                      margin=dict(l=0,r=0,b=0,t=0))\n    fig.update_xaxes(showticklabels=False)\n    fig.update_yaxes(showticklabels=False)\n    \n    if identifier in annotations.keys():\n        for label_id, (label, polygon) in enumerate(annotations[identifier]):\n\n            mask = skimage.draw.polygon2mask((IMG_SIZE, IMG_SIZE), polygon)\n            properties = skimage.measure.regionprops(mask.astype(np.uint8))[0]\n\n            hoverinfo = f'<b>label: {label}</b><br>' \n            for property_name in properties_list:\n                hoverinfo += f'<b>{property_name}: {properties[property_name]:.2f}</b><br>' \n\n            x, y = zip(*polygon)\n            fig.add_trace(\n                go.Scatter(x=x, y=y, name=label_id, mode='lines', fill='toself', \n                           showlegend=False, hovertemplate=hoverinfo, hoveron='points+fills',\n                           line={\"color\": color_dict[label]}, opacity=0.5)\n            )    \n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:46:36.765166Z","iopub.execute_input":"2023-07-31T19:46:36.765647Z","iopub.status.idle":"2023-07-31T19:46:36.777829Z","shell.execute_reply.started":"2023-07-31T19:46:36.765601Z","shell.execute_reply":"2023-07-31T19:46:36.776568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_tile(np.random.choice(dataset1_ids), annotations, class_color_dict)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:46:36.779297Z","iopub.execute_input":"2023-07-31T19:46:36.779678Z","iopub.status.idle":"2023-07-31T19:46:39.432671Z","shell.execute_reply.started":"2023-07-31T19:46:36.779640Z","shell.execute_reply":"2023-07-31T19:46:39.431734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_tiles(identifiers, annotations=None, color_dict=None, \n               nrows=5, ncols=6, divide_by=4 ,\n               properties_list = ['area', 'eccentricity']):\n    \n    fig = make_subplots(rows=nrows, cols=ncols,\n                        shared_xaxes=True, shared_yaxes=True, \n                        horizontal_spacing=0, vertical_spacing=0,\n                        subplot_titles=identifiers[:nrows * ncols])\n    fig.update_layout(width=IMG_SIZE*ncols/divide_by,\n                      height=IMG_SIZE*nrows/divide_by,\n                      margin=dict(l=0,r=0,b=0,t=0))\n    fig.update_xaxes(showticklabels=False)\n    fig.update_yaxes(showticklabels=False)\n    \n    for i in range(nrows * ncols):\n        \n        identifier = identifiers[i]\n        path = Path(TRAIN_DIR) / f\"{identifier}.{IMG_EXT}\"\n        img = PIL.Image.open(path).convert(\"RGB\")\n\n        fig_bgd = px.imshow(img, binary_string=True)\n        fig_bgd.update_traces(hoverinfo='skip')\n\n        fig.add_trace(fig_bgd.data[0], row=1+i//ncols, col=1+i%ncols) \n        \n        if annotations is not None and identifier in annotations.keys():\n            for label_id, (label, polygon) in enumerate(annotations[identifier]):\n\n                mask = skimage.draw.polygon2mask((IMG_SIZE, IMG_SIZE), polygon)\n                properties = skimage.measure.regionprops(mask.astype(np.uint8))[0]\n\n                hoverinfo = f'<b>label: {label}</b><br>' \n                for property_name in properties_list:\n                    hoverinfo += f'<b>{property_name}: {properties[property_name]:.2f}</b><br>' \n\n                x, y = zip(*polygon)\n                fig.add_trace(\n                    go.Scatter(x=x, y=y, name=label_id, mode='lines', fill='toself', \n                               showlegend=False, hovertemplate=hoverinfo, hoveron='points+fills',\n                               line={\"color\": color_dict[label]}, opacity=0.5),\n                    row=1+i//ncols, col=1+i%ncols\n                )\n\n    fig.update_annotations(yshift=-40, bgcolor ='rgba(255,255,255,0.75)')\n\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:46:39.433578Z","iopub.execute_input":"2023-07-31T19:46:39.433865Z","iopub.status.idle":"2023-07-31T19:46:39.447691Z","shell.execute_reply.started":"2023-07-31T19:46:39.433840Z","shell.execute_reply":"2023-07-31T19:46:39.446439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_annotations(annotations):\n    new_annotations = {}\n    anomalies_ids = []\n\n    for identifier in annotations.keys():\n\n        annotations_list = annotations[identifier]\n        revised_annotations = []\n        skip = False\n\n        for (label, polygon) in annotations_list:\n\n            for i, (l, p) in enumerate(revised_annotations):\n                # New Polygon Inside Old Polygon -> Skip New\n                if ShapelyPolygon(p).buffer(0).contains(ShapelyPolygon(polygon).buffer(0)):\n                    skip = True\n                    break\n                # Old Polygon Inside New Polygon -> Delete Old\n                if ShapelyPolygon(p).buffer(0).within(ShapelyPolygon(polygon).buffer(0)):\n                    revised_annotations.pop(i)\n            if skip:\n                continue\n            revised_annotations.append([label, polygon])\n\n        if len(annotations_list) != len(revised_annotations):\n            anomalies_ids.append(identifier)\n\n        new_annotations[identifier] = revised_annotations\n\n    return new_annotations, anomalies_ids ","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:46:39.454082Z","iopub.execute_input":"2023-07-31T19:46:39.454585Z","iopub.status.idle":"2023-07-31T19:46:39.470171Z","shell.execute_reply.started":"2023-07-31T19:46:39.454526Z","shell.execute_reply":"2023-07-31T19:46:39.468879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_annotations, anomalies_ids = clean_annotations(annotations)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:46:39.471239Z","iopub.execute_input":"2023-07-31T19:46:39.472275Z","iopub.status.idle":"2023-07-31T19:47:29.867629Z","shell.execute_reply.started":"2023-07-31T19:46:39.472137Z","shell.execute_reply":"2023-07-31T19:47:29.866526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_tiles(anomalies_ids, annotations, class_color_dict, nrows=2, ncols=3, divide_by=2)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:47:29.868846Z","iopub.execute_input":"2023-07-31T19:47:29.869249Z","iopub.status.idle":"2023-07-31T19:47:33.900031Z","shell.execute_reply.started":"2023-07-31T19:47:29.869222Z","shell.execute_reply":"2023-07-31T19:47:33.898674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_tiles(anomalies_ids, new_annotations, class_color_dict, nrows=2, ncols=3, divide_by=2)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:47:33.901347Z","iopub.execute_input":"2023-07-31T19:47:33.901671Z","iopub.status.idle":"2023-07-31T19:47:37.744629Z","shell.execute_reply.started":"2023-07-31T19:47:33.901643Z","shell.execute_reply":"2023-07-31T19:47:37.742466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Some images are mislabeled. When multiple annotations overlap, we keep the larger one. For instance, when a `blood_vessel` annotation is present inside of a `glomerulus` annotation, we keep the latter. This is consistent with the evaluation rules. ","metadata":{}},{"cell_type":"code","source":"annotations = new_annotations","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:47:37.746202Z","iopub.execute_input":"2023-07-31T19:47:37.746511Z","iopub.status.idle":"2023-07-31T19:47:37.754696Z","shell.execute_reply.started":"2023-07-31T19:47:37.746484Z","shell.execute_reply":"2023-07-31T19:47:37.753617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_grid(identifiers, annotations=None, color_dict=None,\n              nrows=5, ncols=6, divide_by=4, dpi=100):\n    figsize = (((IMG_SIZE/divide_by)*ncols)/dpi, ((IMG_SIZE/divide_by)*nrows)/dpi)\n    fig, axs = plt.subplots(nrows, ncols, figsize=figsize, dpi=dpi)\n    fig.subplots_adjust(left=0, bottom=0, right=1, top=1, wspace=0, hspace=0)\n    axs = axs.flatten()\n    for i, ax in enumerate(axs):\n        identifier = identifiers[i]\n        path = Path(TRAIN_DIR) / f\"{identifier}.{IMG_EXT}\"\n        img = PIL.Image.open(path).convert(\"RGB\")\n        ax.imshow(img)\n        if annotations is not None and identifier in annotations.keys():\n            for label, polygon in annotations[identifier]:        \n                xx, yy = zip(*polygon)\n                ax.fill(xx, yy, facecolor=color_dict[label], alpha=0.5)\n        ax.set_xticks([])\n        ax.set_yticks([])\n        ax.set_aspect(\"equal\")\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:47:37.756196Z","iopub.execute_input":"2023-07-31T19:47:37.756575Z","iopub.status.idle":"2023-07-31T19:47:37.770797Z","shell.execute_reply.started":"2023-07-31T19:47:37.756527Z","shell.execute_reply":"2023-07-31T19:47:37.769336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_grid(dataset1_ids, annotations, class_color_dict)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:47:37.772172Z","iopub.execute_input":"2023-07-31T19:47:37.772635Z","iopub.status.idle":"2023-07-31T19:47:44.517651Z","shell.execute_reply.started":"2023-07-31T19:47:37.772604Z","shell.execute_reply":"2023-07-31T19:47:44.516484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_grid(dataset2_ids, annotations, class_color_dict)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:47:44.519039Z","iopub.execute_input":"2023-07-31T19:47:44.519337Z","iopub.status.idle":"2023-07-31T19:47:51.097320Z","shell.execute_reply.started":"2023-07-31T19:47:44.519311Z","shell.execute_reply":"2023-07-31T19:47:51.095981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Instances from the `blood_vessel` class have high variability. Some are small and circular, others are large and elongated. This is expected since these are actually three types of structures: arterioles, capillaries, and venules. Furthermore, their appearence depends on the direction the tissue was sliced.   \n> Images from `Dataset 1` have more `unsure` annotations. This is consistent with the fact that they have been expert reviewed. Images from `Dataset 2` seem to have more `glomerulus` annotations.   \n> Other anatomical structures are present but are not annotated.  ","metadata":{}},{"cell_type":"code","source":"show_grid(dataset3_ids)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:47:51.099036Z","iopub.execute_input":"2023-07-31T19:47:51.099449Z","iopub.status.idle":"2023-07-31T19:47:55.057165Z","shell.execute_reply.started":"2023-07-31T19:47:51.099413Z","shell.execute_reply":"2023-07-31T19:47:55.055880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> We can see that images from `Dataset 3` are similar in content to images from `Dataset 1`  and `Dataset 2` and therefore could be useful in a semi-supervised learning appraoch.  \n> Some images from `Dataset 3` are pinkish in hue. In order to have a robust model, we should incorporate color-based augmentation methods.","metadata":{}},{"cell_type":"markdown","source":"##  Stats from annotations","metadata":{}},{"cell_type":"code","source":"def get_polygons(identifiers, annotations, class_names):\n    polygons = []\n    for identifier in identifiers:\n        polygons_list = []\n        for (label, polygon) in annotations[identifier]:\n            label_int = class_names.index(label)\n            polygons_list.append([label_int, polygon])\n        polygons.append(polygons_list)\n    return polygons","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:47:55.058537Z","iopub.execute_input":"2023-07-31T19:47:55.059365Z","iopub.status.idle":"2023-07-31T19:47:55.065395Z","shell.execute_reply.started":"2023-07-31T19:47:55.059330Z","shell.execute_reply":"2023-07-31T19:47:55.064175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def count_labels(polygons, class_names, normalize=False):\n    counts = [0] * (len(class_names) - 1)\n    for tile_annotation in polygons:\n        for label, _ in tile_annotation:\n            counts[label - 1] += 1\n    if normalize:\n        sum_counts = sum(counts)\n        counts = [c / sum_counts for c in counts]\n    return counts","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:47:55.066844Z","iopub.execute_input":"2023-07-31T19:47:55.067242Z","iopub.status.idle":"2023-07-31T19:47:55.080044Z","shell.execute_reply.started":"2023-07-31T19:47:55.067201Z","shell.execute_reply":"2023-07-31T19:47:55.078835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_annotations = []\nfor identifier in annotations.keys():\n    polygons = get_polygons([identifier], annotations, class_names)\n    num_annotations.append(count_labels(polygons, class_names)) \nnum_annotations = np.array(num_annotations)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:47:55.081984Z","iopub.execute_input":"2023-07-31T19:47:55.082695Z","iopub.status.idle":"2023-07-31T19:47:55.109262Z","shell.execute_reply.started":"2023-07-31T19:47:55.082661Z","shell.execute_reply":"2023-07-31T19:47:55.108013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Average Number of Annotations per Image:\", num_annotations.sum(axis=1).mean())\nprint(\"Average Number of Blood Vessels per Image:\", num_annotations[:, 0].mean())\nprint(\"Average Number of Glomeruli per Image:\", num_annotations[:, 1].mean())\nprint(\"Average Number of Unsure Annotations per Image:\", num_annotations[:, 2].mean())","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:47:55.111142Z","iopub.execute_input":"2023-07-31T19:47:55.111490Z","iopub.status.idle":"2023-07-31T19:47:55.123391Z","shell.execute_reply.started":"2023-07-31T19:47:55.111461Z","shell.execute_reply":"2023-07-31T19:47:55.122415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"polygons1 = get_polygons(dataset1_ids, annotations, class_names)\ncount_labels(polygons1, class_names)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:47:55.124736Z","iopub.execute_input":"2023-07-31T19:47:55.125672Z","iopub.status.idle":"2023-07-31T19:47:55.147140Z","shell.execute_reply.started":"2023-07-31T19:47:55.125637Z","shell.execute_reply":"2023-07-31T19:47:55.145884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"polygons2 = get_polygons(dataset2_ids, annotations, class_names)\ncount_labels(polygons2, class_names)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:47:55.148826Z","iopub.execute_input":"2023-07-31T19:47:55.149158Z","iopub.status.idle":"2023-07-31T19:47:55.175756Z","shell.execute_reply.started":"2023-07-31T19:47:55.149131Z","shell.execute_reply":"2023-07-31T19:47:55.174455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Transform Polygon Annotations into Masks ","metadata":{}},{"cell_type":"code","source":"def prepare_gt_semantic_seg(identifiers, annotations, class_names, erode=0, binary=False):\n    \"\"\"\n    Transform polygon annotations into masks\n    \"\"\"        \n    mask_shape = (IMG_SIZE, IMG_SIZE)\n    masks, polygons = [], []\n        \n    for identifier in identifiers:\n        \n        labeled_mask = np.zeros(mask_shape, dtype=\"uint8\")\n        overlap = np.zeros(mask_shape, dtype=\"uint8\")\n        polygons_list = [] # [[label_int, polygon_list], ...]\n\n        for (label, polygon) in annotations[identifier]:\n            label_int = class_names.index(label)\n            polygons_list.append([label_int, polygon])\n            mask = skimage.draw.polygon2mask(mask_shape, polygon).T\n            if erode:\n                mask = skimage.morphology.erosion(mask, skimage.morphology.square(7))\n                \n            intersection = (labeled_mask * mask) > 0\n            \n            if intersection.sum():\n                overlap += intersection\n            \n            labeled_mask += (label_int * mask).astype(\"uint8\")\n         \n        # Overlap is treated as background\n        labeled_mask[overlap > 0] = 0\n        \n        masks.append(labeled_mask)\n        polygons.append(polygons_list)\n    \n    masks = np.array(masks) # (N, H, W)\n    \n    if binary:\n        # One-Hot Encode \n        masks = np.eye(len(class_names), dtype=np.uint8)[masks] \n        masks = masks[:,:,:,1:] # (N, H, W, C)\n    \n    return masks, polygons","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:47:55.176991Z","iopub.execute_input":"2023-07-31T19:47:55.177344Z","iopub.status.idle":"2023-07-31T19:47:55.187582Z","shell.execute_reply.started":"2023-07-31T19:47:55.177316Z","shell.execute_reply":"2023-07-31T19:47:55.186698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"masks, polygons = prepare_gt_semantic_seg(dataset1_ids, annotations, class_names)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:47:55.189031Z","iopub.execute_input":"2023-07-31T19:47:55.190040Z","iopub.status.idle":"2023-07-31T19:48:51.978102Z","shell.execute_reply.started":"2023-07-31T19:47:55.190006Z","shell.execute_reply":"2023-07-31T19:48:51.977013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_grid_seperate(identifiers, masks, polygons, class_names,\n                       nsamples=4, divide_by=2, transpose=False, dpi=100):\n    if transpose:\n        nrows, ncols = len(class_names) - 1, nsamples\n    else:\n        nrows, ncols = nsamples, len(class_names) - 1\n        \n    figsize = ((IMG_SIZE/divide_by)*ncols)/dpi, ((IMG_SIZE/divide_by)*nrows)/dpi\n    fig, axs = plt.subplots(nrows, ncols, figsize=figsize, dpi=dpi)\n    \n    for i in range(nsamples):\n        \n        identifier = identifiers[i]\n        path = Path(TRAIN_DIR) / f\"{identifier}.{IMG_EXT}\"\n        img = PIL.Image.open(path).convert(\"RGB\")\n        mask = masks[i]\n        polygons_list = polygons[i]\n        \n        for j in range(1, len(class_names)):\n            \n            ax = axs[j-1, i] if transpose else axs[i, j-1] \n            ax.imshow(img)\n            \n            # Multiclass Masks\n            if mask.ndim == 2:\n                ax.imshow((mask==j)*j, alpha=0.25, cmap=\"turbo\", \n                          vmin=0, vmax=len(class_names),\n                          interpolation=\"nearest\")\n            # Binary Masks\n            elif mask.ndim == 3:\n                ax.imshow(mask[:,:,j-1]*j, alpha=0.25, cmap=\"turbo\", \n                          vmin=0, vmax=len(class_names), \n                          interpolation=\"nearest\")\n            ax.set_xticks([])\n            ax.set_yticks([])\n            ax.set_aspect(\"equal\")\n            \n            for label, polygon in polygons_list:\n                xx, yy = zip(*polygon)\n                if label == j:\n                    ax.fill(xx, yy, facecolor='none', edgecolor=\"orange\", linewidth=1)\n                else:\n                    ax.fill(xx, yy, facecolor='none', edgecolor=\"orange\", linewidth=1, linestyle=\"--\")\n    fig.subplots_adjust(left=0, bottom=0, right=1, top=1, wspace=0, hspace=0)\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:48:51.979774Z","iopub.execute_input":"2023-07-31T19:48:51.980389Z","iopub.status.idle":"2023-07-31T19:48:51.993542Z","shell.execute_reply.started":"2023-07-31T19:48:51.980348Z","shell.execute_reply":"2023-07-31T19:48:51.991966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_grid_seperate(dataset1_ids, masks, polygons, class_names, 5)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:48:51.995429Z","iopub.execute_input":"2023-07-31T19:48:51.995888Z","iopub.status.idle":"2023-07-31T19:48:54.993200Z","shell.execute_reply.started":"2023-07-31T19:48:51.995845Z","shell.execute_reply":"2023-07-31T19:48:54.992135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_grid_seperate(dataset1_ids, masks, polygons, class_names, 15, 8, transpose=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:48:54.994549Z","iopub.execute_input":"2023-07-31T19:48:54.995395Z","iopub.status.idle":"2023-07-31T19:49:03.300570Z","shell.execute_reply.started":"2023-07-31T19:48:54.995356Z","shell.execute_reply":"2023-07-31T19:49:03.299607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inspect Sources","metadata":{}},{"cell_type":"code","source":"wsi_df = pd.read_csv(WSI_TILE_CSV)\n# Consider only annotated tiles\nwsi_df = wsi_df[wsi_df[\"dataset\"] != 3] ","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:49:03.301751Z","iopub.execute_input":"2023-07-31T19:49:03.302129Z","iopub.status.idle":"2023-07-31T19:49:03.320936Z","shell.execute_reply.started":"2023-07-31T19:49:03.302104Z","shell.execute_reply":"2023-07-31T19:49:03.319899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wsi_df[\"source\"] = wsi_df[[\"dataset\", \"source_wsi\"]].apply(tuple, axis=1)\nsource_combinations = [(1, 1), (1, 2), (2, 1), (2, 2), (2, 3), (2, 4)]\nassert set(source_combinations) == set(wsi_df[\"source\"].values)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:49:03.322590Z","iopub.execute_input":"2023-07-31T19:49:03.323007Z","iopub.status.idle":"2023-07-31T19:49:03.345022Z","shell.execute_reply.started":"2023-07-31T19:49:03.322969Z","shell.execute_reply":"2023-07-31T19:49:03.344199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(nrows=3, ncols=2, sharex=True, sharey=True, figsize=[6, 8])\naxs = axs.flatten()\naxs[0].invert_yaxis()\nfor ax, source in zip(axs, source_combinations):\n    wsi_df[wsi_df[\"source\"]==source].plot.scatter(x=\"i\", y=\"j\", marker=\"+\",\n                                                  ax=ax, title=f\"Dataset {source[0]} WSI {source[1]}\")\n    ax.set(aspect='equal')\n    ax.tick_params(axis=\"x\", labelrotation=90)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:49:03.346189Z","iopub.execute_input":"2023-07-31T19:49:03.346699Z","iopub.status.idle":"2023-07-31T19:49:04.871681Z","shell.execute_reply.started":"2023-07-31T19:49:03.346670Z","shell.execute_reply":"2023-07-31T19:49:04.870609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wsi_ids = set(wsi_df[\"source_wsi\"].values)\nfig, axs = plt.subplots(nrows=2, ncols=2, sharex=True, sharey=True, figsize=[6, 8])\naxs = axs.flatten()\naxs[0].invert_yaxis()\nfor wsi_id, ax in enumerate(axs, 1):\n    wsi_df[wsi_df[\"source_wsi\"]==wsi_id].plot.scatter(x=\"i\", y=\"j\", marker=\"+\",\n                                                      c=\"dataset\", cmap=\"viridis\",\n                                                      vmin=1, vmax=2,\n                                                      ax=ax, title=f\"WSI{wsi_id}\")\n    ax.set(aspect='equal')\nfor cax in fig.get_axes()[len(fig.get_axes())//2:]:\n    cax.remove()","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:49:04.872969Z","iopub.execute_input":"2023-07-31T19:49:04.873301Z","iopub.status.idle":"2023-07-31T19:49:06.006714Z","shell.execute_reply.started":"2023-07-31T19:49:04.873273Z","shell.execute_reply":"2023-07-31T19:49:06.005625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> For testing the performance of our model, we should use tiles from `Dataset 1` because the annotations were expert reviewed. We should also make sure to have instances from both `WSI 1` and `WSI 2`.","metadata":{}},{"cell_type":"markdown","source":"## Splitting the data","metadata":{}},{"cell_type":"code","source":"def shuffle_sequence(seq):\n    array = np.array(seq)\n    np.random.shuffle(array)\n    return array.tolist()","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:49:06.008287Z","iopub.execute_input":"2023-07-31T19:49:06.008860Z","iopub.status.idle":"2023-07-31T19:49:06.014164Z","shell.execute_reply.started":"2023-07-31T19:49:06.008828Z","shell.execute_reply":"2023-07-31T19:49:06.013126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def retrieve_identifiers(wsi_df, source, shuffle=True):\n    array = wsi_df[wsi_df[\"source\"]==source].id.values.tolist()\n    if shuffle:    \n        return shuffle_sequence(array)\n    return array","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:49:06.015931Z","iopub.execute_input":"2023-07-31T19:49:06.016864Z","iopub.status.idle":"2023-07-31T19:49:06.027208Z","shell.execute_reply.started":"2023-07-31T19:49:06.016823Z","shell.execute_reply":"2023-07-31T19:49:06.026040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"source11 = retrieve_identifiers(wsi_df, (1, 1))\nsource12 = retrieve_identifiers(wsi_df, (1, 2))\nsource21 = retrieve_identifiers(wsi_df, (2, 1))\nsource22 = retrieve_identifiers(wsi_df, (2, 2))\nsource23 = retrieve_identifiers(wsi_df, (2, 3))\nsource24 = retrieve_identifiers(wsi_df, (2, 4))\n\n\nvalid_ids = source11[:50] + source12[:50]\ntrain_ids = source11[50:] + source12[50:] + source21 + source22 + source23 + source24","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:49:06.028463Z","iopub.execute_input":"2023-07-31T19:49:06.028924Z","iopub.status.idle":"2023-07-31T19:49:06.046690Z","shell.execute_reply.started":"2023-07-31T19:49:06.028886Z","shell.execute_reply":"2023-07-31T19:49:06.045619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_masks, train_polygons = prepare_gt_semantic_seg(train_ids, annotations, class_names)\nvalid_masks, valid_polygons = prepare_gt_semantic_seg(valid_ids, annotations, class_names)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:49:06.053366Z","iopub.execute_input":"2023-07-31T19:49:06.053747Z","iopub.status.idle":"2023-07-31T19:52:28.791071Z","shell.execute_reply.started":"2023-07-31T19:49:06.053718Z","shell.execute_reply":"2023-07-31T19:52:28.790206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count_labels(train_polygons, class_names)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:52:28.792601Z","iopub.execute_input":"2023-07-31T19:52:28.793364Z","iopub.status.idle":"2023-07-31T19:52:28.804927Z","shell.execute_reply.started":"2023-07-31T19:52:28.793330Z","shell.execute_reply":"2023-07-31T19:52:28.803617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count_labels(valid_polygons, class_names)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:52:28.806022Z","iopub.execute_input":"2023-07-31T19:52:28.806308Z","iopub.status.idle":"2023-07-31T19:52:28.820628Z","shell.execute_reply.started":"2023-07-31T19:52:28.806285Z","shell.execute_reply":"2023-07-31T19:52:28.819455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_grid_seperate(train_ids, train_masks, train_polygons, class_names, 15, 8, transpose=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:52:28.822011Z","iopub.execute_input":"2023-07-31T19:52:28.822299Z","iopub.status.idle":"2023-07-31T19:52:36.628452Z","shell.execute_reply.started":"2023-07-31T19:52:28.822275Z","shell.execute_reply":"2023-07-31T19:52:36.627631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_grid_seperate(valid_ids, valid_masks, valid_polygons, class_names, 15, 8, transpose=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:52:36.629685Z","iopub.execute_input":"2023-07-31T19:52:36.630572Z","iopub.status.idle":"2023-07-31T19:52:44.168787Z","shell.execute_reply.started":"2023-07-31T19:52:36.630506Z","shell.execute_reply":"2023-07-31T19:52:44.167711Z"},"trusted":true},"execution_count":null,"outputs":[]}]}