{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nimport os\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport pandas as pd\n\nimport matplotlib.pyplot as plt\nfrom matplotlib.patches import Polygon, Patch\nfrom matplotlib.collections import PatchCollection\nimport seaborn as sns\n\nfrom PIL import Image\n\nfrom IPython.display import display","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:15:11.549560Z","iopub.execute_input":"2023-06-20T11:15:11.549959Z","iopub.status.idle":"2023-06-20T11:15:12.232370Z","shell.execute_reply.started":"2023-06-20T11:15:11.549928Z","shell.execute_reply":"2023-06-20T11:15:12.231446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. Data\nThe data is distributed in the folder test and train where we have the images","metadata":{}},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/hubmap-hacking-the-human-vasculature\"\nfor dirname, _, filenames in os.walk(DATA_DIR):\n    if dirname.find(\"ipynb\")>-1:\n        continue\n    print(f\"Directory: {dirname}\")\n    for filename in filenames[:10]:\n        print(f\"\\t* {filename}\")","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:15:18.502510Z","iopub.execute_input":"2023-06-20T11:15:18.502881Z","iopub.status.idle":"2023-06-20T11:15:19.443136Z","shell.execute_reply.started":"2023-06-20T11:15:18.502849Z","shell.execute_reply":"2023-06-20T11:15:19.442083Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"polygons_df = pd.read_json(os.path.join(DATA_DIR,\"polygons.jsonl\"), lines=True)\ndisplay(polygons_df.head(10))\npolygons_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:15:19.445161Z","iopub.execute_input":"2023-06-20T11:15:19.445512Z","iopub.status.idle":"2023-06-20T11:15:23.951128Z","shell.execute_reply.started":"2023-06-20T11:15:19.445484Z","shell.execute_reply":"2023-06-20T11:15:23.950073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.1 Distribution of data","metadata":{"execution":{"iopub.status.busy":"2023-06-12T22:28:58.772616Z","iopub.execute_input":"2023-06-12T22:28:58.773017Z","iopub.status.idle":"2023-06-12T22:28:58.778496Z","shell.execute_reply.started":"2023-06-12T22:28:58.772987Z","shell.execute_reply":"2023-06-12T22:28:58.777311Z"}}},{"cell_type":"code","source":"annotations_list = polygons_df['annotations'].explode()\n\n# Extract the 'type' field from each dictionary\ntypes = annotations_list.apply(lambda x: x['type'])\n\n# Count the occurrences of each 'type'\ntype_counts = types.value_counts()\nplt.pie(type_counts, labels=type_counts.index, autopct=lambda pct: f'{pct:.1f}% ({int(pct * sum(type_counts)/100)})')\n\n# Add a title\nplt.title(\"annotation Counts\")\n\n# Display the chart\nplt.show()\nplt.close()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:15:23.952405Z","iopub.execute_input":"2023-06-20T11:15:23.952942Z","iopub.status.idle":"2023-06-20T11:15:24.133254Z","shell.execute_reply.started":"2023-06-20T11:15:23.952914Z","shell.execute_reply":"2023-06-20T11:15:24.132176Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets look at the file `tile_meta.csv`. It has columns `id`, `source_wsi`, `dataset`, `i`, `j`\n\n* source_wsi Identifies the WSI this tile was extracted from.\n* {i|j} The location of the upper-left corner within the WSI where the tile was extracted.\n* dataset The dataset this tile belongs to, as described above.","metadata":{}},{"cell_type":"code","source":"tile_df = pd.read_csv(os.path.join(DATA_DIR, \"tile_meta.csv\"))\ndisplay(tile_df.head(10))\nprint(tile_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:15:24.134917Z","iopub.execute_input":"2023-06-20T11:15:24.135606Z","iopub.status.idle":"2023-06-20T11:15:24.168910Z","shell.execute_reply.started":"2023-06-20T11:15:24.135566Z","shell.execute_reply":"2023-06-20T11:15:24.167904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(5,5))\ndataset_counts = tile_df.dataset.value_counts()\n# plt.pie(dataset_counts,labels=dataset_counts.index, autopct='%1.1f%%')\nplt.pie(dataset_counts,labels=dataset_counts.index, autopct=lambda pct: f'{pct:.1f}% ({int(pct * sum(dataset_counts)/100)})')\n# Additional customization (optional)\nplt.axis('equal')  # Equal aspect ratio ensures a circular pie\nplt.title('Counts images from the Dataset')\n\n# Show the pie chart\nplt.show()\nplt.close()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:15:24.170516Z","iopub.execute_input":"2023-06-20T11:15:24.171190Z","iopub.status.idle":"2023-06-20T11:15:24.316204Z","shell.execute_reply.started":"2023-06-20T11:15:24.171152Z","shell.execute_reply":"2023-06-20T11:15:24.315084Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets also look at `wsi_meta.csv`.  wsi_meta.csv Metadata for the Whole Slide Images the tiles were extracted from.\n* `source_wsi` Identifies the WSI.\n* `age`, `sex`, `race`, `height`, `weight`, and `bmi` demographic information about the tissue donor.","metadata":{}},{"cell_type":"code","source":"wsi_df = pd.read_csv(os.path.join(DATA_DIR, \"wsi_meta.csv\"))\ndisplay(wsi_df)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:15:24.317878Z","iopub.execute_input":"2023-06-20T11:15:24.318548Z","iopub.status.idle":"2023-06-20T11:15:24.342659Z","shell.execute_reply.started":"2023-06-20T11:15:24.318509Z","shell.execute_reply":"2023-06-20T11:15:24.341302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Annotated vs Non Annotated data\nassert polygons_df.id.isin(tile_df.id).all()\ntotal_images_count = len(tile_df.id)\nannotated_images_count = len(polygons_df.id)\nnot_annotated_images_count = total_images_count - annotated_images_count\n\nannotation_counts_df = pd.DataFrame({\n    \"annotated\": [annotated_images_count],\n    \"not_annotated\": [not_annotated_images_count]\n})\n\nprint(f\"Not all the images provided are annotated. Here are the counts for total images {total_images_count} we have only {annotated_images_count} annotated and {not_annotated_images_count} are not annotated. it would be interesting to see the metadata and see how many of the datasets are annotated.\")\ndisplay(annotation_counts_df)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:15:24.346390Z","iopub.execute_input":"2023-06-20T11:15:24.346824Z","iopub.status.idle":"2023-06-20T11:15:24.366396Z","shell.execute_reply.started":"2023-06-20T11:15:24.346780Z","shell.execute_reply":"2023-06-20T11:15:24.364615Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(5,5))\ndataset_counts = tile_df[tile_df.id.isin(polygons_df.id)].dataset.value_counts()\n# plt.pie(dataset_counts,labels=dataset_counts.index, autopct='%1.1f%%')\nplt.pie(dataset_counts,labels=dataset_counts.index, autopct=lambda pct: f'{pct:.1f}% ({int(pct * sum(dataset_counts)/100)})')\n# Additional customization (optional)\nplt.axis('equal')  # Equal aspect ratio ensures a circular pie\nplt.title('Counts annotated images from the Dataset')\n\n# Show the pie chart\nplt.show()\nplt.close()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:15:24.368617Z","iopub.execute_input":"2023-06-20T11:15:24.369296Z","iopub.status.idle":"2023-06-20T11:15:24.500064Z","shell.execute_reply.started":"2023-06-20T11:15:24.369254Z","shell.execute_reply":"2023-06-20T11:15:24.499000Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax0, ax1) = plt.subplots(nrows=1, ncols=2, figsize=(10,5))\n\nsource_wsi_counts_1 = tile_df[tile_df.id.isin(polygons_df.id) & (tile_df.dataset==1)].source_wsi.value_counts()\nvalues = source_wsi_counts_1.values\nlabels = [f\"wsi src {x}\" for x in source_wsi_counts_1.index]\nax0.pie(values,labels=labels, autopct=lambda pct: f'{pct:.1f}% ({int(pct * sum(source_wsi_counts_1)/100)})')\nax0.set_title('Counts images from the wsi source in DATASET 1')\n\nsource_wsi_counts_2 = tile_df[tile_df.id.isin(polygons_df.id) & (tile_df.dataset==2)].source_wsi.value_counts()\nvalues = source_wsi_counts_2.values\nlabels = [f\"wsi src {x}\" for x in source_wsi_counts_2.index]\nax1.pie(values,labels=labels, autopct=lambda pct: f'{pct:.1f}% ({int(pct * sum(source_wsi_counts_1)/100)})')\nax1.set_title('Counts images from the wsi source in DATASET 2')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:39:06.993921Z","iopub.execute_input":"2023-06-20T11:39:06.994296Z","iopub.status.idle":"2023-06-20T11:39:07.268948Z","shell.execute_reply.started":"2023-06-20T11:39:06.994266Z","shell.execute_reply":"2023-06-20T11:39:07.267613Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.2 Visualization\nvisualizing a few images with annotation masked on them. Using Blue color highlighted parts for blood vessels. Green for glomerulus and Red for Unsure.","metadata":{}},{"cell_type":"code","source":"images_data = []\nrows = 5\ncolumns = 5\nfor index, row in polygons_df.head(rows*columns).iterrows():\n    annotations = row['annotations']\n    image_id = row[\"id\"]\n    image_data = {\n        \"image_path\": f'{DATA_DIR}/train/{image_id}.tif',\n        \"annotations\": []\n    }\n    \n    for annotation in annotations:\n        annotation_type = annotation['type']\n        coordinates = annotation['coordinates']\n        for segment in coordinates:\n            image_data[\"annotations\"].append({\n                \"image_id\": image_id,\n                \"type\": annotation_type,\n                \"segmentation\": segment\n            })\n    images_data.append(image_data)\n\nfig, axes = plt.subplots(nrows=rows, ncols=columns, figsize=(10,10))\n\ncolors = {\n    'blood_vessel': \"r\",\n    \"unsure\": \"b\",\n    \"glomerulus\": 'g'\n}\n\n# Iterate over the annotations\nfor i, image_data in enumerate(images_data):\n    # Extract the segmentation coordinates\n    row = i//rows\n    col = i % columns\n    \n    ax = axes[row, col]\n    \n    \n    image = Image.open(image_data['image_path'])\n    ax.imshow(image)\n    polygons = []\n    for annotation in image_data['annotations']:\n        segmentation = annotation['segmentation']\n        color = colors[annotation['type']]\n        polygon = Polygon(segmentation, closed=True, alpha=0.4, edgecolor=color, facecolor=color)\n        polygons.append(polygon)\n    \n    # Add all polygons to the axes at once using add_collection()\n    collection = PatchCollection(polygons, match_original=True)\n    ax.add_collection(collection)\n\n    # Set aspect ratio and axis off\n    ax.set_aspect('equal')\n    ax.axis('off')\n    \n\n# Adjust the spacing between subplots\nplt.tight_layout()\n\n# Add legend outside the grid\nhandles = [Patch(facecolor=color, edgecolor=color) for color in colors.values()]\nlabels = list(colors.keys())\nfig.legend(handles, labels, loc='upper right')\n    \n# Show the image with annotations\nplt.show()\nplt.close()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:15:24.743150Z","iopub.execute_input":"2023-06-20T11:15:24.743920Z","iopub.status.idle":"2023-06-20T11:15:27.450376Z","shell.execute_reply.started":"2023-06-20T11:15:24.743879Z","shell.execute_reply":"2023-06-20T11:15:27.449430Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}