{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# HuBMAP - Image and Masks Visualizations ","metadata":{}},{"cell_type":"code","source":"import os\nimport random\nimport tifffile\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom skimage import draw\n\n\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-26T11:14:04.494359Z","iopub.execute_input":"2023-05-26T11:14:04.494804Z","iopub.status.idle":"2023-05-26T11:14:05.089662Z","shell.execute_reply.started":"2023-05-26T11:14:04.494773Z","shell.execute_reply":"2023-05-26T11:14:05.088426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data directory\ndata_dir = \"/kaggle/input/hubmap-hacking-the-human-vasculature/\"","metadata":{"execution":{"iopub.status.busy":"2023-05-26T11:14:40.464318Z","iopub.execute_input":"2023-05-26T11:14:40.464782Z","iopub.status.idle":"2023-05-26T11:14:40.470803Z","shell.execute_reply.started":"2023-05-26T11:14:40.464749Z","shell.execute_reply":"2023-05-26T11:14:40.469295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read polygons json data\njsonl_path = os.path.join(data_dir, 'polygons.jsonl')\npolygon_data = pd.read_json(path_or_buf=jsonl_path, lines=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T11:15:29.357588Z","iopub.execute_input":"2023-05-26T11:15:29.358071Z","iopub.status.idle":"2023-05-26T11:15:34.216033Z","shell.execute_reply.started":"2023-05-26T11:15:29.358038Z","shell.execute_reply":"2023-05-26T11:15:34.214667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"polygon_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T11:15:38.783994Z","iopub.execute_input":"2023-05-26T11:15:38.785115Z","iopub.status.idle":"2023-05-26T11:15:38.845164Z","shell.execute_reply.started":"2023-05-26T11:15:38.785075Z","shell.execute_reply":"2023-05-26T11:15:38.844175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Details about this dataframe\nThe abouve dataframe represent training data. But for upcoming notebooks, I am going to split this data into train, test & validation dataset\n\nNow, \n\n**id**: image id, and image path is ``data_dir/train/<id>.jpg``\n\n**annotations**: it contains all the segmentaion data in the following format\n\n\n```\n[\n        {\n            'type':str,\n            'coordinates':[[\n                            [int,int]\n                            ]]\n        },\n        ...\n\n]\n ```","metadata":{}},{"cell_type":"code","source":"print(f\"TOTAL DATA POINTS: {len(polygon_data)}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-26T11:22:24.307215Z","iopub.execute_input":"2023-05-26T11:22:24.307698Z","iopub.status.idle":"2023-05-26T11:22:24.314595Z","shell.execute_reply.started":"2023-05-26T11:22:24.307669Z","shell.execute_reply":"2023-05-26T11:22:24.313339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### For our further analysis, I am flatten the abouve annotation data","metadata":{}},{"cell_type":"code","source":"# Flatten the list of dictionaries\ndf_flattened = polygon_data.explode('annotations')\ndf_flattened = pd.concat([df_flattened.drop(['annotations'], axis=1), df_flattened['annotations'].apply(pd.Series)], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T11:22:36.298680Z","iopub.execute_input":"2023-05-26T11:22:36.299111Z","iopub.status.idle":"2023-05-26T11:22:42.287756Z","shell.execute_reply.started":"2023-05-26T11:22:36.299082Z","shell.execute_reply":"2023-05-26T11:22:42.286357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_flattened.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T11:22:42.824196Z","iopub.execute_input":"2023-05-26T11:22:42.824667Z","iopub.status.idle":"2023-05-26T11:22:43.037906Z","shell.execute_reply.started":"2023-05-26T11:22:42.824633Z","shell.execute_reply":"2023-05-26T11:22:43.036513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# type distribution \ndf_flattened['type'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T11:23:36.937283Z","iopub.execute_input":"2023-05-26T11:23:36.937761Z","iopub.status.idle":"2023-05-26T11:23:36.955897Z","shell.execute_reply.started":"2023-05-26T11:23:36.937731Z","shell.execute_reply":"2023-05-26T11:23:36.954687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We can see a class imbalance in our data, which we will handle in upcominmg notebooks","metadata":{}},{"cell_type":"markdown","source":"# Now, let's visualize some sample image and masks","metadata":{}},{"cell_type":"code","source":"sample_id = random.choices(df_flattened['id'].tolist(),k=20)\nsample_data = df_flattened[df_flattened['id'].isin(sample_id)]","metadata":{"execution":{"iopub.status.busy":"2023-05-26T11:25:08.221910Z","iopub.execute_input":"2023-05-26T11:25:08.222398Z","iopub.status.idle":"2023-05-26T11:25:08.237413Z","shell.execute_reply.started":"2023-05-26T11:25:08.222363Z","shell.execute_reply":"2023-05-26T11:25:08.235877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a 3x2 subplot grid\nfig, axes = plt.subplots(5, 4, figsize=(20, 20))\n\n# Flatten the axes array to iterate over it\naxes = axes.flatten()\nfor idx, id in enumerate(sample_id):\n    image_path = os.path.join(data_dir,\"train\",f\"{id}.tif\")\n    image = tifffile.imread(image_path)\n    \n    tmp_df = sample_data[sample_data['id']==str(id)]\n    for row in tmp_df.iterrows():\n        cords = row[1]['coordinates']\n        \n        if row[1]['type']==\"blood_vessel\":\n            c=\"r\"\n        elif row[1]['type']==\"glomerulus\":\n            c=\"b\"\n        elif row[1]['type']==\"unsure\":\n            c=\"green\"\n        x, y = np.array([i[0] for i in cords[0]]), np.asarray([i[1] for i in cords[0]])\n        x_offset = x - np.min(x)\n        y_offset = np.max(y) - y\n        imageing = draw.polygon2mask(\n                (512, 512),\n                np.stack((y_offset, x_offset), axis=1)\n                )\n        \n        \n        axes[idx].scatter(x, y, s=0)\n        axes[idx].fill(x, y, c, alpha=0.5)\n    axes[idx].imshow(image, interpolation='nearest', cmap='gray')\n    axes[idx].axis('off')","metadata":{"execution":{"iopub.status.busy":"2023-05-26T11:25:38.486293Z","iopub.execute_input":"2023-05-26T11:25:38.487295Z","iopub.status.idle":"2023-05-26T11:25:50.428891Z","shell.execute_reply.started":"2023-05-26T11:25:38.487242Z","shell.execute_reply":"2023-05-26T11:25:50.427972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Don't forget to hit upvote icon 🔺**","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}