{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Data and masks vizualization of the HuBMAP data","metadata":{}},{"cell_type":"markdown","source":"### Goal\n\nLocate microvasculature structures (blood vessels) within human kidney histology slides.\n\n### Data Description\n\nTiles extracted from five Whole Slide Images (WSI) split into two datasets. Tiles from Dataset 1 have annotations that have been expert reviewed. Dataset 2 comprises the remaining tiles from these same WSIs and contain sparse annotations that have not been expert reviewed.\n\nAll of the test set tiles are from Dataset 1.\n\nTwo of the WSIs make up the training set, two WSIs make up the public test set, and one WSI makes up the private test set.\nThe training data includes Dataset 2 tiles from the public test WSI, but not from the private test WSI\n\n### Preparation\nGet into the directory where the 'test' and 'train' folders are stored. Run the notebook locally there.","metadata":{}},{"cell_type":"code","source":"!pip install pycocotools","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, glob\nfrom PIL import Image\nimport json\nimport glob\nimport cv2\nimport numpy.ma as ma\nimport matplotlib.pyplot as plt\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-06-12T10:52:01.641600Z","iopub.execute_input":"2023-06-12T10:52:01.643363Z","iopub.status.idle":"2023-06-12T10:52:01.851386Z","shell.execute_reply.started":"2023-06-12T10:52:01.643170Z","shell.execute_reply":"2023-06-12T10:52:01.849951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_PATH = \"/train/*\"\nTEST_PATH = \"/test/*\"","metadata":{"execution":{"iopub.status.busy":"2023-06-12T10:55:19.579508Z","iopub.execute_input":"2023-06-12T10:55:19.579947Z","iopub.status.idle":"2023-06-12T10:55:19.586517Z","shell.execute_reply.started":"2023-06-12T10:55:19.579914Z","shell.execute_reply":"2023-06-12T10:55:19.584905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = glob.glob(TRAIN_PATH)\ntest = glob.glob(TEST_PATH)\n\n# Create a dictionary to hold the annotations for each image\nannotations = {}\n\nwith open('polygons.jsonl', 'r') as polygons:\n    for line in polygons:\n        polygons_data = json.loads(line)\n        \n        # Get the image ID and the list of annotations for this image\n        image_id = polygons_data['id']\n        image_annotations = polygons_data['annotations']\n\n        # Store the annotations in the dictionary\n        annotations[image_id] = image_annotations\n        \n# Create our image map for our train data and test data\nimage_map = {impath.split('/')[-1].split('.')[0]: impath for impath in train}\nimage_map_test = {impath.split('/')[-1].split('.')[0]: impath for impath in test}","metadata":{"execution":{"iopub.status.busy":"2023-06-12T10:55:22.061263Z","iopub.execute_input":"2023-06-12T10:55:22.061711Z","iopub.status.idle":"2023-06-12T10:55:28.616211Z","shell.execute_reply.started":"2023-06-12T10:55:22.061680Z","shell.execute_reply":"2023-06-12T10:55:28.614815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_to_mask(annotations):\n    \"\"\"\n    Converts annotations to a binary mask.\n\n    Parameters:\n    - annotations: A list of annotations for a single image.\n\n    Returns:\n    - mask: The binary mask.\n    \"\"\"\n    # Set the image dimensions directly\n    image_dimensions = (512, 512)\n    \n    # Create an empty mask of the same size as the image\n    mask = np.zeros(image_dimensions, dtype=np.uint8)\n    \n    # For each annotation\n    for annotation in annotations:\n        coordinates = np.array(annotation['coordinates'])\n        coordinates = coordinates.reshape(-1, 1, 2)\n        # Draw the polygon on the mask\n        cv2.fillPoly(mask, [coordinates], 255)\n\n    return mask","metadata":{"execution":{"iopub.status.busy":"2023-06-12T10:57:32.866884Z","iopub.execute_input":"2023-06-12T10:57:32.868528Z","iopub.status.idle":"2023-06-12T10:57:32.878540Z","shell.execute_reply.started":"2023-06-12T10:57:32.868467Z","shell.execute_reply":"2023-06-12T10:57:32.876778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create directory to store masks\nMASK_PATH = \"/masks\"\n\nos.makedirs(MASK_PATH, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T10:59:11.806774Z","iopub.execute_input":"2023-06-12T10:59:11.807273Z","iopub.status.idle":"2023-06-12T10:59:11.813816Z","shell.execute_reply.started":"2023-06-12T10:59:11.807238Z","shell.execute_reply":"2023-06-12T10:59:11.812412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create dictionary to store train data\ntrain_dict = {}\n\n# Save masks in the directory and fill the dictionary\nfor scan in annotations:\n    data = convert_to_mask(annotations[f'{scan}'])\n    mask_img = Image.fromarray((data).astype(np.uint8))  # convert to an image\n    filepath = f\"{MASK_PATH}/{scan}.png\"  # specify file path\n    mask_img.save(filepath)\n    train_dict[f'{scan}'] = [image_map[f'{scan}'], filepath]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Vizualize the first 5 images with masks\ncount = 0\nfor key in train_dict:\n    if count <= 3:\n        # Load image\n        img = cv2.imread(train_dict[f'{key}'][0])\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)  # OpenCV uses BGR color order\n\n        # Load mask\n        mask = cv2.imread(train_dict[f'{key}'][1])\n\n        # Display image and mask\n        plt.figure(figsize=(15,15))\n        plt.subplot(1,3,1)\n        plt.imshow(img, 'gray', interpolation='none')\n        plt.subplot(1,3,2)\n        plt.imshow(img, 'gray', interpolation='none')\n        plt.imshow(mask, 'jet', interpolation='none', alpha=0.7)\n        plt.subplot(1,3,3)\n        plt.imshow(mask, 'jet', interpolation='none')\n        count += 1\n    else: \n        break\n\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}