{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Hey everyone, this is my first time participating in a kaggle competion, I am looking forward to learning a lot about machine learning and deep learning. Any comments and suggestions would be appreciated.\n\nCurrent version: Data Import, preprocessing, and Binary Mask Generation","metadata":{}},{"cell_type":"code","source":"# imports\nimport glob\nimport os\nimport numpy as np\nimport json\nimport matplotlib.pyplot as plt\nfrom matplotlib.patches import Polygon\nimport cv2\nfrom scipy.ndimage import label","metadata":{"execution":{"iopub.status.busy":"2023-05-25T22:17:47.160160Z","iopub.execute_input":"2023-05-25T22:17:47.160686Z","iopub.status.idle":"2023-05-25T22:17:47.167843Z","shell.execute_reply.started":"2023-05-25T22:17:47.160651Z","shell.execute_reply":"2023-05-25T22:17:47.166575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = glob.glob(\"/kaggle/input/hubmap-hacking-the-human-vasculature/train/*\")\ntest = glob.glob(\"/kaggle/input/hubmap-hacking-the-human-vasculature/test/*\")","metadata":{"execution":{"iopub.status.busy":"2023-05-25T22:17:47.170426Z","iopub.execute_input":"2023-05-25T22:17:47.170830Z","iopub.status.idle":"2023-05-25T22:17:47.211892Z","shell.execute_reply.started":"2023-05-25T22:17:47.170798Z","shell.execute_reply":"2023-05-25T22:17:47.210609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a dictionary to hold the annotations for each image\nannotations = {}\n\n# Open the annotations file\nwith open('/kaggle/input/hubmap-hacking-the-human-vasculature/polygons.jsonl', 'r') as f:\n    # For each line in the file\n    for line in f:\n        # Parse the line as JSON\n        annotation = json.loads(line)\n\n        # Get the image ID and the list of annotations for this image\n        image_id = annotation['id']\n        image_annotations = annotation['annotations']\n\n        # Store the annotations in the dictionary\n        annotations[image_id] = image_annotations","metadata":{"execution":{"iopub.status.busy":"2023-05-25T22:17:47.214747Z","iopub.execute_input":"2023-05-25T22:17:47.215156Z","iopub.status.idle":"2023-05-25T22:17:51.643855Z","shell.execute_reply.started":"2023-05-25T22:17:47.215123Z","shell.execute_reply":"2023-05-25T22:17:51.642313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# map images to json annotations\nimage_map = {impath.split('/')[-1].split('.')[0]: impath for impath in train}\nimage_map.update({impath.split('/')[-1].split('.')[0]: impath for impath in test})","metadata":{"execution":{"iopub.status.busy":"2023-05-25T22:17:51.645884Z","iopub.execute_input":"2023-05-25T22:17:51.646334Z","iopub.status.idle":"2023-05-25T22:17:51.663143Z","shell.execute_reply.started":"2023-05-25T22:17:51.646295Z","shell.execute_reply":"2023-05-25T22:17:51.661501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Preprocess Image**\n\nThis is still a work in progress, and right now no real preprocessing is being done. This is created here for future work and experimentation \n\nI have been experimenting with stain normalization that can be refrenced here [Staintool](http://https://www.kaggle.com/competitions/hubmap-hacking-the-human-vasculature/discussion/412396) but i am having issues running it in kaggle kernel. works fine in colab and local, The needed libraries are not included in kaggle kernel.\n\nso right now it only converts to rgb and normalizes the images to 0-1","metadata":{}},{"cell_type":"code","source":"def preprocess_image(image):\n    # Convert the image from BGR to RGB\n    image_rgb = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n    # Resize image\n    image_resized = cv2.resize(image_rgb, (512 , 512))\n\n    return image_resized","metadata":{"execution":{"iopub.status.busy":"2023-05-25T22:17:51.667039Z","iopub.execute_input":"2023-05-25T22:17:51.667597Z","iopub.status.idle":"2023-05-25T22:17:51.680445Z","shell.execute_reply.started":"2023-05-25T22:17:51.667553Z","shell.execute_reply":"2023-05-25T22:17:51.678917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocessed_images = {}\nfor i, image_id in enumerate(list(annotations.keys())):\n    # Open the image file\n    if image_id in image_map:\n        image = cv2.imread(image_map[image_id])\n        preprocessed_image = preprocess_image(image.copy())\n        \n        # Normalize image to 0-1\n        preprocessed_image_normalized = preprocessed_image / 255.0\n\n        preprocessed_images[image_id] = preprocessed_image_normalized\n\n#         # For the first 5 images, display the original and preprocessed image side by side\n#         if i < 5:\n#             plt.figure(figsize=(10, 5))\n\n#             plt.subplot(1, 2, 1)\n#             plt.imshow(cv2.cvtColor(image, cv2.COLOR_BGR2RGB))\n#             plt.title(\"Original Image\")\n#             plt.axis('off')\n\n#             plt.subplot(1, 2, 2)\n#             plt.imshow(preprocessed_image_normalized)\n#             plt.title(\"Preprocessed Image\")\n#             plt.axis('off')\n\n#             plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-25T22:17:51.682446Z","iopub.execute_input":"2023-05-25T22:17:51.682906Z","iopub.status.idle":"2023-05-25T22:18:23.979502Z","shell.execute_reply.started":"2023-05-25T22:17:51.682866Z","shell.execute_reply":"2023-05-25T22:18:23.977886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Create binary masks, display results for first 5","metadata":{}},{"cell_type":"code","source":"def display_images_masks_overlays(image, masks, image_id):\n    overlay = image.copy()\n    colors = [(0, 255, 0), (255, 0, 0), (0, 0, 255)]  # colors for blood vessels, glomerulus and unsure respectively\n\n    # Initialize dictionary to hold annotation counts\n    annotation_counts = {}\n    \n    for idx, (annotation_type, mask) in enumerate(masks[image_id].items(), 2):\n        # Count the number of distinct areas in the mask\n        labeled, num_areas = label(mask)\n        annotation_counts[annotation_type] = num_areas\n\n        # update overlay\n        overlay[mask > 0] = colors[idx-2]\n\n    # print annotation counts\n    print(f\"Annotation counts for image {image_id}: {annotation_counts}\")\n\n    # prepare for subplot\n    plt.figure(figsize=(25,5))\n\n    # plot the original image\n    plt.subplot(1,5,1)\n    plt.imshow(image)\n    plt.title('Image')\n    plt.axis('off')\n    \n    for idx, (annotation_type, mask) in enumerate(masks[image_id].items(), 2):\n        # plot the binary masks\n        plt.subplot(1,5,idx)\n        plt.imshow(mask, cmap='gray')\n        plt.title(f'{annotation_type} mask')\n        plt.axis('off')\n\n    # plot the overlay\n    plt.subplot(1,5,5)\n    plt.imshow(image)  # Use RGB image\n    plt.imshow(overlay, alpha=0.4)  # change alpha to adjust transparency\n    plt.title('Overlay')\n    plt.axis('off')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-25T22:22:30.336573Z","iopub.execute_input":"2023-05-25T22:22:30.337214Z","iopub.status.idle":"2023-05-25T22:22:30.351386Z","shell.execute_reply.started":"2023-05-25T22:22:30.337174Z","shell.execute_reply":"2023-05-25T22:22:30.349825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"masks = {}\nannotation_types = ['blood_vessel', 'glomerulus', 'unsure']\nfor image_id in annotations.keys():\n    if image_id in preprocessed_images:\n        polygons = annotations[image_id]\n        # Load the preprocessed image\n        image = preprocessed_images[image_id]\n\n        # Create an empty mask of the same size as the image\n        masks[image_id] = {annotation_type: np.zeros(image.shape[:2], dtype=np.uint8) for annotation_type in annotation_types}\n\n        # For each polygon\n        for polygon in polygons:\n            annotation_type = polygon['type']\n            lines = np.array(polygon['coordinates'])\n            lines = lines.reshape(-1, 1, 2)\n            # Draw the polygon on the mask\n            cv2.fillPoly(masks[image_id][annotation_type], [lines], 255)\n\n        # Display the image and its masks if it's one of the first 5 images\n        if len(masks) <= 5:\n            display_images_masks_overlays(image, masks, image_id)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}