{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!cp -r /kaggle/input/jsonlines/ /kaggle/working/jsonlines\n\n\n!pip install /kaggle/working/jsonlines/jsonlines-3.1.0  --no-index --find-links=/kaggle/working/jsonlines/ \n\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-26T08:23:08.330364Z","iopub.execute_input":"2023-06-26T08:23:08.330661Z","iopub.status.idle":"2023-06-26T08:23:24.674094Z","shell.execute_reply.started":"2023-06-26T08:23:08.330634Z","shell.execute_reply":"2023-06-26T08:23:24.672955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp -r /kaggle/input/pycocotools/ /kaggle/working/pycocotools\n!pip install /kaggle/working/pycocotools/pycocotools-2.0.6  --no-index --find-links=/kaggle/working/pycocotools/ ","metadata":{"execution":{"iopub.status.busy":"2023-06-26T08:25:14.126053Z","iopub.execute_input":"2023-06-26T08:25:14.126500Z","iopub.status.idle":"2023-06-26T08:25:47.358606Z","shell.execute_reply.started":"2023-06-26T08:25:14.126466Z","shell.execute_reply":"2023-06-26T08:25:47.357479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!pip install imantics --quiet","metadata":{"execution":{"iopub.status.busy":"2023-06-26T08:25:47.361709Z","iopub.execute_input":"2023-06-26T08:25:47.362123Z","iopub.status.idle":"2023-06-26T08:25:47.371707Z","shell.execute_reply.started":"2023-06-26T08:25:47.362092Z","shell.execute_reply":"2023-06-26T08:25:47.369450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp -r /kaggle/input/xmljson-whl/xmljson-0.2.1-py2.py3-none-any.whl /kaggle/working/xmljson-whl/xmljson-0.2.1-py2.py3-none-any.whl\n\n!pip install /kaggle/input/xmljson-whl/xmljson-0.2.1-py2.py3-none-any.whl  --no-index --find-links=/kaggle/working/xmljson-whl/xmljson-0.2.1-py2.py3-none-any.whl\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-26T08:25:47.373668Z","iopub.execute_input":"2023-06-26T08:25:47.373999Z","iopub.status.idle":"2023-06-26T08:25:59.170249Z","shell.execute_reply.started":"2023-06-26T08:25:47.373969Z","shell.execute_reply":"2023-06-26T08:25:59.169162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp -r /kaggle/input/imantics/ /kaggle/working/imantics\n!pip install /kaggle/working/imantics/imantics-0.1.12  --no-index --find-links=/kaggle/working/imantics/ \n","metadata":{"execution":{"iopub.status.busy":"2023-06-26T08:25:59.174743Z","iopub.execute_input":"2023-06-26T08:25:59.175061Z","iopub.status.idle":"2023-06-26T08:26:13.136816Z","shell.execute_reply.started":"2023-06-26T08:25:59.175033Z","shell.execute_reply":"2023-06-26T08:26:13.135559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Loading the data is taken from https://www.kaggle.com/code/stpeteishii/hubmap-unet/notebook (Thanks for sharing!)","metadata":{}},{"cell_type":"markdown","source":"## U-net ++\n\nU-Net++ consists of an encoder and decoder path similar to U-Net, but it also includes nested and dense skip pathways. These additional pathways aim to alleviate the \"unknown\" boundary issue by shortening the semantic gap between the feature maps of the encoder and the decoder sub-networks. This architecture allows the model to capture more fine-grained details and thus yields more accurate segmentation maps.\n\nIn U-Net++, the output is not only from the final layer, but also from intermediate layers, which are fused together for the final prediction. This is especially beneficial in scenarios where certain features may only be visible at certain resolutions. By effectively combining these multi-resolution features, U-Net++ can often provide more accurate segmentation results.","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport pandas as pd\nimport jsonlines\nimport numpy as np\nfrom typing import List, Tuple\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport cv2\nimport tensorflow as tf\nimport json\nimport os\nimport imantics\nfrom PIL import Image\nfrom skimage.transform import resize\nimport random\nfrom sklearn.model_selection import train_test_split\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2023-06-26T08:26:33.412916Z","iopub.execute_input":"2023-06-26T08:26:33.413334Z","iopub.status.idle":"2023-06-26T08:26:42.868126Z","shell.execute_reply.started":"2023-06-26T08:26:33.413288Z","shell.execute_reply":"2023-06-26T08:26:42.867169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os\nfrom collections import Counter\n\nfrom PIL import Image\nimport imageio\nimport json \nimport cv2\nimport ipywidgets as widgets\nimport IPython.display as ipd\nimport glob\nfrom pycocotools import _mask as coco_mask\nimport logging\n\nimport base64\nimport numpy as np\nimport typing as t\nimport zlib\n\nimport tensorflow as tf\nfrom keras import backend as K\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Conv2D, Conv2DTranspose, MaxPooling2D, Dropout, concatenate\n\nprint(f\"tensorflow version: {tf.__version__}\")","metadata":{"execution":{"iopub.status.busy":"2023-06-26T08:26:42.869753Z","iopub.execute_input":"2023-06-26T08:26:42.870467Z","iopub.status.idle":"2023-06-26T08:26:43.025404Z","shell.execute_reply.started":"2023-06-26T08:26:42.870433Z","shell.execute_reply":"2023-06-26T08:26:43.024458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport base64\nimport zlib\nimport numpy as np\nimport pandas as pd\nfrom tensorflow.keras.models import load_model\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nfrom sklearn.cluster import DBSCAN\nfrom pycocotools import _mask as coco_mask","metadata":{"execution":{"iopub.status.busy":"2023-06-26T08:27:58.721494Z","iopub.execute_input":"2023-06-26T08:27:58.721883Z","iopub.status.idle":"2023-06-26T08:27:59.046456Z","shell.execute_reply.started":"2023-06-26T08:27:58.721851Z","shell.execute_reply":"2023-06-26T08:27:59.045507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def jaccard_coef(y_true, y_pred):\n    y_true_f = K.flatten(y_true)\n    y_pred_f = K.flatten(y_pred)\n    intersection = K.sum(y_true_f * y_pred_f)\n    return (intersection + 1.0) / (K.sum(y_true_f) + K.sum(y_pred_f) - intersection + 1.0)\n\n\ndef jaccard_loss(y_true, y_pred):\n    return -jaccard_coef(y_true, y_pred)\n\nwith tf.keras.utils.custom_object_scope({'jaccard_loss': jaccard_loss, 'jaccard_coef':jaccard_coef}):\n    model = load_model('/kaggle/input/model-5/model.h5')","metadata":{"execution":{"iopub.status.busy":"2023-06-26T08:28:01.227429Z","iopub.execute_input":"2023-06-26T08:28:01.227827Z","iopub.status.idle":"2023-06-26T08:28:08.529543Z","shell.execute_reply.started":"2023-06-26T08:28:01.227787Z","shell.execute_reply":"2023-06-26T08:28:08.528504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tta=True","metadata":{"execution":{"iopub.status.busy":"2023-06-26T08:43:18.744331Z","iopub.execute_input":"2023-06-26T08:43:18.744746Z","iopub.status.idle":"2023-06-26T08:43:18.749689Z","shell.execute_reply.started":"2023-06-26T08:43:18.744715Z","shell.execute_reply":"2023-06-26T08:43:18.748808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_watershed(prediction, plot=False):\n    normalized_prediction = cv2.normalize(prediction, None, 0, 255, cv2.NORM_MINMAX, cv2.CV_8UC3)\n    ret, thresh = cv2.threshold(normalized_prediction, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n    \n    kernel = np.ones((3,3),np.uint8)\n    opening = cv2.morphologyEx(thresh,cv2.MORPH_OPEN,kernel, iterations = 4)\n    \n    sure_bg = cv2.dilate(opening,kernel,iterations=3)\n    \n    dist_transform = cv2.distanceTransform(opening,cv2.DIST_L2,5)\n    ret, sure_fg = cv2.threshold(dist_transform,0.15*dist_transform.max(),255,0)\n    \n    sure_fg = np.uint8(sure_fg)\n    unknown = cv2.subtract(sure_bg,sure_fg)\n    \n    ret3, markers = cv2.connectedComponents(sure_fg)\n    \n    markers_norm = np.int32(markers + 10)\n    markers_norm[unknown == 255] = 0\n    \n    prediction_rgb = cv2.cvtColor(prediction.astype(np.float32), cv2.COLOR_GRAY2BGR)\n    image_cv = np.uint8(prediction_rgb)\n\n    markers = cv2.watershed(image_cv, markers_norm)\n    \n    if plot:\n        fig, axs = plt.subplots(6, 1, figsize=(6, 6 * 6))\n        \n        axs[0].imshow(thresh, cmap='gray')\n        axs[0].set_title(\"Thresholded Prediction\")\n        \n        axs[1].imshow(opening, cmap='gray')\n        axs[1].set_title('Prediction after removing noise')\n            \n        axs[2].imshow(sure_bg, cmap='gray')\n        axs[2].set_title('Sure Background')\n        \n        axs[3].imshow(sure_fg, cmap='gray')\n        axs[3].set_title('Sure Foreground')\n        \n        axs[4].imshow(unknown, cmap='gray')\n        axs[4].set_title('Unknown Region')\n        \n        axs[5].imshow(markers, cmap='gray')\n        axs[5].set_title('Watershed image (instantiated image)')\n        \n        plt.show()\n    return markers\n\ndef get_binary_masks(markers, background_label=10, marker_label=-1):\n    \"\"\"\n    takes watershed markers and \n    \"\"\"\n    unique_labels = np.unique(markers)\n    \n    if background_label in unique_labels:\n        unique_labels = unique_labels[unique_labels != background_label]\n    if marker_label in unique_labels:\n        unique_labels = unique_labels[unique_labels != marker_label]\n\n    binary_masks = []\n\n    for label in unique_labels:\n        mask = np.where(markers == label, 1, 0).astype(np.uint8)\n        binary_masks.append(mask)\n        \n    return binary_masks\n\ndef plot_binary_masks(prediction):\n    markers = get_watershed(prediction)\n    binary_masks = get_binary_masks(markers)\n\n    fig, axs = plt.subplots(len(binary_masks), 1, figsize=(6,len(binary_masks) * 6))\n    for i, mask in enumerate(binary_masks):\n        axs[i].imshow(mask, cmap='gray')\n        axs[i].set_title(f\"Instance Segmentation {i + 1}\")\n        \ndef encode_binary_mask(mask: np.ndarray) -> str:\n    # Check input mask\n    if mask.dtype != np.bool:\n        raise ValueError(\n            \"encode_binary_mask expects a binary mask, received dtype == %s\" %\n            mask.dtype)\n\n    mask = np.squeeze(mask)\n    if len(mask.shape) != 2:\n        raise ValueError(\n            \"encode_binary_mask expects a 2d mask, received shape == %s\" %\n            mask.shape)\n\n    # convert input mask to expected COCO API input --\n    mask_to_encode = mask.reshape(mask.shape[0], mask.shape[1], 1)\n    mask_to_encode = mask_to_encode.astype(np.uint8)\n    mask_to_encode = np.asfortranarray(mask_to_encode)\n\n    # RLE encode mask --\n    encoded_mask = coco_mask.encode(mask_to_encode)[0][\"counts\"]\n\n    # compress and base64 encoding --\n    binary_str = zlib.compress(encoded_mask, zlib.Z_BEST_COMPRESSION)\n    base64_str = base64.b64encode(binary_str)\n    return base64_str\n\ndef encode_all_masks(binary_masks):\n    encoded_masks = []\n    for mask in binary_masks:\n        encoded_mask = encode_binary_mask(mask.astype(np.bool))\n        encoded_masks.append(encoded_mask.decode('utf-8'))\n    prefixed_masks = ['0 1.0 ' + mask for mask in encoded_masks]\n    encoded_preds = ' '.join(prefixed_masks)\n    return encoded_preds\n\n\ndef create_submission(directory, model, plot=True, tta=True):\n    file_paths = [os.path.join(directory, path) for path in os.listdir(directory)]\n    predictions = []\n    image_ids = []\n    for path in file_paths:\n        image_id = os.path.splitext(os.path.basename(path))[0]\n        image=cv2.imread(path)/255\n        resized_image = cv2.resize(image, (512, 512), interpolation=cv2.INTER_AREA)\n        if resized_image.shape[2] == 1:\n            resized_image = cv2.cvtColor(resized_image, cv2.COLOR_GRAY2BGR) \n\n        if tta:\n            # Apply TTA\n            flips = [[-1],[-1, 0],[0]]  # flip vertically, horizontally, and both\n            tta_predictions = []\n            for flip in flips:\n                flipped_img = np.flip(resized_image, flip)\n                flipped_pred = model.predict(np.expand_dims(flipped_img, axis=0))[0]\n                tta_predictions.append(np.flip(flipped_pred, flip))  # flip prediction back to normal\n            # Averaging predictions\n            prediction = np.mean(tta_predictions, axis=0)\n        else:\n            prediction = model.predict(np.expand_dims(resized_image, axis=0))[0]\n\n        markers = get_watershed(prediction, plot=False)\n        binary_masks = get_binary_masks(markers)\n        encoded_preds = encode_all_masks(binary_masks)\n        \n        image_ids.append(image_id)\n        predictions.append(encoded_preds)\n    \n    if plot:\n        fig,axs = plt.subplots(1, 3, figsize=(24, 8))\n        axs[0].imshow(resized_image)\n        axs[0].set_title(f\"Image {image_id}\")\n        \n        axs[1].imshow(prediction, cmap='gray')\n        axs[1].set_title(f\"U-net Prediction\")\n        \n        axs[2].imshow(markers, cmap='gray')\n        axs[2].set_title(\"watershed + postprocessing\")\n        \n    df = pd.DataFrame({'id': image_ids, 'height': 512, 'width':512 ,'prediction_string': predictions})\n\n    return df\n","metadata":{"execution":{"iopub.status.busy":"2023-06-26T08:43:21.384111Z","iopub.execute_input":"2023-06-26T08:43:21.384520Z","iopub.status.idle":"2023-06-26T08:43:21.431212Z","shell.execute_reply.started":"2023-06-26T08:43:21.384488Z","shell.execute_reply":"2023-06-26T08:43:21.430273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = create_submission(\"/kaggle/input/hubmap-hacking-the-human-vasculature/test\", model)\ndf.to_csv('submission.csv', index=False)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-26T08:43:32.322216Z","iopub.execute_input":"2023-06-26T08:43:32.322896Z","iopub.status.idle":"2023-06-26T08:43:33.883254Z","shell.execute_reply.started":"2023-06-26T08:43:32.322861Z","shell.execute_reply":"2023-06-26T08:43:33.881855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}