{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!cp -r /kaggle/input/jsonlines/ /kaggle/working/jsonlines\n\n\n!pip install /kaggle/working/jsonlines/jsonlines-3.1.0  --no-index --find-links=/kaggle/working/jsonlines/ \n\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-23T10:23:48.200522Z","iopub.execute_input":"2023-06-23T10:23:48.201143Z","iopub.status.idle":"2023-06-23T10:24:05.238416Z","shell.execute_reply.started":"2023-06-23T10:23:48.201115Z","shell.execute_reply":"2023-06-23T10:24:05.237251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp -r /kaggle/input/pycocotools/ /kaggle/working/pycocotools\n!pip install /kaggle/working/pycocotools/pycocotools-2.0.6  --no-index --find-links=/kaggle/working/pycocotools/ ","metadata":{"execution":{"iopub.status.busy":"2023-06-23T10:24:08.300003Z","iopub.execute_input":"2023-06-23T10:24:08.300406Z","iopub.status.idle":"2023-06-23T10:24:41.646741Z","shell.execute_reply.started":"2023-06-23T10:24:08.300370Z","shell.execute_reply":"2023-06-23T10:24:41.645500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!pip install imantics --quiet","metadata":{"execution":{"iopub.status.busy":"2023-06-23T10:24:41.650277Z","iopub.execute_input":"2023-06-23T10:24:41.650687Z","iopub.status.idle":"2023-06-23T10:24:41.658362Z","shell.execute_reply.started":"2023-06-23T10:24:41.650647Z","shell.execute_reply":"2023-06-23T10:24:41.657301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp -r /kaggle/input/xmljson-whl/xmljson-0.2.1-py2.py3-none-any.whl /kaggle/working/xmljson-whl/xmljson-0.2.1-py2.py3-none-any.whl\n\n!pip install /kaggle/input/xmljson-whl/xmljson-0.2.1-py2.py3-none-any.whl  --no-index --find-links=/kaggle/working/xmljson-whl/xmljson-0.2.1-py2.py3-none-any.whl\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-23T10:24:41.659435Z","iopub.execute_input":"2023-06-23T10:24:41.659735Z","iopub.status.idle":"2023-06-23T10:24:53.420588Z","shell.execute_reply.started":"2023-06-23T10:24:41.659709Z","shell.execute_reply":"2023-06-23T10:24:53.419434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp -r /kaggle/input/imantics/ /kaggle/working/imantics\n!pip install /kaggle/working/imantics/imantics-0.1.12  --no-index --find-links=/kaggle/working/imantics/ \n","metadata":{"execution":{"iopub.status.busy":"2023-06-23T10:26:32.908541Z","iopub.execute_input":"2023-06-23T10:26:32.908963Z","iopub.status.idle":"2023-06-23T10:26:47.269159Z","shell.execute_reply.started":"2023-06-23T10:26:32.908928Z","shell.execute_reply":"2023-06-23T10:26:47.268011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Loading the data is taken from https://www.kaggle.com/code/stpeteishii/hubmap-unet/notebook (Thanks for sharing!)","metadata":{}},{"cell_type":"markdown","source":"## U-net ++\n\nU-Net++ consists of an encoder and decoder path similar to U-Net, but it also includes nested and dense skip pathways. These additional pathways aim to alleviate the \"unknown\" boundary issue by shortening the semantic gap between the feature maps of the encoder and the decoder sub-networks. This architecture allows the model to capture more fine-grained details and thus yields more accurate segmentation maps.\n\nIn U-Net++, the output is not only from the final layer, but also from intermediate layers, which are fused together for the final prediction. This is especially beneficial in scenarios where certain features may only be visible at certain resolutions. By effectively combining these multi-resolution features, U-Net++ can often provide more accurate segmentation results.","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport pandas as pd\nimport jsonlines\nimport numpy as np\nfrom typing import List, Tuple\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport cv2\nimport tensorflow as tf\nimport json\nimport os\nimport imantics\nfrom PIL import Image\nfrom skimage.transform import resize\nimport random\nfrom sklearn.model_selection import train_test_split\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2023-06-23T10:26:47.272927Z","iopub.execute_input":"2023-06-23T10:26:47.273280Z","iopub.status.idle":"2023-06-23T10:26:56.121987Z","shell.execute_reply.started":"2023-06-23T10:26:47.273249Z","shell.execute_reply":"2023-06-23T10:26:56.120960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_dir = '/kaggle/input/hubmap-hacking-the-human-vasculature'\nannote_dir = f'{base_dir}/polygons.jsonl'\nimages_dir = f'{base_dir}/train' \ntest_images_dir = f'{base_dir}/test' ","metadata":{"execution":{"iopub.status.busy":"2023-06-23T10:26:56.123283Z","iopub.execute_input":"2023-06-23T10:26:56.124476Z","iopub.status.idle":"2023-06-23T10:26:56.129109Z","shell.execute_reply.started":"2023-06-23T10:26:56.124441Z","shell.execute_reply":"2023-06-23T10:26:56.128248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_size = 512\ninput_image_size = (512,512)","metadata":{"execution":{"iopub.status.busy":"2023-06-23T10:26:56.131602Z","iopub.execute_input":"2023-06-23T10:26:56.132529Z","iopub.status.idle":"2023-06-23T10:26:56.139681Z","shell.execute_reply.started":"2023-06-23T10:26:56.132497Z","shell.execute_reply":"2023-06-23T10:26:56.138757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_listdir = os.listdir(images_dir)\nrandom_images = np.random.choice(images_listdir, size = 9, replace = False)","metadata":{"execution":{"iopub.status.busy":"2023-06-23T10:26:56.141134Z","iopub.execute_input":"2023-06-23T10:26:56.141673Z","iopub.status.idle":"2023-06-23T10:26:56.386818Z","shell.execute_reply.started":"2023-06-23T10:26:56.141629Z","shell.execute_reply":"2023-06-23T10:26:56.385858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_image(path):\n    img = cv2.imread(path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = cv2.resize(img, (image_size, image_size))\n    return img","metadata":{"execution":{"iopub.status.busy":"2023-06-23T10:26:56.388464Z","iopub.execute_input":"2023-06-23T10:26:56.389135Z","iopub.status.idle":"2023-06-23T10:26:56.394572Z","shell.execute_reply.started":"2023-06-23T10:26:56.389102Z","shell.execute_reply":"2023-06-23T10:26:56.393738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rows = 3\ncols = 3\nfig, ax = plt.subplots(rows, cols, figsize = (12,12))\n\nfor i, ax in enumerate(ax.flat):\n    if i < len(random_images):\n        img = read_image(f\"{images_dir}/{random_images[i]}\")\n        #print(img.shape)\n        ax.set_title(f\"{random_images[i]}\")\n        ax.imshow(img)\n        ax.axis('off')","metadata":{"execution":{"iopub.status.busy":"2023-06-23T10:26:56.396298Z","iopub.execute_input":"2023-06-23T10:26:56.397074Z","iopub.status.idle":"2023-06-23T10:26:58.368376Z","shell.execute_reply.started":"2023-06-23T10:26:56.397042Z","shell.execute_reply":"2023-06-23T10:26:58.367451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(annote_dir) as file:\n    for i,line in enumerate(file):\n        json_data = json.loads(line)\n        # Process the individual JSON object here\n        print(i,json_data.keys())\n        print(i,json_data['annotations'][0].keys())\n        break","metadata":{"execution":{"iopub.status.busy":"2023-06-23T10:26:58.369391Z","iopub.execute_input":"2023-06-23T10:26:58.369698Z","iopub.status.idle":"2023-06-23T10:26:58.385855Z","shell.execute_reply.started":"2023-06-23T10:26:58.369671Z","shell.execute_reply":"2023-06-23T10:26:58.384821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(annote_dir) as annotes:\n    for i,annote in enumerate(annotes):\n        if i<1:\n            annotej=json.loads(annote)#str to dict\n            print(annotej.keys())#error\n            print(annotej['id'])\n            print(len(annotej['annotations']))\n            print(annotej['annotations'][0].keys())\n            print(annotej['annotations'][0]['type'])\n            #print(annotej['annotations'][0]['coordinates'])","metadata":{"execution":{"iopub.status.busy":"2023-06-23T10:26:58.387416Z","iopub.execute_input":"2023-06-23T10:26:58.387807Z","iopub.status.idle":"2023-06-23T10:26:58.761482Z","shell.execute_reply.started":"2023-06-23T10:26:58.387777Z","shell.execute_reply":"2023-06-23T10:26:58.760476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_dir","metadata":{"execution":{"iopub.status.busy":"2023-06-23T10:26:58.768368Z","iopub.execute_input":"2023-06-23T10:26:58.770923Z","iopub.status.idle":"2023-06-23T10:26:58.780794Z","shell.execute_reply.started":"2023-06-23T10:26:58.770886Z","shell.execute_reply":"2023-06-23T10:26:58.779830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"annotejs=[]\nwith open(annote_dir) as annotes:\n    for i,annote in enumerate(annotes):\n        annotej=json.loads(annote)#str to dict\n        annotejs+=[annotej]","metadata":{"execution":{"iopub.status.busy":"2023-06-23T10:26:58.782496Z","iopub.execute_input":"2023-06-23T10:26:58.783128Z","iopub.status.idle":"2023-06-23T10:27:03.492563Z","shell.execute_reply.started":"2023-06-23T10:26:58.783094Z","shell.execute_reply":"2023-06-23T10:27:03.491437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MASKS=np.zeros((1,image_size, image_size, 1), dtype=bool)\nIMAGES=np.zeros((1,image_size, image_size, 3),dtype=np.uint8)\n\nfor j,annotej in enumerate(annotejs[0:501]):##the smaller, the faster\n    #print(j)\n    masks = np.zeros((len(annotej[\"annotations\"]), image_size, image_size, 1), dtype=bool)\n\n    idi=annotej['id']\n    path=os.path.join(images_dir,idi+'.tif')\n    image=read_image(path)\n\n    for i,annotation in enumerate(annotej[\"annotations\"]):\n        segmentation = annotation[\"coordinates\"]\n        cur_mask = imantics.Polygons(segmentation).mask(*input_image_size).array\n        cur_mask = np.expand_dims(resize(cur_mask, (image_size, image_size), mode='constant', preserve_range=True), 2)\n        masks[i] = masks[i] | cur_mask \n        \n    mask2=np.sum(masks, axis=0) \n    mask2_ex = np.expand_dims(mask2, axis=0)\n    image_ex = np.expand_dims(image, axis=0)\n\n    MASKS=np.vstack([MASKS, mask2_ex])\n    IMAGES=np.vstack([IMAGES, image_ex])","metadata":{"execution":{"iopub.status.busy":"2023-06-23T10:27:03.494116Z","iopub.execute_input":"2023-06-23T10:27:03.494501Z","iopub.status.idle":"2023-06-23T10:29:49.295876Z","shell.execute_reply.started":"2023-06-23T10:27:03.494465Z","shell.execute_reply":"2023-06-23T10:29:49.294873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images=np.array(IMAGES)[1:501]\nmasks=np.array(MASKS)[1:501]\nprint(images.shape,masks.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-23T10:29:49.297231Z","iopub.execute_input":"2023-06-23T10:29:49.297784Z","iopub.status.idle":"2023-06-23T10:29:49.866577Z","shell.execute_reply.started":"2023-06-23T10:29:49.297751Z","shell.execute_reply":"2023-06-23T10:29:49.865661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_train, images_test, masks_train, masks_test = train_test_split(\n    images, masks, test_size=0.05, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-06-23T10:29:49.871876Z","iopub.execute_input":"2023-06-23T10:29:49.877408Z","iopub.status.idle":"2023-06-23T10:29:50.420156Z","shell.execute_reply.started":"2023-06-23T10:29:49.877371Z","shell.execute_reply":"2023-06-23T10:29:50.419230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(images_train), len(masks_train))","metadata":{"execution":{"iopub.status.busy":"2023-06-23T10:29:50.424626Z","iopub.execute_input":"2023-06-23T10:29:50.427099Z","iopub.status.idle":"2023-06-23T10:29:50.435657Z","shell.execute_reply.started":"2023-06-23T10:29:50.427063Z","shell.execute_reply":"2023-06-23T10:29:50.434642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n\n# Check if GPU is available\nif tf.test.gpu_device_name():\n    print('Default GPU Device: {}'.format(tf.test.gpu_device_name()))\nelse:\n    print(\"Please install GPU version of TF\")\n","metadata":{"execution":{"iopub.status.busy":"2023-06-23T10:29:50.439404Z","iopub.execute_input":"2023-06-23T10:29:50.441460Z","iopub.status.idle":"2023-06-23T10:29:52.973356Z","shell.execute_reply.started":"2023-06-23T10:29:50.441426Z","shell.execute_reply":"2023-06-23T10:29:52.972235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Generator class used in training to avoid memmory constraints","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.utils import Sequence\nfrom skimage.transform import resize\nimport cv2\nimport numpy as np\nimport os\nimport json\nimport imantics\n\nclass DataGen(Sequence):\n    def __init__(self, annotejs, images_dir, batch_size=8, image_size=512):\n        self.annotejs = annotejs\n        self.images_dir = images_dir\n        self.batch_size = batch_size\n        self.image_size = image_size\n        self.on_epoch_end()\n    \n    def __load__(self, annotej):\n        # Preparing the mask array\n        masks = np.zeros((self.image_size, self.image_size, 1), dtype=bool)\n\n        idi=annotej['id']\n        image_path = os.path.join(self.images_dir, idi+'.tif')\n\n        # Read and resize image\n        image = cv2.imread(image_path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image = cv2.resize(image, (self.image_size, self.image_size))\n\n        # Iterate over each annotation\n        for i,annotation in enumerate(annotej[\"annotations\"]):\n            segmentation = annotation[\"coordinates\"]\n            cur_mask = imantics.Polygons(segmentation).mask(*input_image_size).array\n            cur_mask = np.expand_dims(resize(cur_mask, (self.image_size, self.image_size), mode='constant', preserve_range=True), 2)\n            masks = masks | cur_mask \n        \n        ## Normalize \n        image = image/255.0\n        # Binarize the mask\n        masks = masks > 0\n\n        return image, masks\n    \n    def __getitem__(self, index):\n        if(index+1)*self.batch_size > len(self.annotejs):\n            self.batch_size = len(self.annotejs) - index*self.batch_size\n\n        # Select the right batch\n        batch_annotejs = self.annotejs[index*self.batch_size : (index+1)*self.batch_size]\n\n        images = []\n        masks  = []\n\n        # Load each image and mask pair\n        for annotej in batch_annotejs:\n            _img, _mask = self.__load__(annotej)\n            images.append(_img)\n            masks.append(_mask)\n\n        # Return the batch\n        return np.array(images), np.array(masks)\n    \n    def on_epoch_end(self):\n        pass\n    \n    def __len__(self):\n        return int(np.ceil(len(self.annotejs)/float(self.batch_size)))\n","metadata":{"execution":{"iopub.status.busy":"2023-06-23T10:29:52.974782Z","iopub.execute_input":"2023-06-23T10:29:52.975310Z","iopub.status.idle":"2023-06-23T10:29:52.992103Z","shell.execute_reply.started":"2023-06-23T10:29:52.975274Z","shell.execute_reply":"2023-06-23T10:29:52.991081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model + Train loop (uncomment for training)","metadata":{}},{"cell_type":"code","source":"from keras.callbacks import ModelCheckpoint\n\ndef conv_block(x, filters):\n    x = tf.keras.layers.Conv2D(filters, (3, 3), padding=\"same\")(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.Activation(\"relu\")(x)\n    return x\n\ndef U_net_plus_plus(input_shape, num_classes):\n    inputs = tf.keras.layers.Input(input_shape)\n\n    # First part of U-Net\n    c1 = conv_block(inputs, 64)\n    p1 = tf.keras.layers.MaxPooling2D((2, 2))(c1)\n\n    c2 = conv_block(p1, 128)\n    p2 = tf.keras.layers.MaxPooling2D((2, 2))(c2)\n\n    c3 = conv_block(p2, 256)\n    p3 = tf.keras.layers.MaxPooling2D((2, 2))(c3)\n\n    c4 = conv_block(p3, 512)\n    p4 = tf.keras.layers.MaxPooling2D((2, 2))(c4)\n\n    c5 = conv_block(p4, 1024)\n\n    # Second part of U-Net++\n    u6 = tf.keras.layers.Conv2DTranspose(512, (2, 2), strides=(2, 2), padding='same')(c5)\n    u6 = tf.keras.layers.concatenate([u6, c4])\n    c6 = conv_block(u6, 512)\n\n    u7 = tf.keras.layers.Conv2DTranspose(256, (2, 2), strides=(2, 2), padding='same')(c6)\n    u7 = tf.keras.layers.concatenate([u7, c3])\n    c7 = conv_block(u7, 256)\n\n    u8 = tf.keras.layers.Conv2DTranspose(128, (2, 2), strides=(2, 2), padding='same')(c7)\n    u8 = tf.keras.layers.concatenate([u8, c2])\n    c8 = conv_block(u8, 128)\n\n    u9 = tf.keras.layers.Conv2DTranspose(64, (2, 2), strides=(2, 2), padding='same')(c8)\n    u9 = tf.keras.layers.concatenate([u9, c1], axis=3)\n    c9 = conv_block(u9, 64)\n\n    outputs = tf.keras.layers.Conv2D(1, (1, 1), activation='sigmoid')(c9)\n\n\n    model = tf.keras.models.Model(inputs=[inputs], outputs=[outputs])\n\n    return model\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-23T10:29:52.993782Z","iopub.execute_input":"2023-06-23T10:29:52.994218Z","iopub.status.idle":"2023-06-23T10:29:53.017030Z","shell.execute_reply.started":"2023-06-23T10:29:52.994165Z","shell.execute_reply":"2023-06-23T10:29:53.015865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n\"\"\"\n# Split the annotejs list into training and validation\nannotejs_train, annotejs_val = train_test_split(annotejs, test_size=0.2, random_state=42)\n\n# Instantiate the data generators\ntrain_gen = DataGen(annotejs_train, images_dir, batch_size=2, image_size=512)\nval_gen = DataGen(annotejs_val, images_dir, batch_size=2, image_size=512)\n\n# Instantiate the model\nmodel = U_net_plus_plus((512, 512, 3), 2)\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n\n# Check model summary\nmodel.summary()\n\n# Save the model after every epoch\ncheckpointer = ModelCheckpoint('model.h5', verbose=1, save_best_only=True)\n\n# Train the model using the data generator\nhistory = model.fit(train_gen,\n                    validation_data=val_gen,\n                    epochs=12,\n                    callbacks=[checkpointer])\n\n# Save the model after training\nmodel.save(\"model.h5\")\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-06-22T20:53:10.671738Z","iopub.execute_input":"2023-06-22T20:53:10.672144Z","iopub.status.idle":"2023-06-22T20:53:10.679209Z","shell.execute_reply.started":"2023-06-22T20:53:10.672114Z","shell.execute_reply":"2023-06-22T20:53:10.678257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference","metadata":{}},{"cell_type":"markdown","source":"Inference code was not functional. But you can upload the trained model into this inference pipeline:\n\n- https://www.kaggle.com/code/josipvrdoljak/fork-of-u-net-inference","metadata":{}},{"cell_type":"markdown","source":"## Inference offers poor result, Things to do: Use a pre-trained backbone like EffNet or ResNet","metadata":{}},{"cell_type":"code","source":"#df\n","metadata":{"execution":{"iopub.status.busy":"2023-06-22T21:41:16.150336Z","iopub.execute_input":"2023-06-22T21:41:16.151252Z","iopub.status.idle":"2023-06-22T21:41:16.161801Z","shell.execute_reply.started":"2023-06-22T21:41:16.151207Z","shell.execute_reply":"2023-06-22T21:41:16.160651Z"},"trusted":true},"execution_count":null,"outputs":[]}]}