{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# HuBMAP UNET\n","metadata":{}},{"cell_type":"markdown","source":"https://arxiv.org/pdf/1505.04597.pdf","metadata":{}},{"cell_type":"markdown","source":"U-Net is a convolutional neural network (CNN) architecture that was specifically designed for semantic segmentation tasks, such as predicting masks. It was proposed by researchers at the Computer Science Department of the University of Freiburg in 2015.\n\nThe U-Net architecture consists of an encoder-decoder network with skip connections. The encoder portion of the network is responsible for capturing high-level features from the input image, while the decoder portion aims to reconstruct the output mask from those features.\n\nTraining U-Net typically requires a large dataset of input images with corresponding ground truth masks. The network is trained using a loss function that compares the predicted mask with the ground truth mask, such as the pixel-wise binary cross-entropy loss. The weights of the network are updated using backpropagation and stochastic gradient descent or other optimization algorithms.\n\nOverall, U-Net is a powerful architecture for predicting masks in semantic segmentation tasks, thanks to its encoder-decoder structure and skip connections that allow it to capture detailed spatial information and produce accurate predictions.","metadata":{}},{"cell_type":"code","source":"!pip install imantics --quiet","metadata":{"execution":{"iopub.status.busy":"2023-06-18T10:07:47.052783Z","iopub.execute_input":"2023-06-18T10:07:47.053730Z","iopub.status.idle":"2023-06-18T10:07:58.792479Z","shell.execute_reply.started":"2023-06-18T10:07:47.053677Z","shell.execute_reply":"2023-06-18T10:07:58.791118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport cv2\nimport tensorflow as tf\nimport json\nimport os\nimport imantics\nfrom PIL import Image\nfrom skimage.transform import resize\nimport random\nfrom sklearn.model_selection import train_test_split\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2023-06-18T10:07:58.796209Z","iopub.execute_input":"2023-06-18T10:07:58.796909Z","iopub.status.idle":"2023-06-18T10:07:58.811383Z","shell.execute_reply.started":"2023-06-18T10:07:58.796871Z","shell.execute_reply":"2023-06-18T10:07:58.810196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_dir = '/kaggle/input/hubmap-hacking-the-human-vasculature'\nannote_dir = f'{base_dir}/polygons.jsonl'\nimages_dir = f'{base_dir}/train' \ntest_images_dir = f'{base_dir}/test' ","metadata":{"execution":{"iopub.status.busy":"2023-06-18T10:07:58.814905Z","iopub.execute_input":"2023-06-18T10:07:58.815245Z","iopub.status.idle":"2023-06-18T10:07:58.826674Z","shell.execute_reply.started":"2023-06-18T10:07:58.815196Z","shell.execute_reply":"2023-06-18T10:07:58.825625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_size = 512\ninput_image_size = (512,512)","metadata":{"execution":{"iopub.status.busy":"2023-06-18T10:07:58.830475Z","iopub.execute_input":"2023-06-18T10:07:58.831311Z","iopub.status.idle":"2023-06-18T10:07:58.837708Z","shell.execute_reply.started":"2023-06-18T10:07:58.831270Z","shell.execute_reply":"2023-06-18T10:07:58.836552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_listdir = os.listdir(images_dir)\nrandom_images = np.random.choice(images_listdir, size = 9, replace = False)","metadata":{"execution":{"iopub.status.busy":"2023-06-18T10:07:58.839247Z","iopub.execute_input":"2023-06-18T10:07:58.840468Z","iopub.status.idle":"2023-06-18T10:07:58.854647Z","shell.execute_reply.started":"2023-06-18T10:07:58.840429Z","shell.execute_reply":"2023-06-18T10:07:58.853518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_image(path):\n    img = cv2.imread(path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = cv2.resize(img, (image_size, image_size))\n    return img","metadata":{"execution":{"iopub.status.busy":"2023-06-18T10:07:58.866753Z","iopub.execute_input":"2023-06-18T10:07:58.867859Z","iopub.status.idle":"2023-06-18T10:07:58.874308Z","shell.execute_reply.started":"2023-06-18T10:07:58.867811Z","shell.execute_reply":"2023-06-18T10:07:58.873084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rows = 3\ncols = 3\nfig, ax = plt.subplots(rows, cols, figsize = (12,12))\n\nfor i, ax in enumerate(ax.flat):\n    if i < len(random_images):\n        img = read_image(f\"{images_dir}/{random_images[i]}\")\n        #print(img.shape)\n        ax.set_title(f\"{random_images[i]}\")\n        ax.imshow(img)\n        ax.axis('off')","metadata":{"execution":{"iopub.status.busy":"2023-06-18T11:08:44.360853Z","iopub.execute_input":"2023-06-18T11:08:44.361566Z","iopub.status.idle":"2023-06-18T11:08:45.677765Z","shell.execute_reply.started":"2023-06-18T11:08:44.361524Z","shell.execute_reply":"2023-06-18T11:08:45.673975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# added added added\nwith open(annote_dir) as file:\n    for i,line in enumerate(file):\n        json_data = json.loads(line)\n        # Process the individual JSON object here\n        print(i,json_data.keys())\n        print(i,json_data['annotations'][0].keys())\n        break\n","metadata":{"execution":{"iopub.status.busy":"2023-06-18T10:08:00.008448Z","iopub.execute_input":"2023-06-18T10:08:00.009147Z","iopub.status.idle":"2023-06-18T10:08:00.021650Z","shell.execute_reply.started":"2023-06-18T10:08:00.009108Z","shell.execute_reply":"2023-06-18T10:08:00.020314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(annote_dir) as annotes:\n    for i,annote in enumerate(annotes):\n        if i<1:\n            annotej=json.loads(annote)#str to dict\n            print(annotej.keys())#error\n            print(annotej['id'])\n            print(len(annotej['annotations']))\n            print(annotej['annotations'][0].keys())\n            print(annotej['annotations'][0]['type'])\n            #print(annotej['annotations'][0]['coordinates'])","metadata":{"execution":{"iopub.status.busy":"2023-06-18T10:08:00.023436Z","iopub.execute_input":"2023-06-18T10:08:00.024181Z","iopub.status.idle":"2023-06-18T10:08:00.064198Z","shell.execute_reply.started":"2023-06-18T10:08:00.024144Z","shell.execute_reply":"2023-06-18T10:08:00.063138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_dir","metadata":{"execution":{"iopub.status.busy":"2023-06-18T10:08:00.075035Z","iopub.execute_input":"2023-06-18T10:08:00.075613Z","iopub.status.idle":"2023-06-18T10:08:00.089293Z","shell.execute_reply.started":"2023-06-18T10:08:00.075573Z","shell.execute_reply":"2023-06-18T10:08:00.088066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"annotejs=[]\nwith open(annote_dir) as annotes:\n    for i,annote in enumerate(annotes):\n        annotej=json.loads(annote)#str to dict\n        annotejs+=[annotej]","metadata":{"execution":{"iopub.status.busy":"2023-06-18T10:08:00.090883Z","iopub.execute_input":"2023-06-18T10:08:00.091659Z","iopub.status.idle":"2023-06-18T10:08:05.495805Z","shell.execute_reply.started":"2023-06-18T10:08:00.091617Z","shell.execute_reply":"2023-06-18T10:08:05.493249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MASKS=np.zeros((1,image_size, image_size, 1), dtype=bool)\nIMAGES=np.zeros((1,image_size, image_size, 3),dtype=np.uint8)\n\nfor j,annotej in enumerate(annotejs[0:501]):##the smaller, the faster\n    #print(j)\n    masks = np.zeros((len(annotej[\"annotations\"]), image_size, image_size, 1), dtype=bool)\n\n    idi=annotej['id']\n    path=os.path.join(images_dir,idi+'.tif')\n    image=read_image(path)\n\n    for i,annotation in enumerate(annotej[\"annotations\"]):\n        segmentation = annotation[\"coordinates\"]\n        cur_mask = imantics.Polygons(segmentation).mask(*input_image_size).array\n        cur_mask = np.expand_dims(resize(cur_mask, (image_size, image_size), mode='constant', preserve_range=True), 2)\n        masks[i] = masks[i] | cur_mask \n        \n    mask2=np.sum(masks, axis=0) \n    mask2_ex = np.expand_dims(mask2, axis=0)\n    image_ex = np.expand_dims(image, axis=0)\n\n    MASKS=np.vstack([MASKS, mask2_ex])\n    IMAGES=np.vstack([IMAGES, image_ex])\n","metadata":{"execution":{"iopub.status.busy":"2023-06-18T10:08:05.498010Z","iopub.execute_input":"2023-06-18T10:08:05.498877Z","iopub.status.idle":"2023-06-18T10:08:18.083485Z","shell.execute_reply.started":"2023-06-18T10:08:05.498814Z","shell.execute_reply":"2023-06-18T10:08:18.082299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images=np.array(IMAGES)[1:501]\nmasks=np.array(MASKS)[1:501]\nprint(images.shape,masks.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-18T10:08:18.085195Z","iopub.execute_input":"2023-06-18T10:08:18.085620Z","iopub.status.idle":"2023-06-18T10:08:18.202158Z","shell.execute_reply.started":"2023-06-18T10:08:18.085577Z","shell.execute_reply":"2023-06-18T10:08:18.201055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_train, images_test, masks_train, masks_test = train_test_split(\n    images, masks, test_size=0.05, random_state=42)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-18T10:08:18.203544Z","iopub.execute_input":"2023-06-18T10:08:18.204243Z","iopub.status.idle":"2023-06-18T10:08:18.293454Z","shell.execute_reply.started":"2023-06-18T10:08:18.204189Z","shell.execute_reply":"2023-06-18T10:08:18.292274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(images_train), len(masks_train))","metadata":{"execution":{"iopub.status.busy":"2023-06-18T10:08:18.294945Z","iopub.execute_input":"2023-06-18T10:08:18.296011Z","iopub.status.idle":"2023-06-18T10:08:18.303298Z","shell.execute_reply.started":"2023-06-18T10:08:18.295969Z","shell.execute_reply":"2023-06-18T10:08:18.301057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# U-NET","metadata":{}},{"cell_type":"code","source":"def conv_block(input, num_filters):\n    conv = tf.keras.layers.Conv2D(num_filters, 3, padding=\"same\")(input)\n    conv = tf.keras.layers.BatchNormalization()(conv)\n    conv = tf.keras.layers.Activation(\"relu\")(conv)\n    conv = tf.keras.layers.Conv2D(num_filters, 3, padding=\"same\")(conv)\n    conv = tf.keras.layers.BatchNormalization()(conv)\n    conv = tf.keras.layers.Activation(\"relu\")(conv)\n    return conv\n\ndef encoder_block(input, num_filters):\n    skip = conv_block(input, num_filters)\n    pool = tf.keras.layers.MaxPool2D((2, 2))(skip)\n    return skip, pool\n\ndef decoder_block(input, skip, num_filters):\n    up_conv = tf.keras.layers.Conv2DTranspose(num_filters, (2, 2), strides=2, padding=\"same\")(input)\n    conv = tf.keras.layers.Concatenate()([up_conv, skip])\n    conv = conv_block(conv, num_filters)\n    return conv\n\ndef Unet(input_shape):\n    inputs = tf.keras.layers.Input(input_shape)\n\n    skip1, pool1 = encoder_block(inputs, 64)\n    skip2, pool2 = encoder_block(pool1, 128)\n    skip3, pool3 = encoder_block(pool2, 256)\n    skip4, pool4 = encoder_block(pool3, 512)\n\n    bridge = conv_block(pool4, 1024)\n\n    decode1 = decoder_block(bridge, skip4, 512)\n    decode2 = decoder_block(decode1, skip3, 256)\n    decode3 = decoder_block(decode2, skip2, 128)\n    decode4 = decoder_block(decode3, skip1, 64)\n\n    outputs = tf.keras.layers.Conv2D(1, 1, padding=\"same\", activation=\"sigmoid\")(decode4)\n\n    model = tf.keras.models.Model(inputs, outputs, name=\"U-Net\")\n    return model\n\nunet_model = Unet((512, 512, 3))\nunet_model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\nunet_model.summary()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-18T10:08:18.316202Z","iopub.execute_input":"2023-06-18T10:08:18.317393Z","iopub.status.idle":"2023-06-18T10:08:19.016302Z","shell.execute_reply.started":"2023-06-18T10:08:18.317329Z","shell.execute_reply":"2023-06-18T10:08:19.015439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train","metadata":{}},{"cell_type":"code","source":"unet_result = unet_model.fit(\n    images_train, masks_train, \n    validation_split = 0.2, batch_size = 4, epochs = 120)","metadata":{"execution":{"iopub.status.busy":"2023-06-18T10:08:19.017457Z","iopub.execute_input":"2023-06-18T10:08:19.018020Z","iopub.status.idle":"2023-06-18T10:11:46.964324Z","shell.execute_reply.started":"2023-06-18T10:08:19.017988Z","shell.execute_reply":"2023-06-18T10:11:46.963095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict","metadata":{}},{"cell_type":"code","source":"unet_predict = unet_model.predict(images_test)","metadata":{"execution":{"iopub.status.busy":"2023-06-18T10:11:46.976565Z","iopub.execute_input":"2023-06-18T10:11:46.976972Z","iopub.status.idle":"2023-06-18T10:11:47.740200Z","shell.execute_reply.started":"2023-06-18T10:11:46.976931Z","shell.execute_reply":"2023-06-18T10:11:47.738968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unet_predict1 = (unet_predict > 0.3).astype(np.uint8)\nunet_predict2 = (unet_predict > 0.4).astype(np.uint8)\nunet_predict3 = (unet_predict > 0.5).astype(np.uint8)\nunet_predict4 = (unet_predict > 0.6).astype(np.uint8)","metadata":{"execution":{"iopub.status.busy":"2023-06-18T10:11:47.741860Z","iopub.execute_input":"2023-06-18T10:11:47.742268Z","iopub.status.idle":"2023-06-18T10:11:47.750499Z","shell.execute_reply.started":"2023-06-18T10:11:47.742205Z","shell.execute_reply":"2023-06-18T10:11:47.749325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_result(og, unet, target, p):\n    \n    fig, axs = plt.subplots(1, 3, figsize=(12,12))\n    axs[0].set_title(\"Original\")\n    axs[0].imshow(og)\n    axs[0].axis('off')\n    \n    axs[1].set_title(\"U-Net: p>\"+str(p))\n    axs[1].imshow(unet)\n    axs[1].axis('off')\n    \n    axs[2].set_title(\"Ground Truth\")\n    axs[2].imshow(target)\n    axs[2].axis('off')\n\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-18T10:11:47.764639Z","iopub.execute_input":"2023-06-18T10:11:47.765150Z","iopub.status.idle":"2023-06-18T10:11:47.773633Z","shell.execute_reply.started":"2023-06-18T10:11:47.765110Z","shell.execute_reply":"2023-06-18T10:11:47.772199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_test_idx = random.sample(range(len(unet_predict)), 3)\nfor idx in show_test_idx: \n    show_result(images_test[idx], unet_predict1[idx], masks_test[idx], 0.1)\n    show_result(images_test[idx], unet_predict2[idx], masks_test[idx], 0.2)\n    show_result(images_test[idx], unet_predict3[idx], masks_test[idx], 0.3)\n    show_result(images_test[idx], unet_predict4[idx], masks_test[idx], 0.4)","metadata":{"execution":{"iopub.status.busy":"2023-06-18T10:11:47.775510Z","iopub.execute_input":"2023-06-18T10:11:47.775945Z","iopub.status.idle":"2023-06-18T10:11:48.581361Z","shell.execute_reply.started":"2023-06-18T10:11:47.775908Z","shell.execute_reply":"2023-06-18T10:11:48.580277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}