{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <center>HuBMAP Image Cutting<br/><br/>Demonstrative Notebook</center>\n\n\nThis notebook demonstrates:\n- Reducing image size by cutting it into parts;\n- Increase training/validation dataset size;\n- Use 1D arrays as core model input/output.\n\nTo-Do:\n- Improve the model (overfitting);\n- Train the model.\n\n<b style='color:red;'>Note</b>: This notebook is a demonstrative one. To have a result:\n- set MAX_IDS_COUNT constant to negative value to load all training images;\n- turn on GPU;\n- tune the AI model structure ant raining;\n- tune BW_IMAGE_THRESHOLD constant;\n- tune MIN_REGIONS_POINTS constant;\n- update the code to use presisions for predictions (To-Do). \n\n<a id='toc'></a>\n# Table of Contents\n<div style='border-width: 2px;\n              border-bottom-width:4px;\n              border-bottom-color:#00FF00;\n              border-bottom-style: solid;'></div>\n\n\n- [1 | Prerequisites](#1)\n\n- [2 | Load Data](#2)\n\n- [3 | AI Model](#3)\n\n- [4 | Predictions](#4)\n\n\n<a id='1'></a>\n# 1 | Prerequisites\n<div style='border-width: 2px;\n              border-bottom-width:4px;\n              border-bottom-color:#00FF00;\n              border-bottom-style: solid;'></div>","metadata":{}},{"cell_type":"code","source":"# Needs for the encoding results\n!pip install --no-index --no-deps /kaggle/input/pycocotools-206/wheels/*.whl","metadata":{"execution":{"iopub.status.busy":"2023-07-21T14:42:02.711812Z","iopub.execute_input":"2023-07-21T14:42:02.712337Z","iopub.status.idle":"2023-07-21T14:42:14.473180Z","shell.execute_reply.started":"2023-07-21T14:42:02.712296Z","shell.execute_reply":"2023-07-21T14:42:14.471124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Imports necessary libraries\nimport os\nimport glob\nimport json\n\nimport cv2\nimport matplotlib.pyplot as plt\nfrom matplotlib import rcParams\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport random\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, BatchNormalization\nfrom tensorflow.keras.optimizers import Adam\nimport keras.models as kmodels\n\nimport base64\nfrom pycocotools import _mask as coco_mask\nimport typing as t\nimport zlib\n\n\n# Set random seed to have stable results for runnings\nrseed = 55\nrandom.seed(rseed)\nnp.random.seed(rseed)\ntf.random.set_seed(rseed)\nos.environ['PYTHONHASHSEED'] = str(rseed)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-21T14:42:57.322397Z","iopub.execute_input":"2023-07-21T14:42:57.322937Z","iopub.status.idle":"2023-07-21T14:43:07.676876Z","shell.execute_reply.started":"2023-07-21T14:42:57.322897Z","shell.execute_reply":"2023-07-21T14:43:07.675591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Hom namy parts by X and Y (Total IMAGE_PIECES x IMAGE_PIECES pieces)\nIMAGE_PIECES = 8\n\n# Each 512 x 512 image breaks into 8 * 8 = 64 images with size 64 * 64\n# Reduce input/output size of the model allowing increase of internal structures\n# Remaining the same number of params (model size in RAM)\n# Number of training/validation dataset items is increased in 64\n\n# How many IDs (pictures) will be used in training. Set it to negative to load all images\nMAX_IDS_COUNT = 5","metadata":{"execution":{"iopub.status.busy":"2023-07-21T14:43:14.598683Z","iopub.execute_input":"2023-07-21T14:43:14.599603Z","iopub.status.idle":"2023-07-21T14:43:14.606938Z","shell.execute_reply.started":"2023-07-21T14:43:14.599561Z","shell.execute_reply":"2023-07-21T14:43:14.605339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Path constants\nPATH_ANNOTATION = '/kaggle/input/hubmap-hacking-the-human-vasculature/polygons.jsonl'\nPATH_TRAIN = '/kaggle/input/hubmap-hacking-the-human-vasculature/train/'\nPATH_TEST = '/kaggle/input/hubmap-hacking-the-human-vasculature/test/'","metadata":{"execution":{"iopub.status.busy":"2023-07-21T15:16:08.955946Z","iopub.execute_input":"2023-07-21T15:16:08.957855Z","iopub.status.idle":"2023-07-21T15:16:08.964628Z","shell.execute_reply.started":"2023-07-21T15:16:08.957803Z","shell.execute_reply":"2023-07-21T15:16:08.963378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Procedures\n\ndef load_ann_regions(maxr):\n    \"\"\"\n    Load maxr annotation regions (for debug).\n    If maxr < 0, all records will be loaded (final).\n    \"\"\"\n    reg_ann_df = pd.DataFrame({'iid':[], 'type':[], 'coord':[]})\n    ids = []\n    i = 0\n    last_id = None\n    with open(PATH_ANNOTATION, 'r') as f:\n        for line in f:\n            antn = json.loads(line)\n            img_id = antn['id']\n            img_antn = antn['annotations']\n            \n            if(last_id != img_id):\n                ids.append(img_id)\n                i += 1\n                last_id = img_id\n            \n            if(i == maxr):\n                break\n                \n            for ant in img_antn:\n                tp = ant['type']\n                crd = ant['coordinates']\n                reg_ann_df.loc[len(reg_ann_df.index)] = np.array([img_id, tp, crd], dtype = object)\n    return (ids, reg_ann_df)\n    \n\ndef get_image_pieces_img(path, reg_df, img_id):\n    \"\"\"\n    Loads image by img_id, creates mask, and cut them.\n    \"\"\"\n    img = Image.open(path + img_id + '.tif') #.convert('L')\n        \n    imgs = []\n    sz = 512 / IMAGE_PIECES\n    for ix in range(IMAGE_PIECES):\n        for iy in range(IMAGE_PIECES):\n            bx = ix * sz\n            by = iy * sz\n            box = (bx, by, bx + sz, by + sz)\n            imgs.append(img.crop(box))\n    \n    return imgs\n\n\ndef get_image_pieces_msk(reg_df, img_id):\n    \"\"\"\n    Loads annotation regions by img_id, creates mask, and cut them.\n    \"\"\"\n    msk = np.zeros((512,512), dtype = np.uint8)\n    mdf = reg_df[reg_df.iid == img_id]\n    for index, row in mdf.iterrows():\n        crd = np.array(row['coord'])\n        crd = crd.reshape(-1, 1, 2)\n        # Draw the polygon on the mask\n        cv2.fillPoly(msk, [crd], 255)\n        \n    msk_img = Image.fromarray(msk.astype(np.uint8))\n    msks = []\n    sz = 512 / IMAGE_PIECES\n    for ix in range(IMAGE_PIECES):\n        for iy in range(IMAGE_PIECES):\n            bx = ix * sz\n            by = iy * sz\n            box = (bx, by, bx + sz, by + sz)\n            msks.append(msk_img.crop(box))\n            \n    return msks\n\n\ndef img_2_array(img):\n    \"\"\"\n    Converts color image into 1D float array conaining grayscale pixels.\n    \"\"\"\n    ar_img_out = []\n    ar_img = np.asarray(img)\n    \n    for row in ar_img:\n        for pxl in row:\n            p = 0.2989 * pxl[0] + 0.5870 * pxl[1] + 0.1140 * pxl[2]\n            ar_img_out.append(float(p / 255.0))\n\n    return ar_img_out\n\n\ndef msk_2_array(img):\n    \"\"\"\n    Converts mask image to array.\n    \"\"\"\n    ar_img_out = []\n    ar_img = np.asarray(img)\n    \n    for row in ar_img:\n        for pxl in row:\n            ar_img_out.append(pxl/255.0)\n\n    return ar_img_out\n\n\n# Create training dataset for X and Y separately reduces RAM use\n# You can combine both procedures to one if you have more than 32 Gb RAM\n\ndef prepare_training_df_x(path, reg_df, ids):\n    \"\"\"\n    Create x part of training dataset by list of IDs\n    \"\"\"\n    ax = []\n    \n    ii = 0\n    sz = len(ids)\n    \n    for id in ids:\n        ips = get_image_pieces_img(path, reg_df, id)    \n        for i in range(len(ips)):\n            arx = img_2_array(ips[i])\n            ax.append(arx)\n            \n        print ('Record:', ii, 'of', sz, 'done       ', end = \"\\r\")\n        ii += 1\n\n    print ('Records:', sz, 'done                                      ')\n    return np.array(ax, dtype = 'float32')\n\n\ndef prepare_training_df_y(reg_df, ids):\n    \"\"\"\n    Create y part of training dataset by list of IDs\n    \"\"\"\n    ay = []\n    \n    ii = 0\n    sz = len(ids)\n    \n    for id in ids:\n        ips = get_image_pieces_msk(reg_df, id)    \n        for i in range(len(ips)):\n            ary = msk_2_array(ips[i])\n            ay.append(ary)\n            \n        print ('Record:', ii, 'of', sz, 'done       ', end = \"\\r\")\n        ii += 1\n\n    print ('Records:', sz, 'done                                      ')\n    return np.array(ay, dtype = 'float32')","metadata":{"execution":{"iopub.status.busy":"2023-07-21T15:25:58.449951Z","iopub.execute_input":"2023-07-21T15:25:58.450568Z","iopub.status.idle":"2023-07-21T15:25:58.478772Z","shell.execute_reply.started":"2023-07-21T15:25:58.450525Z","shell.execute_reply":"2023-07-21T15:25:58.477284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[↑↑↑ Table of Contents ↑↑↑](#toc)\n\n\n\n<a id='2'></a>\n# 2 | Load Data\n<div style='border-width: 2px;\n              border-bottom-width:4px;\n              border-bottom-color:#00FF00;\n              border-bottom-style: solid;'></div>              ","metadata":{}},{"cell_type":"code","source":"# Load Annotation Regions\nids, reg_ann_df = load_ann_regions(MAX_IDS_COUNT)\n\n# Safe Annotation Regions if need\n#reg_ann_df.to_csv('ann_regions.csv', index = False)\n\nreg_ann_df","metadata":{"execution":{"iopub.status.busy":"2023-07-21T15:20:44.775913Z","iopub.execute_input":"2023-07-21T15:20:44.777752Z","iopub.status.idle":"2023-07-21T15:20:45.760323Z","shell.execute_reply.started":"2023-07-21T15:20:44.777696Z","shell.execute_reply":"2023-07-21T15:20:45.759309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# See one part example\nimg_demo = get_image_pieces_img(PATH_TRAIN, reg_ann_df, ids[0])\nimg_demo[2]","metadata":{"execution":{"iopub.status.busy":"2023-07-21T15:25:05.407628Z","iopub.execute_input":"2023-07-21T15:25:05.408110Z","iopub.status.idle":"2023-07-21T15:25:05.443287Z","shell.execute_reply.started":"2023-07-21T15:25:05.408078Z","shell.execute_reply":"2023-07-21T15:25:05.441734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# See the mask of the image\nmsk_demo = get_image_pieces_msk(reg_ann_df, ids[0])\nmsk_demo[2]","metadata":{"execution":{"iopub.status.busy":"2023-07-21T15:26:04.938733Z","iopub.execute_input":"2023-07-21T15:26:04.939228Z","iopub.status.idle":"2023-07-21T15:26:04.965953Z","shell.execute_reply.started":"2023-07-21T15:26:04.939193Z","shell.execute_reply":"2023-07-21T15:26:04.964789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare x training set\nax = prepare_training_df_x(PATH_TRAIN, reg_ann_df, ids)\n# Save it if need\n#np.save('tr_ax', ax)\nprint(ax.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-21T15:26:32.539819Z","iopub.execute_input":"2023-07-21T15:26:32.540266Z","iopub.status.idle":"2023-07-21T15:26:46.979959Z","shell.execute_reply.started":"2023-07-21T15:26:32.540232Z","shell.execute_reply":"2023-07-21T15:26:46.978703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare x training set\nay = prepare_training_df_y(reg_ann_df, ids)\n# Save it if need\n# np.save(path_tmp_base + 'tr_ay', ay)\nprint(ay.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-21T15:27:20.787176Z","iopub.execute_input":"2023-07-21T15:27:20.787632Z","iopub.status.idle":"2023-07-21T15:27:25.507584Z","shell.execute_reply.started":"2023-07-21T15:27:20.787596Z","shell.execute_reply":"2023-07-21T15:27:25.505961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[↑↑↑ Table of Contents ↑↑↑](#toc)\n\n\n\n<a id='3'></a>\n# 3 | AI Model\n<div style='border-width: 2px;\n              border-bottom-width:4px;\n              border-bottom-color:#00FF00;\n              border-bottom-style: solid;'></div>      ","metadata":{}},{"cell_type":"code","source":"# Create AI model\n\ndef create_model():\n    \"\"\"\n    Create AI model\n    \"\"\"\n    random.seed(123456)\n    model = Sequential()\n    model.add(Dense(64 * 64, input_dim = 64 * 64, activation = \"relu\"))\n    model.add(BatchNormalization())\n    model.add(Dropout(0.1))\n    model.add(Dense(64 * 64, activation = \"relu\"))\n    model.add(Dropout(0.1))\n    model.add(BatchNormalization())\n    model.add(Dense(64 * 64, activation = \"relu\"))\n    model.add(Dropout(0.1))\n    model.add(BatchNormalization())\n    model.add(Dense(64 * 64, activation = \"relu\"))\n    model.add(Dropout(0.1))\n    model.add(BatchNormalization())\n    model.add(Dense(64 * 64, activation=\"sigmoid\"))\n\n    model.compile(\n        optimizer = Adam(learning_rate = 1e-5),\n        loss = 'mean_squared_error',\n        metrics=[\n            tf.keras.metrics.BinaryAccuracy(name='accuracy')\n        ]\n    )\n        \n    return model\n\nmodel = create_model()\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-21T15:28:40.178143Z","iopub.execute_input":"2023-07-21T15:28:40.179833Z","iopub.status.idle":"2023-07-21T15:28:41.349222Z","shell.execute_reply.started":"2023-07-21T15:28:40.179775Z","shell.execute_reply":"2023-07-21T15:28:41.345527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ax2 = np.load(path_tmp_base + 'tr_ax.npy')\n#ay2 = np.load(path_tmp_base + 'tr_ay.npy')\n\n#ax2 = ax2.reshape(ax2.shape)\n#ay2 = ay2.reshape(ay2.shape)\n\n#np.save(path_tmp_base + 'tr_ax2', ax2)\n#np.save(path_tmp_base + 'tr_ay2', ay2)\n\n#print(ax2.shape)\n#print(ay2.shape)\n#print(ax2.dtype)\n#print(ay2.dtype)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model\n\nEPOCHS = 20\n\ndef lr_scheduler(epoch, lr):\n    if epoch < 200:\n        return 1e-5\n    else:\n        return 5e-6\n\n\nhistory = model.fit(\n    ax,\n    ay,\n    batch_size = 16,\n    epochs = EPOCHS,\n    verbose = 1,\n    # callbacks = [earlyStopping, mcp_save, reduce_lr_loss],\n    validation_split = 0.25,\n    \n    callbacks=[\n        tf.keras.callbacks.LearningRateScheduler(\n            # lambda epoch: 1e-5 * 10 ** (epoch / 30)\n            # lambda epoch: 1e-3 - (epoch * 1e-5)\n            lr_scheduler\n        )\n    ]\n)","metadata":{"execution":{"iopub.status.busy":"2023-07-21T15:29:29.871586Z","iopub.execute_input":"2023-07-21T15:29:29.872317Z","iopub.status.idle":"2023-07-21T15:35:51.866150Z","shell.execute_reply.started":"2023-07-21T15:29:29.872279Z","shell.execute_reply":"2023-07-21T15:35:51.864842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Some diagrams useful for tunint the model\n\nrcParams['figure.figsize'] = (18, 8)\nrcParams['axes.spines.top'] = False\nrcParams['axes.spines.right'] = False \n\nplt.plot(\n    np.arange(1, EPOCHS + 1),\n    history.history['loss'], \n    label='Loss', lw=3\n)\nplt.plot(\n    np.arange(1, EPOCHS + 1),\n    history.history['accuracy'], \n    label='Accuracy', lw=3\n)\nplt.plot(\n    np.arange(1, EPOCHS + 1),\n    history.history['lr'], \n    label='Learning rate', color='#000', lw=3, linestyle='--'\n)\nplt.title('Evaluation metrics', size=20)\nplt.xlabel('Epoch', size=14)\nplt.legend();","metadata":{"execution":{"iopub.status.busy":"2023-07-21T15:42:41.047938Z","iopub.execute_input":"2023-07-21T15:42:41.049657Z","iopub.status.idle":"2023-07-21T15:42:41.820808Z","shell.execute_reply.started":"2023-07-21T15:42:41.049588Z","shell.execute_reply":"2023-07-21T15:42:41.819962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Some diagrams useful for tunint the model\n\nplt.plot(history.history['accuracy'], label='Accuracy')\nplt.plot(history.history['val_accuracy'], label='Val accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-21T15:43:16.505107Z","iopub.execute_input":"2023-07-21T15:43:16.505577Z","iopub.status.idle":"2023-07-21T15:43:16.922359Z","shell.execute_reply.started":"2023-07-21T15:43:16.505541Z","shell.execute_reply":"2023-07-21T15:43:16.921080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Some diagrams useful for tunint the model\n\nplt.plot(history.history['loss'], label='Loss')\nplt.plot(history.history['val_loss'], label='Val loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-21T15:43:37.716315Z","iopub.execute_input":"2023-07-21T15:43:37.716766Z","iopub.status.idle":"2023-07-21T15:43:38.123260Z","shell.execute_reply.started":"2023-07-21T15:43:37.716733Z","shell.execute_reply":"2023-07-21T15:43:38.121527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the model if need\n# model.save(path_model_base + 'mdl.keras')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[↑↑↑ Table of Contents ↑↑↑](#toc)\n\n\n\n<a id='4'></a>\n# 4 | Predictions\n<div style='border-width: 2px;\n              border-bottom-width:4px;\n              border-bottom-color:#00FF00;\n              border-bottom-style: solid;'></div>    ","metadata":{}},{"cell_type":"code","source":"# Load Testing Data and divide images into pieces\n\ndef get_image_pieces(img):\n    \"\"\"\n    Divides the image into pieces\n    \"\"\"\n    \n    imgs = []\n    sz = 512 / IMAGE_PIECES\n    for ix in range(IMAGE_PIECES):\n        for iy in range(IMAGE_PIECES):\n            bx = ix * sz\n            by = iy * sz\n            box = (bx, by, bx + sz, by + sz)\n            imp = img.crop(box)\n            imgs.append(imp)\n    \n    return imgs\n\n\ndef get_image_pieces_from_path():\n    \"\"\"\n    Loads images, divide them into pieces, and return them and their IDs\n    \"\"\"\n\n    imgs = []\n    img_ids = []\n    fllist = os.listdir(PATH_TEST)\n    \n    for fn in fllist:\n        img_id = fn[:12]\n        img_ids.append(img_id)\n        img = Image.open(PATH_TEST + img_id + '.tif')\n        imgs_ar = get_image_pieces(img)\n        imgs.extend(imgs_ar)\n    \n    return imgs,img_ids\n\nimgs_test,img_ids_test = get_image_pieces_from_path()\n\nprint(len(imgs_test))\nprint(len(img_ids_test))","metadata":{"execution":{"iopub.status.busy":"2023-07-21T15:44:30.528863Z","iopub.execute_input":"2023-07-21T15:44:30.529390Z","iopub.status.idle":"2023-07-21T15:44:30.581458Z","shell.execute_reply.started":"2023-07-21T15:44:30.529350Z","shell.execute_reply":"2023-07-21T15:44:30.579886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make predictions\n\ndef img_2_array(img):\n    \"\"\"\n    Converts color image to 1D array corresponding to grayscale float base\n    \"\"\"\n    \n    ar_img_out = []\n    ar_img = np.asarray(img)\n    \n    for row in ar_img:\n        for pxl in row:\n            p = 0.2989 * pxl[0] + 0.5870 * pxl[1] + 0.1140 * pxl[2]\n            ar_img_out.append(float(p / 255.0))\n\n    return ar_img_out\n\n\nimgs_ar_test = []\n\nfor img in imgs_test:\n    imgs_ar_test.append(img_2_array(img))\n\npreds_test = model.predict(imgs_ar_test)\n\nlen(preds_test)","metadata":{"execution":{"iopub.status.busy":"2023-07-21T15:44:44.275969Z","iopub.execute_input":"2023-07-21T15:44:44.276517Z","iopub.status.idle":"2023-07-21T15:44:49.344533Z","shell.execute_reply.started":"2023-07-21T15:44:44.276480Z","shell.execute_reply":"2023-07-21T15:44:49.343497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combine predicted parts into original size\n\nBW_IMAGE_THRESHOLD = 0.75\n\n\ndef calc_monochrome_img_ar(ar, threshold):\n    \"\"\"\n    Creates a monochrome image array from an array\n    \"\"\"\n    \n    ar2 = np.copy(ar)\n    for i in range(len(ar2)): \n        if(ar2[i] < threshold):\n            ar2[i] = 0\n        else:\n            ar2[i] = 1\n    return ar2\n\n\nsz = int(512 / IMAGE_PIECES)\nmsks_preds = []\nmasks_count = int(len(preds_test) / (IMAGE_PIECES * IMAGE_PIECES))\n\nprint(masks_count)\n\nipart = 0\n\nfor imsk in range(masks_count):\n    img  = Image.new( mode = 'L', size = (512, 512) )\n    for ix in range(IMAGE_PIECES):\n        for iy in range(IMAGE_PIECES):\n            img_ar = calc_monochrome_img_ar(preds_test[ipart], BW_IMAGE_THRESHOLD)\n            img_ar = (img_ar * 255).astype(np.uint8)\n            img_ar = np.reshape(img_ar, (sz, sz))\n            img_part = Image.fromarray(img_ar, mode='L')                \n            img.paste(img_part, (ix * sz, iy * sz))\n            ipart += 1\n                \n    msks_preds.append(img)\n\nlen(msks_preds)","metadata":{"execution":{"iopub.status.busy":"2023-07-21T15:45:05.549531Z","iopub.execute_input":"2023-07-21T15:45:05.550050Z","iopub.status.idle":"2023-07-21T15:45:06.472044Z","shell.execute_reply.started":"2023-07-21T15:45:05.550014Z","shell.execute_reply":"2023-07-21T15:45:06.470579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Look at 1st predicted image\nmsks_preds[0]","metadata":{"execution":{"iopub.status.busy":"2023-07-21T15:45:10.592045Z","iopub.execute_input":"2023-07-21T15:45:10.592600Z","iopub.status.idle":"2023-07-21T15:45:10.609192Z","shell.execute_reply.started":"2023-07-21T15:45:10.592555Z","shell.execute_reply":"2023-07-21T15:45:10.607720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Regions containing less points will not be considered\nMIN_REGIONS_POINTS = 20\n\n# Discover regions in predicted images\n\ndef get_region_coordinates(ar_img):\n    \"\"\"\n    Returns set of pixels of the 1st found region.\n    \"\"\"\n    \n    rcrds = set()\n\n    for ix in range(512):\n        for iy in range(512):\n            if(ar_img[ix, iy] == 255):\n                rcrds.add((ix, iy))\n                break\n        if(len(rcrds) > 0):\n            break\n\n    b_new_pixel = (len(rcrds) > 0)\n\n    while b_new_pixel:\n        b_new_pixel = False\n\n        new_pixels = []\n    \n        for crd in rcrds:\n\n            # Left, Left-Up, Left-Down\n            if(crd[0] - 1 >= 0):\n            \n                # Left\n                px_l = ar_img[(crd[0] - 1, crd[1])]\n                if(px_l == 255):\n                    new_pixels.append((crd[0] - 1, crd[1]))\n        \n                # Left-Up\n                if(crd[1] - 1 >= 0):\n                    px_l = ar_img[(crd[0] - 1, crd[1] - 1)]\n                    if(px_l == 255):\n                        new_pixels.append((crd[0] - 1, crd[1] - 1))\n                    \n                # Left-Down\n                if(crd[1] + 1 < 512):\n                    px_l = ar_img[(crd[0] - 1, crd[1] + 1)]\n                    if(px_l == 255):\n                        new_pixels.append((crd[0] - 1, crd[1] + 1))\n        \n        \n            # Right, Right-Up, Right-Down\n            if(crd[0] + 1 < 512):\n            \n                # Left\n                px_l = ar_img[(crd[0] + 1, crd[1])]\n                if(px_l == 255):\n                    new_pixels.append((crd[0] + 1, crd[1]))\n        \n                # Left-Up\n                if(crd[1] - 1 >= 0):\n                    px_l = ar_img[(crd[0] + 1, crd[1] - 1)]\n                    if(px_l == 255):\n                        new_pixels.append((crd[0] + 1, crd[1] - 1))\n                    \n                # Left-Down\n                if(crd[1] + 1 < 512):\n                    px_l = ar_img[(crd[0] + 1, crd[1] + 1)]\n                    if(px_l == 255):\n                        new_pixels.append((crd[0] + 1, crd[1] + 1))\n                    \n            # Up\n            if(crd[1] - 1 >= 0):\n                px_l = ar_img[(crd[0], crd[1] - 1)]\n                if(px_l == 255):\n                    new_pixels.append((crd[0], crd[1] - 1))\n        \n            # Down\n            if(crd[1] + 1 < 512):\n                px_l = ar_img[(crd[0], crd[1] + 1)]\n                if(px_l == 255):\n                    new_pixels.append((crd[0], crd[1] + 1))\n\n        l1 = len(rcrds)\n        rcrds.update(new_pixels)\n        l2 = len(rcrds)\n\n        b_new_pixel = ((l2 - l1) > 0)\n\n    return rcrds\n    \n\ndef get_regions(msk):\n    \"\"\"\n    Returns regions with size > min_reg.\n    \"\"\"\n\n    ar_img = np.asarray(msk).copy()\n    ar_img.setflags(write = 1)\n\n    regions = []\n    \n    while True:\n        reg = get_region_coordinates(ar_img)\n        \n        if (len(reg) == 0):\n            break\n    \n    \n        for crd in reg:\n            ar_img[crd[0]][crd[1]] = 0\n    \n        if (MIN_REGIONS_POINTS <= len(reg)):\n            regions.append(reg)\n    \n    return regions\n\n\nregion_coordinates = []\n\nfor i in range(len(img_ids_test)):\n    region_coordinates.append(get_regions(msks_preds[i]))\n    \nlen(region_coordinates)","metadata":{"execution":{"iopub.status.busy":"2023-07-21T15:45:16.921481Z","iopub.execute_input":"2023-07-21T15:45:16.921942Z","iopub.status.idle":"2023-07-21T15:55:05.851340Z","shell.execute_reply.started":"2023-07-21T15:45:16.921908Z","shell.execute_reply":"2023-07-21T15:55:05.850169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Look at region coordinates\nlen(region_coordinates[0])","metadata":{"execution":{"iopub.status.busy":"2023-07-21T15:56:24.191834Z","iopub.execute_input":"2023-07-21T15:56:24.192356Z","iopub.status.idle":"2023-07-21T15:56:24.201917Z","shell.execute_reply.started":"2023-07-21T15:56:24.192321Z","shell.execute_reply":"2023-07-21T15:56:24.200082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encode result for submission\n\ndef encode_binary_mask(mask: np.ndarray) -> t.Text:\n  \"\"\"Converts a binary mask into OID challenge encoding ascii text.\"\"\"\n\n  # check input mask --\n  if mask.dtype != bool:\n    raise ValueError(\n        \"encode_binary_mask expects a binary mask, received dtype == %s\" %\n        mask.dtype)\n\n  mask = np.squeeze(mask)\n  if len(mask.shape) != 2:\n    raise ValueError(\n        \"encode_binary_mask expects a 2d mask, received shape == %s\" %\n        mask.shape)\n\n  # convert input mask to expected COCO API input --\n  mask_to_encode = mask.reshape(mask.shape[0], mask.shape[1], 1)\n  mask_to_encode = mask_to_encode.astype(np.uint8)\n  mask_to_encode = np.asfortranarray(mask_to_encode)\n\n  # RLE encode mask --\n  encoded_mask = coco_mask.encode(mask_to_encode)[0][\"counts\"]\n\n  # compress and base64 encoding --\n  binary_str = zlib.compress(encoded_mask, zlib.Z_BEST_COMPRESSION)\n  base64_str = base64.b64encode(binary_str)\n  return base64_str\n\n\nsubmission = pd.DataFrame(data = [], columns = ['id','height','width','prediction_string'])\n\nfor i in range(len(img_ids_test)):\n\n    p_string = ''\n    \n    for region in region_coordinates[i]:\n\n        ar = np.zeros(shape = (512, 512), dtype = bool)\n\n        for crd in region:\n            ar[crd[0],crd[1]] = True\n        \n        code = encode_binary_mask(ar)\n\n        if(len(p_string) > 0):\n            p_string += ' '\n            \n        p_string = p_string + '0 0.8 ' + str(code)[2:-1]\n\n    submission.loc[i, 'id'] = img_ids_test[i]\n    submission.loc[i, 'height'] = 512\n    submission.loc[i, 'width'] = 512\n    submission.loc[i, 'prediction_string'] = p_string\n        \nsubmission","metadata":{"execution":{"iopub.status.busy":"2023-07-21T15:58:13.272787Z","iopub.execute_input":"2023-07-21T15:58:13.273354Z","iopub.status.idle":"2023-07-21T15:58:13.320186Z","shell.execute_reply.started":"2023-07-21T15:58:13.273317Z","shell.execute_reply":"2023-07-21T15:58:13.318774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save submission\nsubmission.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2023-07-21T15:58:17.968060Z","iopub.execute_input":"2023-07-21T15:58:17.968591Z","iopub.status.idle":"2023-07-21T15:58:17.980325Z","shell.execute_reply.started":"2023-07-21T15:58:17.968553Z","shell.execute_reply":"2023-07-21T15:58:17.978964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[↑↑↑ Table of Contents ↑↑↑](#toc)","metadata":{}}]}