{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":30201,"databundleVersionId":2750748,"sourceType":"competition"}],"dockerImageVersionId":30152,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport os\nimport cv2 \nimport tensorflow as tf\n\nfrom keras.models import Model, load_model\nfrom keras.layers import Input\nfrom keras.layers.core import Dropout, Lambda\nfrom keras.layers.convolutional import Conv2D, Conv2DTranspose\nfrom keras.layers.pooling import MaxPooling2D\nfrom keras.layers.merge import concatenate\nfrom keras.callbacks import EarlyStopping, ModelCheckpoint","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-08T23:04:59.249066Z","iopub.execute_input":"2024-11-08T23:04:59.250039Z","iopub.status.idle":"2024-11-08T23:05:07.613886Z","shell.execute_reply.started":"2024-11-08T23:04:59.249875Z","shell.execute_reply":"2024-11-08T23:05:07.612873Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"files = 0\nimages = 0\nannotations = 0\n\nfor num, (dirname, _, filenames,) in enumerate(os.walk('/kaggle/input')):\n    for file, filename in enumerate(filenames):\n        if file==2:\n            print(dirname, \"  ... many on this folder\")\n        if filename.endswith((\"xlsx\", \"txt\", \"csv\")):\n            files+=1\n            print(os.path.join(dirname, filename))\n        elif filename.endswith((\"png\", \"jpeg\", \"jpg\",)):\n            images+=1\n        else:\n            annotations+=1\nprint(\"...\\n\")\nprint(\"#\"*10, \"        files: {} images: {} annotations: {}\".format(files, images, annotations))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-08T23:05:07.615576Z","iopub.execute_input":"2024-11-08T23:05:07.615880Z","iopub.status.idle":"2024-11-08T23:05:15.361875Z","shell.execute_reply.started":"2024-11-08T23:05:07.615846Z","shell.execute_reply":"2024-11-08T23:05:15.360876Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Original unet taken from https://www.kaggle.com/stpeteishii/cell-instance-segmentation-unet","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:15.363314Z","iopub.execute_input":"2024-11-08T23:05:15.363645Z","iopub.status.idle":"2024-11-08T23:05:15.369215Z","shell.execute_reply.started":"2024-11-08T23:05:15.363600Z","shell.execute_reply":"2024-11-08T23:05:15.368046Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":" # data understanding","metadata":{}},{"cell_type":"code","source":"# read data first. The csv is the core.\ntrain_data = pd.read_csv('../input/sartorius-cell-instance-segmentation/train.csv')\nprint(train_data.shape)\ntrain_data.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:15.371672Z","iopub.execute_input":"2024-11-08T23:05:15.371966Z","iopub.status.idle":"2024-11-08T23:05:16.055614Z","shell.execute_reply.started":"2024-11-08T23:05:15.371932Z","shell.execute_reply":"2024-11-08T23:05:16.054359Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submision sample\nsample_submission=pd.read_csv('../input/sartorius-cell-instance-segmentation/sample_submission.csv')\nprint(sample_submission.shape)\nsample_submission.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:16.057042Z","iopub.execute_input":"2024-11-08T23:05:16.057303Z","iopub.status.idle":"2024-11-08T23:05:16.080695Z","shell.execute_reply.started":"2024-11-08T23:05:16.057269Z","shell.execute_reply":"2024-11-08T23:05:16.079643Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# we will focus on the train folder, as it contains the images to process. The others not for now.\nprint(\"Number of images in the folder train:\")\nlen(os.listdir('../input/sartorius-cell-instance-segmentation/train'))","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:16.082079Z","iopub.execute_input":"2024-11-08T23:05:16.082326Z","iopub.status.idle":"2024-11-08T23:05:16.091487Z","shell.execute_reply.started":"2024-11-08T23:05:16.082288Z","shell.execute_reply":"2024-11-08T23:05:16.090486Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Number of unique id on the file:\")\ntrain_data.id.unique().shape","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:16.093003Z","iopub.execute_input":"2024-11-08T23:05:16.093280Z","iopub.status.idle":"2024-11-08T23:05:16.116912Z","shell.execute_reply.started":"2024-11-08T23:05:16.093248Z","shell.execute_reply":"2024-11-08T23:05:16.115939Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The id column contains all the train images codes. But each id column is repetead an have several notations. Each of the have diferent counts","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(20,6))\ntrain_data.groupby(\"id\").size().plot.bar();\nplt.xticks([])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:16.118307Z","iopub.execute_input":"2024-11-08T23:05:16.118634Z","iopub.status.idle":"2024-11-08T23:05:18.420391Z","shell.execute_reply.started":"2024-11-08T23:05:16.118598Z","shell.execute_reply":"2024-11-08T23:05:18.419191Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_data[train_data[\"id\"]==\"0030fd0e6378\"][\"id\"].count())\nprint(train_data[train_data[\"id\"]==\"0140b3c8f445\"][\"id\"].count())","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:18.421995Z","iopub.execute_input":"2024-11-08T23:05:18.422323Z","iopub.status.idle":"2024-11-08T23:05:18.460339Z","shell.execute_reply.started":"2024-11-08T23:05:18.422287Z","shell.execute_reply":"2024-11-08T23:05:18.459232Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# other columns\ntrain_data.sample_id.unique().shape","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:18.464143Z","iopub.execute_input":"2024-11-08T23:05:18.464912Z","iopub.status.idle":"2024-11-08T23:05:18.488005Z","shell.execute_reply.started":"2024-11-08T23:05:18.464845Z","shell.execute_reply":"2024-11-08T23:05:18.486846Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data[train_data[\"id\"]==\"0030fd0e6378\"][\"sample_id\"].unique()","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:18.489577Z","iopub.execute_input":"2024-11-08T23:05:18.489865Z","iopub.status.idle":"2024-11-08T23:05:18.512869Z","shell.execute_reply.started":"2024-11-08T23:05:18.489831Z","shell.execute_reply":"2024-11-08T23:05:18.511864Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# sample_id is bounded to the id\ntrain_data.groupby([\"id\", \"sample_id\"]).size().count()","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:18.514417Z","iopub.execute_input":"2024-11-08T23:05:18.514734Z","iopub.status.idle":"2024-11-08T23:05:18.555435Z","shell.execute_reply.started":"2024-11-08T23:05:18.514697Z","shell.execute_reply":"2024-11-08T23:05:18.554387Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#checking for missing values. None\ntrain_data.isnull().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:18.556800Z","iopub.execute_input":"2024-11-08T23:05:18.557159Z","iopub.status.idle":"2024-11-08T23:05:18.628810Z","shell.execute_reply.started":"2024-11-08T23:05:18.557122Z","shell.execute_reply":"2024-11-08T23:05:18.627750Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\ntrain_data.groupby('cell_type').size().plot.bar()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:18.630416Z","iopub.execute_input":"2024-11-08T23:05:18.630761Z","iopub.status.idle":"2024-11-08T23:05:18.778695Z","shell.execute_reply.started":"2024-11-08T23:05:18.630716Z","shell.execute_reply":"2024-11-08T23:05:18.777746Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\ntrain_data.groupby(['width', 'height']).size().plot.bar()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:18.780457Z","iopub.execute_input":"2024-11-08T23:05:18.781020Z","iopub.status.idle":"2024-11-08T23:05:19.179474Z","shell.execute_reply.started":"2024-11-08T23:05:18.780968Z","shell.execute_reply":"2024-11-08T23:05:19.178299Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Summary of the columns:\n    - id: name of the train picture\n    - annotation: info of the target mask of the neuron cells\n    - width, heigh: width and heigh of the images (constant of 704x520)\n    - plate_time: not usefull\n    - sample_date: not usefull\n    - sample_id\n    - Elapsed_timedelta: not usefull","metadata":{}},{"cell_type":"code","source":"# check the images\nimg = cv2.imread(\"../input/sartorius-cell-instance-segmentation/train_semi_supervised/astro[hippo]_D1-1_Vessel-361_2020-09-14_13h00m00s_Ph_1.png\")\nplt.imshow(img);","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:19.181517Z","iopub.execute_input":"2024-11-08T23:05:19.181891Z","iopub.status.idle":"2024-11-08T23:05:19.479643Z","shell.execute_reply.started":"2024-11-08T23:05:19.181844Z","shell.execute_reply":"2024-11-08T23:05:19.478614Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img = cv2.imread(\"../input/sartorius-cell-instance-segmentation/train/0140b3c8f445.png\")\nplt.imshow(img);","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:19.481109Z","iopub.execute_input":"2024-11-08T23:05:19.481409Z","iopub.status.idle":"2024-11-08T23:05:19.835219Z","shell.execute_reply.started":"2024-11-08T23:05:19.481371Z","shell.execute_reply":"2024-11-08T23:05:19.834084Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# add the filters to see it better\nfrom PIL import Image, ImageEnhance\nimg = cv2.imread(\"../input/sartorius-cell-instance-segmentation/train/042c17cd9143.png\")\nimg = np.asarray(ImageEnhance.Contrast(Image.fromarray(img)).enhance(16))\n\nplt.figure()\nplt.imshow(img);","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:19.837198Z","iopub.execute_input":"2024-11-08T23:05:19.837572Z","iopub.status.idle":"2024-11-08T23:05:20.205528Z","shell.execute_reply.started":"2024-11-08T23:05:19.837519Z","shell.execute_reply":"2024-11-08T23:05:20.204498Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# sooo an image id has several anotations... but there are multiple id's repetitions with the same sample_id...\ntrain_data[train_data[\"id\"]==\"0030fd0e6378\"]","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:20.206965Z","iopub.execute_input":"2024-11-08T23:05:20.207226Z","iopub.status.idle":"2024-11-08T23:05:20.247264Z","shell.execute_reply.started":"2024-11-08T23:05:20.207195Z","shell.execute_reply":"2024-11-08T23:05:20.246243Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# how to process annotations.\n\n# read one annotation\nmask_rle = train_data[train_data[\"id\"] == \"0030fd0e6378\"][\"annotation\"].tolist()[0]\nshape=(520, 704, 3)\ns = mask_rle.split()","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:20.248610Z","iopub.execute_input":"2024-11-08T23:05:20.248966Z","iopub.status.idle":"2024-11-08T23:05:20.275352Z","shell.execute_reply.started":"2024-11-08T23:05:20.248919Z","shell.execute_reply":"2024-11-08T23:05:20.274420Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"starts = list(map(lambda x: int(x) - 1, s[0::2])) # get the starting coordinate (on the even positions)\nlengths = list(map(int, s[1::2])) # get the lenght (on the not even positions)\nends = [x + y for x, y in zip(starts, lengths)] # calculate the end point (starting point + lenght)\n\n# create a blank inage\nimg = np.zeros((shape[0] * shape[1], shape[2]), dtype=np.float32)\n\n# fill the positions on the image with a color to create the mask\nfor start, end in zip(starts, ends):\n    img[start : end] = 1\n\nplt.figure()  \nplt.imshow(img.reshape(shape));","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:20.277108Z","iopub.execute_input":"2024-11-08T23:05:20.278273Z","iopub.status.idle":"2024-11-08T23:05:20.670066Z","shell.execute_reply.started":"2024-11-08T23:05:20.278198Z","shell.execute_reply":"2024-11-08T23:05:20.669012Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(train_data[train_data[\"id\"] == \"0030fd0e6378\"][\"annotation\"].tolist())","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:20.671468Z","iopub.execute_input":"2024-11-08T23:05:20.671726Z","iopub.status.idle":"2024-11-08T23:05:20.694384Z","shell.execute_reply.started":"2024-11-08T23:05:20.671694Z","shell.execute_reply":"2024-11-08T23:05:20.693219Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reference: https://www.kaggle.com/ihelon/cell-segmentation-run-length-decoding\n#https://www.kaggle.com/susnato/understanding-run-length-encoding-and-decoding?scriptVersionId=77552323\n# coding packed into one function\n\ndef rle_decode(mask_rle, shape, color=3):     #color=1,3\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height, width, channels) of array to return \n    color: color for the mask\n    Returns numpy array (mask)\n    '''\n    s = mask_rle.split()\n    \n    starts = list(map(lambda x: int(x) - 1, s[0::2]))\n    lengths = list(map(int, s[1::2]))\n    ends = [x + y for x, y in zip(starts, lengths)]\n    \n    img = np.zeros((shape[0] * shape[1], shape[2]), dtype=np.float32)\n            \n    for start, end in zip(starts, ends):\n        img[start : end] = color\n    \n    return img.reshape(shape)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:20.695936Z","iopub.execute_input":"2024-11-08T23:05:20.696264Z","iopub.status.idle":"2024-11-08T23:05:20.707811Z","shell.execute_reply.started":"2024-11-08T23:05:20.696221Z","shell.execute_reply":"2024-11-08T23:05:20.706546Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# other version\n# https://www.kaggle.com/c/sartorius-cell-instance-segmentation/discussion/291627\ndef rle_decode(mask_rle, shape=(520, 704, 1)):\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape)  # Needed to align to RLE direction\n\ndef rle_encode(img):\n    pixels = img.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:20.709564Z","iopub.execute_input":"2024-11-08T23:05:20.710125Z","iopub.status.idle":"2024-11-08T23:05:20.722529Z","shell.execute_reply.started":"2024-11-08T23:05:20.710073Z","shell.execute_reply":"2024-11-08T23:05:20.721575Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# each mask annotation has one area\nmask = train_data[train_data[\"id\"] == \"0030fd0e6378\"][\"annotation\"].tolist()[0]\nimg = rle_decode(mask)\nplt.imshow(img, cmap=\"gray\");","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:20.723796Z","iopub.execute_input":"2024-11-08T23:05:20.724083Z","iopub.status.idle":"2024-11-08T23:05:21.057014Z","shell.execute_reply.started":"2024-11-08T23:05:20.724042Z","shell.execute_reply":"2024-11-08T23:05:21.055891Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mask","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:21.058536Z","iopub.execute_input":"2024-11-08T23:05:21.058858Z","iopub.status.idle":"2024-11-08T23:05:21.065580Z","shell.execute_reply.started":"2024-11-08T23:05:21.058820Z","shell.execute_reply":"2024-11-08T23:05:21.064506Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img.shape","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:21.067062Z","iopub.execute_input":"2024-11-08T23:05:21.067394Z","iopub.status.idle":"2024-11-08T23:05:21.080847Z","shell.execute_reply.started":"2024-11-08T23:05:21.067351Z","shell.execute_reply":"2024-11-08T23:05:21.079760Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rle_encode(img)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:21.087755Z","iopub.execute_input":"2024-11-08T23:05:21.089437Z","iopub.status.idle":"2024-11-08T23:05:21.098961Z","shell.execute_reply.started":"2024-11-08T23:05:21.089380Z","shell.execute_reply":"2024-11-08T23:05:21.097632Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# revert the encodig has a correct recosntruction?\nplt.imshow(rle_decode(rle_encode(img)));","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:21.100474Z","iopub.execute_input":"2024-11-08T23:05:21.100770Z","iopub.status.idle":"2024-11-08T23:05:21.429152Z","shell.execute_reply.started":"2024-11-08T23:05:21.100734Z","shell.execute_reply":"2024-11-08T23:05:21.428107Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# data processing","metadata":{}},{"cell_type":"code","source":"def plot_masks(image_id, colors=False):\n    labels = train_data[train_data[\"id\"] == image_id][\"annotation\"].tolist()\n\n    if colors:\n        mask = np.zeros((520, 704, 3))\n        for label in labels:\n            mask += rle_decode(label, shape=(520, 704, 3), color=np.random.rand(3))\n    else:\n        mask = np.zeros((520, 704, 1))\n        for label in labels:\n            mask += rle_decode(label, shape=(520, 704, 1))\n    mask = mask.clip(0, 1)\n\n    image = cv2.imread(f\"../input/sartorius-cell-instance-segmentation/train/{image_id}.png\")\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n    plt.figure(figsize=(18,6))\n    plt.subplot(1, 3, 1)\n    plt.imshow(image)\n    plt.title('Input image')\n    plt.axis(\"off\")\n    \n    plt.subplot(1, 3, 2)\n    plt.imshow(image)\n    plt.imshow(mask, alpha=0.1)\n    plt.title('Input image with mask')\n    plt.axis(\"off\")\n    \n    plt.subplot(1, 3, 3)\n    plt.imshow(mask)\n    plt.title('Only mask')\n    plt.axis(\"off\")\n    \n    plt.show();","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:21.430945Z","iopub.execute_input":"2024-11-08T23:05:21.431317Z","iopub.status.idle":"2024-11-08T23:05:21.456922Z","shell.execute_reply.started":"2024-11-08T23:05:21.431260Z","shell.execute_reply":"2024-11-08T23:05:21.455391Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_ids = ['0030fd0e6378','0140b3c8f445','01ae5a43a2ab']\n\nfor sample_id in sample_ids:\n    celltype=train_data[train_data['id']==sample_id]['cell_type'].tolist()[0]\n    file_path = '../input/sartorius-cell-instance-segmentation/train/' + sample_id + '.png'\n    image_df = cv2.imread(file_path)\n    print('ID:', sample_id, ', CellType:',celltype)\n    plot_masks(sample_id, colors=False)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:21.460408Z","iopub.execute_input":"2024-11-08T23:05:21.460877Z","iopub.status.idle":"2024-11-08T23:05:23.427697Z","shell.execute_reply.started":"2024-11-08T23:05:21.460827Z","shell.execute_reply":"2024-11-08T23:05:23.426695Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reference: https://www.kaggle.com/keegil/keras-u-net-starter-lb-0-277\nIMG_HEIGHT = 256\nIMG_WIDTH = 256\nIMG_CHANNELS = 3\nTRAIN_PATH = '../input/sartorius-cell-instance-segmentation/train/'\n\ntrain_ids = train_data['id'].unique().tolist()\ntest_ids = sample_submission['id'].unique().tolist()\n\n# Get and resize train images and masks\nX_train = np.zeros((train_data['id'].nunique(), IMG_HEIGHT, IMG_WIDTH, IMG_CHANNELS), dtype=np.uint8)\nY_train = np.zeros((train_data['id'].nunique(), IMG_HEIGHT, IMG_WIDTH, 1), dtype=np.bool)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:23.429374Z","iopub.execute_input":"2024-11-08T23:05:23.429620Z","iopub.status.idle":"2024-11-08T23:05:23.462410Z","shell.execute_reply.started":"2024-11-08T23:05:23.429590Z","shell.execute_reply":"2024-11-08T23:05:23.461305Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tqdm import tqdm\nfor n, id_ in tqdm(enumerate(train_ids), total=len(train_ids)):\n    path = TRAIN_PATH + id_\n    img = cv2.imread(path + '.png')[:,:]\n    img = cv2.resize(img, (IMG_HEIGHT, IMG_WIDTH), interpolation = cv2.INTER_LINEAR)\n    #img = np.expand_dims(img, axis = 2)\n    X_train[n] = img\n    \n    labels = train_data[train_data[\"id\"] == id_][\"annotation\"].tolist()\n    mask = np.zeros((520, 704, 1))\n    for label in labels:\n        mask += rle_decode(label, shape=(520, 704, 1))\n    mask = mask.clip(0, 1)\n    mask = mask[:,:,0]\n\n    mask = np.expand_dims(cv2.resize(mask, (IMG_HEIGHT, IMG_WIDTH), interpolation = cv2.INTER_LINEAR), axis=-1)\n    \n    Y_train[n] = mask\nprint(\"Done\")","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:05:23.463896Z","iopub.execute_input":"2024-11-08T23:05:23.464199Z","iopub.status.idle":"2024-11-08T23:06:19.455515Z","shell.execute_reply.started":"2024-11-08T23:05:23.464163Z","shell.execute_reply":"2024-11-08T23:06:19.454591Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get and resize test images\nX_test = np.zeros((sample_submission['id'].nunique(), IMG_HEIGHT, IMG_WIDTH, IMG_CHANNELS), dtype=np.uint8)\ntest_images_id = []\nfor n, id_ in tqdm(enumerate(test_ids), total=len(test_ids)):\n    path = TRAIN_PATH.replace('train', 'test') + id_\n    img = cv2.imread(path + '.png')[:,:]\n    img = cv2.resize(img, (IMG_HEIGHT, IMG_WIDTH), interpolation = cv2.INTER_LINEAR)\n    #img = np.expand_dims(img, axis = 2)\n    X_test[n] = img\n    test_images_id.append(id_)\nprint(\"Done\")","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:06:19.456860Z","iopub.execute_input":"2024-11-08T23:06:19.457199Z","iopub.status.idle":"2024-11-08T23:06:19.527721Z","shell.execute_reply.started":"2024-11-08T23:06:19.457163Z","shell.execute_reply":"2024-11-08T23:06:19.526678Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(X_train.shape,Y_train.shape,X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:06:19.529130Z","iopub.execute_input":"2024-11-08T23:06:19.529409Z","iopub.status.idle":"2024-11-08T23:06:19.535584Z","shell.execute_reply.started":"2024-11-08T23:06:19.529365Z","shell.execute_reply":"2024-11-08T23:06:19.534608Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_id_num = 40\nplt.imshow(X_train[sample_id_num][:,:,0], cmap = 'gray')\nplt.show()\nplt.imshow(Y_train[sample_id_num][:,:,0])\nplt.show()\n\nprint('Input image:','Min:', X_train[sample_id_num][:,:,0].min(), '; Max:', X_train[sample_id_num][:,:,0].max(), '; Mean:', X_train[sample_id_num][:,:,0].mean())\nprint('Mask:','Min:', Y_train[sample_id_num][:,:,0].min(), '; Max:', Y_train[sample_id_num][:,:,0].max(), '; Mean:', Y_train[sample_id_num][:,:,0].mean())","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:06:19.536970Z","iopub.execute_input":"2024-11-08T23:06:19.537221Z","iopub.status.idle":"2024-11-08T23:06:19.903256Z","shell.execute_reply.started":"2024-11-08T23:06:19.537190Z","shell.execute_reply":"2024-11-08T23:06:19.902233Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# modeling","metadata":{"execution":{"iopub.status.busy":"2021-12-05T11:10:54.148824Z","iopub.execute_input":"2021-12-05T11:10:54.149111Z","iopub.status.idle":"2021-12-05T11:10:54.152922Z","shell.execute_reply.started":"2021-12-05T11:10:54.14908Z","shell.execute_reply":"2021-12-05T11:10:54.152239Z"}}},{"cell_type":"code","source":"#dice_coefficient\ndef dice_coefficient(y_true, y_pred):\n    numerator = 2 * tf.reduce_sum(y_true * y_pred)\n    denominator = tf.reduce_sum(y_true + y_pred)\n    return numerator / (denominator + tf.keras.backend.epsilon())","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:06:19.904564Z","iopub.execute_input":"2024-11-08T23:06:19.904832Z","iopub.status.idle":"2024-11-08T23:06:19.911149Z","shell.execute_reply.started":"2024-11-08T23:06:19.904786Z","shell.execute_reply":"2024-11-08T23:06:19.909912Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Build U-Net model\ninputs = Input((IMG_HEIGHT, IMG_WIDTH, IMG_CHANNELS))\nactivation='elu'\ns = Lambda(lambda x: x / 255) (inputs)\n\nc1 = Conv2D(16, (3, 3), activation=activation, kernel_initializer='he_normal', padding='same') (s)\nc1 = Dropout(0.1) (c1)\nc1 = Conv2D(16, (3, 3), activation=activation, kernel_initializer='he_normal', padding='same') (c1)\np1 = MaxPooling2D((2, 2)) (c1)\n\nc2 = Conv2D(32, (3, 3), activation=activation, kernel_initializer='he_normal', padding='same') (p1)\nc2 = Dropout(0.1) (c2)\nc2 = Conv2D(32, (3, 3), activation=activation, kernel_initializer='he_normal', padding='same') (c2)\np2 = MaxPooling2D((2, 2)) (c2)\n\nc3 = Conv2D(64, (3, 3), activation=activation, kernel_initializer='he_normal', padding='same') (p2)\nc3 = Dropout(0.2) (c3)\nc3 = Conv2D(64, (3, 3), activation=activation, kernel_initializer='he_normal', padding='same') (c3)\np3 = MaxPooling2D((2, 2)) (c3)\n\nc4 = Conv2D(128, (3, 3), activation=activation, kernel_initializer='he_normal', padding='same') (p3)\nc4 = Dropout(0.2) (c4)\nc4 = Conv2D(128, (3, 3), activation=activation, kernel_initializer='he_normal', padding='same') (c4)\np4 = MaxPooling2D(pool_size=(2, 2)) (c4)\n\nc5 = Conv2D(256, (3, 3), activation=activation, kernel_initializer='he_normal', padding='same') (p4)\nc5 = Dropout(0.3) (c5)\nc5 = Conv2D(256, (3, 3), activation=activation, kernel_initializer='he_normal', padding='same') (c5)\n\nu6 = Conv2DTranspose(128, (2, 2), strides=(2, 2), padding='same') (c5)\nu6 = concatenate([u6, c4])\nc6 = Conv2D(128, (3, 3), activation=activation, kernel_initializer='he_normal', padding='same') (u6)\nc6 = Dropout(0.2) (c6)\nc6 = Conv2D(128, (3, 3), activation=activation, kernel_initializer='he_normal', padding='same') (c6)\n\nu7 = Conv2DTranspose(64, (2, 2), strides=(2, 2), padding='same') (c6)\nu7 = concatenate([u7, c3])\nc7 = Conv2D(64, (3, 3), activation=activation, kernel_initializer='he_normal', padding='same') (u7)\nc7 = Dropout(0.2) (c7)\nc7 = Conv2D(64, (3, 3), activation=activation, kernel_initializer='he_normal', padding='same') (c7)\n\nu8 = Conv2DTranspose(32, (2, 2), strides=(2, 2), padding='same') (c7)\nu8 = concatenate([u8, c2])\nc8 = Conv2D(32, (3, 3), activation=activation, kernel_initializer='he_normal', padding='same') (u8)\nc8 = Dropout(0.1) (c8)\nc8 = Conv2D(32, (3, 3), activation=activation, kernel_initializer='he_normal', padding='same') (c8)\n\nu9 = Conv2DTranspose(16, (2, 2), strides=(2, 2), padding='same') (c8)\nu9 = concatenate([u9, c1], axis=3)\nc9 = Conv2D(16, (3, 3), activation=activation, kernel_initializer='he_normal', padding='same') (u9)\nc9 = Dropout(0.1) (c9)\nc9 = Conv2D(16, (3, 3), activation=activation, kernel_initializer='he_normal', padding='same') (c9)\n\noutputs = Conv2D(1, (1, 1), activation='sigmoid') (c9)\n\nmodel = Model(inputs=[inputs], outputs=[outputs])\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=[dice_coefficient])\n#model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:06:19.912639Z","iopub.execute_input":"2024-11-08T23:06:19.912979Z","iopub.status.idle":"2024-11-08T23:06:20.423431Z","shell.execute_reply.started":"2024-11-08T23:06:19.912933Z","shell.execute_reply":"2024-11-08T23:06:20.422456Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Fit model\nearlystopper = EarlyStopping(patience=40, verbose=1)\n#checkpointer = ModelCheckpoint('best_model.h5', verbose=1, save_best_only=True)\n\nresults = model.fit(X_train, Y_train, validation_split=0.12, batch_size=10, epochs=71, \n                    callbacks=[earlystopper])","metadata":{"execution":{"iopub.status.busy":"2024-11-08T23:06:20.425088Z","iopub.execute_input":"2024-11-08T23:06:20.425428Z","iopub.status.idle":"2024-11-09T01:36:27.001877Z","shell.execute_reply.started":"2024-11-08T23:06:20.425382Z","shell.execute_reply":"2024-11-09T01:36:27.000592Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(14,4))\nplt.plot(results.history['loss'])\nplt.plot(results.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('Loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'val'], loc='upper right')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:36:27.003836Z","iopub.execute_input":"2024-11-09T01:36:27.004564Z","iopub.status.idle":"2024-11-09T01:36:27.274125Z","shell.execute_reply.started":"2024-11-09T01:36:27.004517Z","shell.execute_reply":"2024-11-09T01:36:27.273069Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(14,4))\nplt.plot(results.history['dice_coefficient'])\nplt.plot(results.history['val_dice_coefficient'])\nplt.title('dice_coefficient')\nplt.ylabel('dice_coefficient')\nplt.xlabel('epoch')\nplt.legend(['train', 'val'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:36:27.275488Z","iopub.execute_input":"2024-11-09T01:36:27.275726Z","iopub.status.idle":"2024-11-09T01:36:27.549697Z","shell.execute_reply.started":"2024-11-09T01:36:27.275696Z","shell.execute_reply":"2024-11-09T01:36:27.548740Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.shape[0]*0.85","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:36:27.551219Z","iopub.execute_input":"2024-11-09T01:36:27.551556Z","iopub.status.idle":"2024-11-09T01:36:27.558807Z","shell.execute_reply.started":"2024-11-09T01:36:27.551511Z","shell.execute_reply":"2024-11-09T01:36:27.557762Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict on train, val and test\n#model = load_model('best_model.h5', custom_objects={'dice_coefficient': dice_coefficient})\npreds_train = model.predict(X_train[:int(X_train.shape[0]*0.85)], verbose=1)\npreds_test = model.predict(X_test, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:36:27.560221Z","iopub.execute_input":"2024-11-09T01:36:27.560489Z","iopub.status.idle":"2024-11-09T01:36:57.694183Z","shell.execute_reply.started":"2024-11-09T01:36:27.560456Z","shell.execute_reply":"2024-11-09T01:36:57.692857Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# by default I use threshold of 0.5. It is worth optimize it? In some notebooks the use it per class type. \nfrom statistics import mean\ndef get_threshold(Y, pred):\n    scores = list(pred.ravel())\n    mask = list(Y.ravel())\n    \n    idxs=np.argsort(scores)[::-1]\n    mask_sorted=np.array(mask)[idxs]\n    sum_mask_one=np.cumsum(mask_sorted)\n    IoU=sum_mask_one/(np.arange(1,len(mask_sorted)+1)+np.sum(mask_sorted)-sum_mask_one)\n    best_IoU_idx=IoU.argmax()\n    best_threshold=scores[idxs[best_IoU_idx]]\n    best_IoU=IoU[best_IoU_idx]\n\n    return best_threshold, best_IoU\n\n\nimg_thresholds = []         # one for each image\nimg_IoUs = []\nfor Y, P in tqdm(zip(Y_train, preds_train), total=Y_train.shape[0]):\n\n    best_img_threshold, best_img_IoU = get_threshold(Y, P)\n    img_thresholds.append(best_img_threshold)\n    img_IoUs.append(best_img_IoU)\n    \nbest_threshold = np.mean(img_thresholds)\nbest_threshold_spread = np.std(img_thresholds)\navg_IoU = mean(img_IoUs)\n\nprint(f\"Best threshold: {best_threshold:.3g} (+-{best_threshold_spread:.3g}), Avg. Train IoU: {avg_IoU:.3f}\")","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:36:57.695750Z","iopub.execute_input":"2024-11-09T01:36:57.696065Z","iopub.status.idle":"2024-11-09T01:37:22.599327Z","shell.execute_reply.started":"2024-11-09T01:36:57.696032Z","shell.execute_reply":"2024-11-09T01:37:22.598416Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Threshold predictions\ndefault_threshold = False\n\nif default_threshold:\n    threshold=0.5\nelse:\n    threshold=best_threshold\n\n\npreds_train_t = (preds_train > threshold).astype(np.uint8)\npreds_test_t = (preds_test > threshold).astype(np.uint8)\n\n# Create list of upsampled test masks\npreds_test_upsampled = []\nfor i in range(len(preds_test)):\n    preds_test_upsampled.append(cv2.resize(np.squeeze(preds_test[i]), \n                                    (IMG_HEIGHT, IMG_WIDTH), interpolation = cv2.INTER_AREA))","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:22.600605Z","iopub.execute_input":"2024-11-09T01:37:22.600878Z","iopub.status.idle":"2024-11-09T01:37:22.649783Z","shell.execute_reply.started":"2024-11-09T01:37:22.600844Z","shell.execute_reply":"2024-11-09T01:37:22.648846Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Perform a sanity check on some random training samples\nfrom random import randint\nix = randint(0, len(preds_train_t))\nplt.figure(figsize=(18,6))\nplt.subplot(1, 3, 1)\nplt.imshow(X_train[ix])\nplt.title('Input image')\nplt.axis(\"off\")\n\nplt.subplot(1, 3, 2)\nplt.imshow(np.squeeze(Y_train[ix]))\nplt.title('Mask')\nplt.axis(\"off\")\n\nplt.subplot(1, 3, 3)\nplt.imshow(np.squeeze(preds_train_t[ix]))\nplt.title('predicted mask')\nplt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:22.651004Z","iopub.execute_input":"2024-11-09T01:37:22.651257Z","iopub.status.idle":"2024-11-09T01:37:23.058425Z","shell.execute_reply.started":"2024-11-09T01:37:22.651227Z","shell.execute_reply":"2024-11-09T01:37:23.057316Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ix = randint(0, len(preds_train_t))\nplt.figure(figsize=(18,6));\nplt.subplot(1, 3, 1)\nplt.imshow(X_train[ix])\nplt.title('Input image')\nplt.axis(\"off\")\n\nplt.subplot(1, 3, 2)\nplt.imshow(np.squeeze(Y_train[ix]))\nplt.title('Mask')\nplt.axis(\"off\")\n\nplt.subplot(1, 3, 3)\nplt.imshow(np.squeeze(preds_train_t[ix]))\nplt.title('predicted mask')\nplt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:23.060285Z","iopub.execute_input":"2024-11-09T01:37:23.060632Z","iopub.status.idle":"2024-11-09T01:37:23.479684Z","shell.execute_reply.started":"2024-11-09T01:37:23.060591Z","shell.execute_reply":"2024-11-09T01:37:23.478629Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#predictions not empty right?\nprint(np.count_nonzero(preds_train_t[ix]))","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:23.481148Z","iopub.execute_input":"2024-11-09T01:37:23.481455Z","iopub.status.idle":"2024-11-09T01:37:23.487491Z","shell.execute_reply.started":"2024-11-09T01:37:23.481415Z","shell.execute_reply":"2024-11-09T01:37:23.486531Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from matplotlib.colors import ListedColormap\ncmap = ListedColormap(['black', 'gray', 'orange', 'green'])\n\ndef plot_colored(img_Y, img_pred):\n    output = np.zeros_like(img_Y)\n    output = np.where((img_Y == 0) & (img_pred == 1), 1, output)\n    output = np.where((img_Y == 1) & (img_pred == 0), 2, output)\n    output = np.where((img_Y == 1) & (img_pred == 1), 3, output)\n\n    plt.figure(figsize=(10,10))\n    plt.imshow(output, cmap=cmap)\n    plt.xticks([])\n    plt.yticks([]);","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:23.488902Z","iopub.execute_input":"2024-11-09T01:37:23.489173Z","iopub.status.idle":"2024-11-09T01:37:23.503195Z","shell.execute_reply.started":"2024-11-09T01:37:23.489140Z","shell.execute_reply":"2024-11-09T01:37:23.502162Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"N = 4\nfor i in range(N):\n    plot_colored(Y_train[i], preds_train_t[i])\n    plt.show()\n# green: correct prediction\n# gray: false positive (too much)\n# orange: false negative (missed)","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:23.504688Z","iopub.execute_input":"2024-11-09T01:37:23.504977Z","iopub.status.idle":"2024-11-09T01:37:24.110444Z","shell.execute_reply.started":"2024-11-09T01:37:23.504943Z","shell.execute_reply":"2024-11-09T01:37:24.109316Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# prepare the submision file","metadata":{}},{"cell_type":"code","source":"#test_mask: after reshape before fix_overlapping\ntest_masks = [cv2.resize(pred,dsize=(704,520),interpolation=cv2.INTER_CUBIC).reshape(520,704,1) for pred in preds_test_t]\nprint(test_masks[0].shape)\nprint(test_masks[1].shape)\nprint(test_masks[2].shape)","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:24.112083Z","iopub.execute_input":"2024-11-09T01:37:24.112340Z","iopub.status.idle":"2024-11-09T01:37:24.142815Z","shell.execute_reply.started":"2024-11-09T01:37:24.112309Z","shell.execute_reply":"2024-11-09T01:37:24.141797Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# fix_overlap\nhttps://www.kaggle.com/c/sartorius-cell-instance-segmentation/discussion/279995","metadata":{}},{"cell_type":"code","source":"def check_overlap(msk):\n    msk = msk.astype(np.bool).astype(np.uint8)\n    return np.any(np.sum(msk, axis=-1)>1)\n\ndef fix_overlap(msk):\n    \"\"\"\n    Args:\n        mask: multi-channel mask, each channel is an instance of cell, shape:(520,704,None)\n    Returns:\n        multi-channel mask with non-overlapping values, shape:(520,704,None)\n    \"\"\"\n    msk = np.array(msk)\n    msk = np.pad(msk, [[0,0],[0,0],[1,0]])\n    ins_len = msk.shape[-1]\n    msk = np.argmax(msk,axis=-1)\n    msk = tf.keras.utils.to_categorical(msk, num_classes=ins_len)\n    msk = msk[...,1:]\n    msk = msk[...,np.any(msk, axis=(0,1))]\n    return msk\n\ndef remove_isolated_points_from_rle(strin):\n    t2 = strin.split(\" \")\n    a = []\n    for i in range(0, len(t2), 2):\n        if t2[i+1]!=\"1\":\n            a.append(t2[i])\n            a.append(t2[i+1])\n    return ' '.join(a)","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:24.144120Z","iopub.execute_input":"2024-11-09T01:37:24.144395Z","iopub.status.idle":"2024-11-09T01:37:24.157678Z","shell.execute_reply.started":"2024-11-09T01:37:24.144360Z","shell.execute_reply":"2024-11-09T01:37:24.156595Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for test_mask in test_masks:\n    print(check_overlap(test_mask))","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:24.159178Z","iopub.execute_input":"2024-11-09T01:37:24.159482Z","iopub.status.idle":"2024-11-09T01:37:24.181298Z","shell.execute_reply.started":"2024-11-09T01:37:24.159443Z","shell.execute_reply":"2024-11-09T01:37:24.179868Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#test_mask2: after reshape after fix_overlapping. No need right now\n#test_masks2=[]\n#for test_mask in test_masks:\n#    test_mask2 = fix_overlap(test_mask).reshape(520,704,1)\n#    print(test_mask2.shape)\n#    test_masks2+=[test_mask2]\n\n#for test_mask2 in test_masks2:\n #   print(check_overlap(test_mask2))","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:24.182793Z","iopub.execute_input":"2024-11-09T01:37:24.183056Z","iopub.status.idle":"2024-11-09T01:37:24.189662Z","shell.execute_reply.started":"2024-11-09T01:37:24.183025Z","shell.execute_reply":"2024-11-09T01:37:24.188593Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# revert the output codes to the file","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:24.191330Z","iopub.execute_input":"2024-11-09T01:37:24.191650Z","iopub.status.idle":"2024-11-09T01:37:24.203171Z","shell.execute_reply.started":"2024-11-09T01:37:24.191614Z","shell.execute_reply":"2024-11-09T01:37:24.202152Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predicted2 = [rle_encode(test_mask2) for test_mask2 in test_masks]\n#print(predicted2[0])","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:24.204645Z","iopub.execute_input":"2024-11-09T01:37:24.204994Z","iopub.status.idle":"2024-11-09T01:37:24.251368Z","shell.execute_reply.started":"2024-11-09T01:37:24.204951Z","shell.execute_reply":"2024-11-09T01:37:24.249852Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# split the mask into each cluster nucleus","metadata":{}},{"cell_type":"code","source":"# split the mask into each cluster nucleus for the submision\n# seen on https://www.kaggle.com/c/sartorius-cell-instance-segmentation/discussion/288376\ndef post_process(mask, min_size=80, shape=(520, 704,)):\n    num_component, component = cv2.connectedComponents(mask.astype(np.uint8))\n    predictions = []\n    for c in range(1, num_component):\n        p = (component == c)\n        if p.sum() > min_size:\n            a_prediction = np.zeros(shape, np.float32)\n            a_prediction[p] = 1\n            predictions.append(a_prediction)\n    return predictions","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:24.252733Z","iopub.execute_input":"2024-11-09T01:37:24.253036Z","iopub.status.idle":"2024-11-09T01:37:24.261538Z","shell.execute_reply.started":"2024-11-09T01:37:24.253001Z","shell.execute_reply":"2024-11-09T01:37:24.260322Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# test the nucleus thing. Take one simple mask\nplt.imshow(Y_train[4], cmap=\"gray\");","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:24.262896Z","iopub.execute_input":"2024-11-09T01:37:24.263278Z","iopub.status.idle":"2024-11-09T01:37:24.457942Z","shell.execute_reply.started":"2024-11-09T01:37:24.263232Z","shell.execute_reply":"2024-11-09T01:37:24.456992Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# connectedComponents returns the number of compenets of the image and a image with a pixel value of each of them\nnum_component, component = cv2.connectedComponents(Y_train[4].astype(np.uint8))\nprint(num_component)\nplt.imshow(component, cmap=\"gray\");","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:24.459167Z","iopub.execute_input":"2024-11-09T01:37:24.459432Z","iopub.status.idle":"2024-11-09T01:37:24.654560Z","shell.execute_reply.started":"2024-11-09T01:37:24.459400Z","shell.execute_reply":"2024-11-09T01:37:24.653780Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# extraction of the component of value 5 as example\ncompenent_5 = (component == 5)\nplt.imshow(compenent_5, cmap=\"gray\");","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:24.655768Z","iopub.execute_input":"2024-11-09T01:37:24.656050Z","iopub.status.idle":"2024-11-09T01:37:24.901651Z","shell.execute_reply.started":"2024-11-09T01:37:24.656015Z","shell.execute_reply":"2024-11-09T01:37:24.900697Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# all together\n# notice that with the size of the training the min size also is affected. The minimun size on this case is 20\nfinal = post_process(Y_train[4], min_size=20, shape=(IMG_HEIGHT, IMG_WIDTH,))\nprint(final[0].shape)\nplt.imshow(final[0], cmap=\"gray\");","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:24.903440Z","iopub.execute_input":"2024-11-09T01:37:24.903809Z","iopub.status.idle":"2024-11-09T01:37:25.102419Z","shell.execute_reply.started":"2024-11-09T01:37:24.903756Z","shell.execute_reply":"2024-11-09T01:37:25.101321Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# create submision file","metadata":{}},{"cell_type":"code","source":"preds_test_t[0].shape","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:25.103637Z","iopub.execute_input":"2024-11-09T01:37:25.103894Z","iopub.status.idle":"2024-11-09T01:37:25.110925Z","shell.execute_reply.started":"2024-11-09T01:37:25.103863Z","shell.execute_reply":"2024-11-09T01:37:25.109971Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# new version with the mask nucleus split\npredicted_nucleus = []\ntest_nucleus_image_id = []\n\nfor index, s in enumerate(preds_test_t):\n    nucleus = post_process(cv2.resize(s, (704,520,), interpolation = cv2.INTER_LINEAR))\n    for nucl in nucleus:\n        predicted_nucleus.append(nucl)\n        test_nucleus_image_id.append(test_images_id[index])","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:25.112449Z","iopub.execute_input":"2024-11-09T01:37:25.112719Z","iopub.status.idle":"2024-11-09T01:37:25.237293Z","shell.execute_reply.started":"2024-11-09T01:37:25.112664Z","shell.execute_reply":"2024-11-09T01:37:25.236343Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.imshow(predicted_nucleus[0], cmap=\"gray\");\npredicted_nucleus[0].shape","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:25.238592Z","iopub.execute_input":"2024-11-09T01:37:25.238842Z","iopub.status.idle":"2024-11-09T01:37:25.549411Z","shell.execute_reply.started":"2024-11-09T01:37:25.238812Z","shell.execute_reply":"2024-11-09T01:37:25.548535Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predicted2 = [rle_encode(test_mask2) for test_mask2 in predicted_nucleus]\nprint(predicted2[0])\npredicted_filt = [remove_isolated_points_from_rle(s) for s in predicted2]\nprint(predicted_filt[0])","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:25.550719Z","iopub.execute_input":"2024-11-09T01:37:25.550996Z","iopub.status.idle":"2024-11-09T01:37:25.695207Z","shell.execute_reply.started":"2024-11-09T01:37:25.550963Z","shell.execute_reply":"2024-11-09T01:37:25.694147Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submit = sample_submission.copy()\nsubmit = pd.DataFrame({'id':test_nucleus_image_id, 'predicted':predicted_filt})\nprint(submit.shape)\nsubmit.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:25.696557Z","iopub.execute_input":"2024-11-09T01:37:25.696851Z","iopub.status.idle":"2024-11-09T01:37:25.713817Z","shell.execute_reply.started":"2024-11-09T01:37:25.696813Z","shell.execute_reply":"2024-11-09T01:37:25.712812Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submit.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-11-09T01:37:25.715214Z","iopub.execute_input":"2024-11-09T01:37:25.715496Z","iopub.status.idle":"2024-11-09T01:37:25.736951Z","shell.execute_reply.started":"2024-11-09T01:37:25.715448Z","shell.execute_reply":"2024-11-09T01:37:25.736109Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null}]}