{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":30201,"databundleVersionId":2750748,"sourceType":"competition"}],"dockerImageVersionId":30152,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from pathlib import Path\nimport cv2\nimport matplotlib.pyplot as plt\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\nimport tensorflow as tf\nfrom statistics import mean\nfrom matplotlib.colors import ListedColormap","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-08T17:03:35.513157Z","iopub.execute_input":"2024-11-08T17:03:35.513499Z","iopub.status.idle":"2024-11-08T17:03:35.519590Z","shell.execute_reply.started":"2024-11-08T17:03:35.513452Z","shell.execute_reply":"2024-11-08T17:03:35.518608Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# read data first. The csv is the core.\ntrain_data = pd.read_csv('../input/sartorius-cell-instance-segmentation/train.csv')\n# submision sample\nsample_submission=pd.read_csv('../input/sartorius-cell-instance-segmentation/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-11-08T17:03:35.521472Z","iopub.execute_input":"2024-11-08T17:03:35.521777Z","iopub.status.idle":"2024-11-08T17:03:36.212973Z","shell.execute_reply.started":"2024-11-08T17:03:35.521740Z","shell.execute_reply":"2024-11-08T17:03:36.212041Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# https://www.kaggle.com/c/sartorius-cell-instance-segmentation/discussion/291627\ndef rle_decode(mask_rle, shape=(520, 704, 1)):\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape)  # Needed to align to RLE direction\n\ndef rle_encode(img):\n    pixels = img.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T17:03:36.214205Z","iopub.execute_input":"2024-11-08T17:03:36.214466Z","iopub.status.idle":"2024-11-08T17:03:36.223981Z","shell.execute_reply.started":"2024-11-08T17:03:36.214431Z","shell.execute_reply":"2024-11-08T17:03:36.222752Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# each mask annotation has one area\nmask = train_data[train_data[\"id\"] == \"0030fd0e6378\"][\"annotation\"].tolist()[0]\nimg = rle_decode(mask)\nplt.imshow(img, cmap=\"gray\");","metadata":{"execution":{"iopub.status.busy":"2024-11-08T17:03:36.226042Z","iopub.execute_input":"2024-11-08T17:03:36.226895Z","iopub.status.idle":"2024-11-08T17:03:36.695216Z","shell.execute_reply.started":"2024-11-08T17:03:36.226842Z","shell.execute_reply":"2024-11-08T17:03:36.694344Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_masks(image_id, colors=False):\n    labels = train_data[train_data[\"id\"] == image_id][\"annotation\"].tolist()\n\n    if colors:\n        mask = np.zeros((520, 704, 3))\n        for label in labels:\n            mask += rle_decode(label, shape=(520, 704, 3), color=np.random.rand(3))\n    else:\n        mask = np.zeros((520, 704, 1))\n        for label in labels:\n            mask += rle_decode(label, shape=(520, 704, 1))\n    mask = mask.clip(0, 1)\n\n    image = cv2.imread(f\"../input/sartorius-cell-instance-segmentation/train/{image_id}.png\")\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n    plt.figure(figsize=(18,6))\n    plt.subplot(1, 3, 1)\n    plt.imshow(image)\n    plt.title('Input image')\n    plt.axis(\"off\")\n    \n    plt.subplot(1, 3, 2)\n    plt.imshow(image)\n    plt.imshow(mask, alpha=0.1)\n    plt.title('Input image with mask')\n    plt.axis(\"off\")\n    \n    plt.subplot(1, 3, 3)\n    plt.imshow(mask)\n    plt.title('Only mask')\n    plt.axis(\"off\")\n    \n    plt.show();","metadata":{"execution":{"iopub.status.busy":"2024-11-08T17:03:36.697431Z","iopub.execute_input":"2024-11-08T17:03:36.697694Z","iopub.status.idle":"2024-11-08T17:03:36.707755Z","shell.execute_reply.started":"2024-11-08T17:03:36.697655Z","shell.execute_reply":"2024-11-08T17:03:36.706832Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_ids = ['0030fd0e6378','0140b3c8f445','01ae5a43a2ab']\n\nfor sample_id in sample_ids:\n    celltype=train_data[train_data['id']==sample_id]['cell_type'].tolist()[0]\n    file_path = '../input/sartorius-cell-instance-segmentation/train/' + sample_id + '.png'\n    image_df = cv2.imread(file_path)\n    print('ID:', sample_id, ', CellType:',celltype)\n    plot_masks(sample_id, colors=False)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T17:03:36.709194Z","iopub.execute_input":"2024-11-08T17:03:36.709524Z","iopub.status.idle":"2024-11-08T17:03:38.346196Z","shell.execute_reply.started":"2024-11-08T17:03:36.709478Z","shell.execute_reply":"2024-11-08T17:03:38.345293Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reference: https://www.kaggle.com/keegil/keras-u-net-starter-lb-0-277\nIMG_HEIGHT = 520\nIMG_WIDTH = 704\nIMG_CHANNELS = 1\nTRAIN_PATH = '../input/sartorius-cell-instance-segmentation/train/'\n\ntrain_ids = train_data['id'].unique().tolist()\ntest_ids = sample_submission['id'].unique().tolist()\n\n# Get and resize train images and masks\nX_train = np.zeros((train_data['id'].nunique(), IMG_HEIGHT, IMG_WIDTH, IMG_CHANNELS), dtype=np.uint8)\nY_train = np.zeros((train_data['id'].nunique(), IMG_HEIGHT, IMG_WIDTH, 1), dtype=np.bool)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T17:03:38.347376Z","iopub.execute_input":"2024-11-08T17:03:38.347609Z","iopub.status.idle":"2024-11-08T17:03:38.376380Z","shell.execute_reply.started":"2024-11-08T17:03:38.347580Z","shell.execute_reply":"2024-11-08T17:03:38.375503Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tqdm import tqdm\nfor n, id_ in tqdm(enumerate(train_ids), total=len(train_ids)):\n    path = TRAIN_PATH + id_\n    img = cv2.imread(path + '.png')[:,:]\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY).astype(np.float32) -125\n    img = np.expand_dims(img, axis = 2)\n    X_train[n] = img\n    \n    labels = train_data[train_data[\"id\"]\n                        == id_][\"annotation\"].tolist()\n    mask = np.zeros((520, 704, 1))\n    for label in labels:\n        mask += rle_decode(label, shape=(520, 704, 1))\n    mask = mask.clip(0, 1)\n\n    Y_train[n] = mask\nprint(\"Done\")","metadata":{"execution":{"iopub.status.busy":"2024-11-08T17:03:38.377647Z","iopub.execute_input":"2024-11-08T17:03:38.377880Z","iopub.status.idle":"2024-11-08T17:04:27.559245Z","shell.execute_reply.started":"2024-11-08T17:03:38.377852Z","shell.execute_reply":"2024-11-08T17:04:27.558205Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get and resize test images\ntest_images_id = []\nX_test = np.zeros((sample_submission['id'].nunique(), IMG_HEIGHT, IMG_WIDTH, IMG_CHANNELS), dtype=np.uint8)\nfor n, id_ in tqdm(enumerate(test_ids), total=len(test_ids)):\n    path = TRAIN_PATH.replace('train', 'test') + id_\n    img = cv2.imread(path + '.png')[:,:]\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY).astype(np.float32) -125\n    img = np.expand_dims(img, axis = 2)\n    X_test[n] = img\n    test_images_id.append(id_)\nprint(\"Done\")","metadata":{"execution":{"iopub.status.busy":"2024-11-08T17:04:27.561017Z","iopub.execute_input":"2024-11-08T17:04:27.561343Z","iopub.status.idle":"2024-11-08T17:04:27.631520Z","shell.execute_reply.started":"2024-11-08T17:04:27.561297Z","shell.execute_reply":"2024-11-08T17:04:27.630651Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(X_train.shape,Y_train.shape,X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T17:04:27.632721Z","iopub.execute_input":"2024-11-08T17:04:27.632988Z","iopub.status.idle":"2024-11-08T17:04:27.638518Z","shell.execute_reply.started":"2024-11-08T17:04:27.632949Z","shell.execute_reply":"2024-11-08T17:04:27.637662Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_id_num = 40\nplt.imshow(X_train[sample_id_num][:,:,0], cmap = 'gray')\nplt.show()\nplt.imshow(Y_train[sample_id_num][:,:,0])\nplt.show()\n\nprint('Input image:','Min:', X_train[sample_id_num][:,:,0].min(), '; Max:', X_train[sample_id_num][:,:,0].max(), '; Mean:', X_train[sample_id_num][:,:,0].mean())\nprint('Mask:','Min:', Y_train[sample_id_num][:,:,0].min(), '; Max:', Y_train[sample_id_num][:,:,0].max(), '; Mean:', Y_train[sample_id_num][:,:,0].mean())","metadata":{"execution":{"iopub.status.busy":"2024-11-08T17:04:27.639890Z","iopub.execute_input":"2024-11-08T17:04:27.640204Z","iopub.status.idle":"2024-11-08T17:04:28.084373Z","shell.execute_reply.started":"2024-11-08T17:04:27.640163Z","shell.execute_reply":"2024-11-08T17:04:28.083385Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#dice_coefficient\ndef dice_coefficient(y_true, y_pred):\n    numerator = 2 * tf.reduce_sum(y_true * y_pred)\n    denominator = tf.reduce_sum(y_true + y_pred)\n    return numerator / (denominator + tf.keras.backend.epsilon())","metadata":{"execution":{"iopub.status.busy":"2024-11-08T17:04:28.085886Z","iopub.execute_input":"2024-11-08T17:04:28.086240Z","iopub.status.idle":"2024-11-08T17:04:28.092561Z","shell.execute_reply.started":"2024-11-08T17:04:28.086195Z","shell.execute_reply":"2024-11-08T17:04:28.091296Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"from torch import nn\nNN = nn.Sequential(\n    nn.Conv2d(1, 20, kernel_size=5, padding=\"same\"),\n    nn.BatchNorm2d(20),\n    nn.ReLU(),\n    nn.Conv2d(20, 10, kernel_size=1),\n\n    nn.Conv2d(10, 10, kernel_size=5, padding=\"same\"),\n    nn.BatchNorm2d(10),\n    nn.ReLU(),\n    nn.Conv2d(10, 1, kernel_size=1),\n)\nfrom torchsummary import summary\nsummary(model_to_transfer, input_size=[IMG_WIDTH,IMG_HEIGHT,IMG_CHANNELS])\n\"\"\"\n#input_var.shape[-3:]","metadata":{"execution":{"iopub.status.busy":"2024-11-08T17:04:28.094229Z","iopub.execute_input":"2024-11-08T17:04:28.094568Z","iopub.status.idle":"2024-11-08T17:04:28.106081Z","shell.execute_reply.started":"2024-11-08T17:04:28.094523Z","shell.execute_reply":"2024-11-08T17:04:28.105117Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import keras\nfrom keras.models import Model, load_model\nfrom keras import layers\n\nmodel = keras.Sequential([\n    # Convolutional layer 1\n    keras.layers.Conv2D(filters=20, kernel_size=5, strides=1,\n                  padding='same',input_shape=[IMG_WIDTH,IMG_HEIGHT,IMG_CHANNELS],\n                  activation='relu'),\n    keras.layers.BatchNormalization(),\n    \n    # Convolutional layer 2\n    keras.layers.Conv2D(filters=10, kernel_size=1),\n\n    # Convolutional layer 3\n    keras.layers.Conv2D(filters=10, kernel_size=5, strides=1,\n                  padding='same', activation='relu'),\n    keras.layers.BatchNormalization(),\n\n    # Convolutional layer 4\n    keras.layers.Conv2D(filters=1, kernel_size=1),\n])","metadata":{"execution":{"iopub.status.busy":"2024-11-08T17:04:28.109463Z","iopub.execute_input":"2024-11-08T17:04:28.109825Z","iopub.status.idle":"2024-11-08T17:04:28.967710Z","shell.execute_reply.started":"2024-11-08T17:04:28.109790Z","shell.execute_reply":"2024-11-08T17:04:28.966615Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.losses import BinaryCrossentropy\nloss = BinaryCrossentropy(from_logits=True)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T17:04:28.969189Z","iopub.execute_input":"2024-11-08T17:04:28.969514Z","iopub.status.idle":"2024-11-08T17:04:29.338898Z","shell.execute_reply.started":"2024-11-08T17:04:28.969473Z","shell.execute_reply":"2024-11-08T17:04:29.337987Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(optimizer='adam', loss=loss)\n#model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-11-08T17:04:29.340181Z","iopub.execute_input":"2024-11-08T17:04:29.340425Z","iopub.status.idle":"2024-11-08T17:04:29.357136Z","shell.execute_reply.started":"2024-11-08T17:04:29.340395Z","shell.execute_reply":"2024-11-08T17:04:29.356262Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Fit model\nn_epochs = 50\nbatch_size = 32\nfrom keras.callbacks import EarlyStopping\nearlystopper = EarlyStopping(patience=20, verbose=1)\n\nresults = model.fit(X_train, Y_train, validation_split=0.15, batch_size=batch_size, epochs=n_epochs, \n                    callbacks=[earlystopper])\nprint(\"Done!\")","metadata":{"execution":{"iopub.status.busy":"2024-11-08T17:04:29.358491Z","iopub.execute_input":"2024-11-08T17:04:29.359041Z","iopub.status.idle":"2024-11-08T18:53:25.350276Z","shell.execute_reply.started":"2024-11-08T17:04:29.358996Z","shell.execute_reply":"2024-11-08T18:53:25.348943Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(14,4))\nplt.plot(results.history['loss'])\nplt.plot(results.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('Loss')\nplt.xlabel('epoch')\nplt.legend(['loss', 'val_loss'], loc='upper right')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:53:25.352559Z","iopub.execute_input":"2024-11-08T18:53:25.352847Z","iopub.status.idle":"2024-11-08T18:53:25.619822Z","shell.execute_reply.started":"2024-11-08T18:53:25.352816Z","shell.execute_reply":"2024-11-08T18:53:25.618673Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(X_train.shape)\nprint(Y_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:53:25.621536Z","iopub.execute_input":"2024-11-08T18:53:25.621886Z","iopub.status.idle":"2024-11-08T18:53:25.628315Z","shell.execute_reply.started":"2024-11-08T18:53:25.621840Z","shell.execute_reply":"2024-11-08T18:53:25.627254Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds_train = model.predict(X_train, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:53:25.629749Z","iopub.execute_input":"2024-11-08T18:53:25.630026Z","iopub.status.idle":"2024-11-08T18:53:56.128844Z","shell.execute_reply.started":"2024-11-08T18:53:25.629991Z","shell.execute_reply":"2024-11-08T18:53:56.127793Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:53:56.130447Z","iopub.execute_input":"2024-11-08T18:53:56.130727Z","iopub.status.idle":"2024-11-08T18:53:56.137646Z","shell.execute_reply.started":"2024-11-08T18:53:56.130692Z","shell.execute_reply":"2024-11-08T18:53:56.136666Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Threshold predictions\npreds_train_t = (preds_train > 0.5).astype(np.uint8)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:53:56.139019Z","iopub.execute_input":"2024-11-08T18:53:56.139263Z","iopub.status.idle":"2024-11-08T18:53:56.419823Z","shell.execute_reply.started":"2024-11-08T18:53:56.139233Z","shell.execute_reply":"2024-11-08T18:53:56.418701Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.imshow(preds_train_t[0], cmap=\"gray\");","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:53:56.421464Z","iopub.execute_input":"2024-11-08T18:53:56.421724Z","iopub.status.idle":"2024-11-08T18:53:56.662029Z","shell.execute_reply.started":"2024-11-08T18:53:56.421692Z","shell.execute_reply":"2024-11-08T18:53:56.661169Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# unoptimized and slow; any way to speed up?\n\ndef get_threshold(Y, pred):\n    scores = list(pred.ravel())\n    mask = list(Y.ravel())\n    \n    idxs=np.argsort(scores)[::-1]\n    mask_sorted=np.array(mask)[idxs]\n    sum_mask_one=np.cumsum(mask_sorted)\n    IoU=sum_mask_one/(np.arange(1,len(mask_sorted)+1)+np.sum(mask_sorted)-sum_mask_one)\n    best_IoU_idx=IoU.argmax()\n    best_threshold=scores[idxs[best_IoU_idx]]\n    best_IoU=IoU[best_IoU_idx]\n\n    return best_threshold, best_IoU","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:53:56.663433Z","iopub.execute_input":"2024-11-08T18:53:56.663663Z","iopub.status.idle":"2024-11-08T18:53:56.670900Z","shell.execute_reply.started":"2024-11-08T18:53:56.663634Z","shell.execute_reply":"2024-11-08T18:53:56.669723Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(X_train.shape)\nprint(preds_train.shape)\nprint(Y_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:53:56.672459Z","iopub.execute_input":"2024-11-08T18:53:56.672839Z","iopub.status.idle":"2024-11-08T18:53:56.682519Z","shell.execute_reply.started":"2024-11-08T18:53:56.672791Z","shell.execute_reply":"2024-11-08T18:53:56.681624Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"get_threshold(Y_train[0], preds_train[0])","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:53:56.683769Z","iopub.execute_input":"2024-11-08T18:53:56.684325Z","iopub.status.idle":"2024-11-08T18:53:56.881296Z","shell.execute_reply.started":"2024-11-08T18:53:56.684290Z","shell.execute_reply":"2024-11-08T18:53:56.880355Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img_thresholds = []         # one for each image\nimg_IoUs = []\nfor Y, P in tqdm(zip(Y_train, preds_train), total=Y_train.shape[0]):\n\n    best_img_threshold, best_img_IoU = get_threshold(Y, P)\n    img_thresholds.append(best_img_threshold)\n    img_IoUs.append(best_img_IoU)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:53:56.882726Z","iopub.execute_input":"2024-11-08T18:53:56.883078Z","iopub.status.idle":"2024-11-08T18:55:46.627841Z","shell.execute_reply.started":"2024-11-08T18:53:56.883032Z","shell.execute_reply":"2024-11-08T18:55:46.626847Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_threshold = np.mean(img_thresholds)\nbest_threshold_spread = np.std(img_thresholds)\navg_IoU = mean(img_IoUs)\n\nprint(f\"Best threshold: {best_threshold:.3g} (+-{best_threshold_spread:.3g}), Avg. Train IoU: {avg_IoU:.3f}\")","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:46.629137Z","iopub.execute_input":"2024-11-08T18:55:46.629392Z","iopub.status.idle":"2024-11-08T18:55:46.643059Z","shell.execute_reply.started":"2024-11-08T18:55:46.629360Z","shell.execute_reply":"2024-11-08T18:55:46.641851Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dice_coefficient(Y_train, preds_train)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:46.644409Z","iopub.execute_input":"2024-11-08T18:55:46.644693Z","iopub.status.idle":"2024-11-08T18:55:48.347527Z","shell.execute_reply.started":"2024-11-08T18:55:46.644659Z","shell.execute_reply":"2024-11-08T18:55:48.346670Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_Y = (preds_train >= best_threshold)\n    \ndef plot(img_Y, img_pred):\n    output = np.zeros_like(img_Y)\n    output = np.where((img_Y == 0) & (img_pred == 1), 1, output)\n    output = np.where((img_Y == 1) & (img_pred == 0), 2, output)\n    output = np.where((img_Y == 1) & (img_pred == 1), 3, output)\n\n    plt.figure(figsize=(10,10))\n    plt.imshow(output, cmap=ListedColormap(['black', 'gray', 'orange', 'green']))\n    plt.xticks([])\n    plt.yticks([]);","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:48.348839Z","iopub.execute_input":"2024-11-08T18:55:48.349177Z","iopub.status.idle":"2024-11-08T18:55:48.492789Z","shell.execute_reply.started":"2024-11-08T18:55:48.349133Z","shell.execute_reply":"2024-11-08T18:55:48.491734Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"N = 5\nfor i in range(N):\n    img_Y = Y_train[i]\n    img_pred = pred_Y[i]\n    \n    plot(img_Y, img_pred)\n    plt.show()\n\n# green: correct prediction\n# gray: false positive (too much)\n# orange: false negative (missed)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:48.495029Z","iopub.execute_input":"2024-11-08T18:55:48.495336Z","iopub.status.idle":"2024-11-08T18:55:49.658885Z","shell.execute_reply.started":"2024-11-08T18:55:48.495301Z","shell.execute_reply":"2024-11-08T18:55:49.657969Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds_test = model.predict(X_test, verbose=1)\npreds_test_t = (preds_test >= best_threshold).astype(np.uint8)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:49.660398Z","iopub.execute_input":"2024-11-08T18:55:49.660744Z","iopub.status.idle":"2024-11-08T18:55:49.884116Z","shell.execute_reply.started":"2024-11-08T18:55:49.660698Z","shell.execute_reply":"2024-11-08T18:55:49.883214Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds_test_t[1].shape","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:49.885366Z","iopub.execute_input":"2024-11-08T18:55:49.885616Z","iopub.status.idle":"2024-11-08T18:55:49.891501Z","shell.execute_reply.started":"2024-11-08T18:55:49.885586Z","shell.execute_reply":"2024-11-08T18:55:49.890703Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Test samples\nfrom random import randint\nix = randint(0, len(preds_test_t)-1)\nprint(ix)\nplt.imshow(X_test[ix])\nplt.show()\nplt.imshow(np.squeeze(preds_test_t[ix]))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:49.892758Z","iopub.execute_input":"2024-11-08T18:55:49.893023Z","iopub.status.idle":"2024-11-08T18:55:50.297082Z","shell.execute_reply.started":"2024-11-08T18:55:49.892988Z","shell.execute_reply":"2024-11-08T18:55:50.296179Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(preds_test_t[0].shape)\nprint(preds_test_t[1].shape)\nprint(preds_test_t[2].shape)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:50.298447Z","iopub.execute_input":"2024-11-08T18:55:50.298700Z","iopub.status.idle":"2024-11-08T18:55:50.305150Z","shell.execute_reply.started":"2024-11-08T18:55:50.298667Z","shell.execute_reply":"2024-11-08T18:55:50.304039Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def check_overlap(msk):\n    msk = msk.astype(np.bool).astype(np.uint8)\n    return np.any(np.sum(msk, axis=-1)>1)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:50.306955Z","iopub.execute_input":"2024-11-08T18:55:50.307358Z","iopub.status.idle":"2024-11-08T18:55:50.317491Z","shell.execute_reply.started":"2024-11-08T18:55:50.307288Z","shell.execute_reply":"2024-11-08T18:55:50.316558Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for test_mask in preds_test_t:\n    print(check_overlap(test_mask))","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:50.319338Z","iopub.execute_input":"2024-11-08T18:55:50.319720Z","iopub.status.idle":"2024-11-08T18:55:50.333613Z","shell.execute_reply.started":"2024-11-08T18:55:50.319670Z","shell.execute_reply":"2024-11-08T18:55:50.332400Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# split the mask into each cluster nucleus for the submision\n# seen on https://www.kaggle.com/c/sartorius-cell-instance-segmentation/discussion/288376\ndef post_process(mask, min_size=80):\n    num_component, component = cv2.connectedComponents(mask.astype(np.uint8))\n    predictions = []\n    for c in range(1, num_component):\n        p = (component == c)\n        if p.sum() > min_size:\n            a_prediction = np.zeros((520, 704), np.float32)\n            a_prediction[p] = 1\n            predictions.append(a_prediction)\n    return predictions","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:50.335117Z","iopub.execute_input":"2024-11-08T18:55:50.335459Z","iopub.status.idle":"2024-11-08T18:55:50.345101Z","shell.execute_reply.started":"2024-11-08T18:55:50.335413Z","shell.execute_reply":"2024-11-08T18:55:50.343323Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# test the nucleus thing\nplt.imshow(Y_train[4], cmap=\"gray\");","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:50.347463Z","iopub.execute_input":"2024-11-08T18:55:50.347889Z","iopub.status.idle":"2024-11-08T18:55:50.597872Z","shell.execute_reply.started":"2024-11-08T18:55:50.347842Z","shell.execute_reply":"2024-11-08T18:55:50.596906Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_component, component = cv2.connectedComponents(Y_train[4].astype(np.uint8))\nnum_component","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:50.599195Z","iopub.execute_input":"2024-11-08T18:55:50.599427Z","iopub.status.idle":"2024-11-08T18:55:50.610657Z","shell.execute_reply.started":"2024-11-08T18:55:50.599399Z","shell.execute_reply":"2024-11-08T18:55:50.609731Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.imshow(component, cmap=\"gray\");","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:50.615466Z","iopub.execute_input":"2024-11-08T18:55:50.615728Z","iopub.status.idle":"2024-11-08T18:55:50.850019Z","shell.execute_reply.started":"2024-11-08T18:55:50.615696Z","shell.execute_reply":"2024-11-08T18:55:50.848956Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"compenent_1 = (component == 1)\nplt.imshow(compenent_1, cmap=\"gray\");","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:50.851197Z","iopub.execute_input":"2024-11-08T18:55:50.851450Z","iopub.status.idle":"2024-11-08T18:55:51.079206Z","shell.execute_reply.started":"2024-11-08T18:55:50.851416Z","shell.execute_reply":"2024-11-08T18:55:51.078105Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final = post_process(Y_train[4])\nfinal[0].shape","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:51.080895Z","iopub.execute_input":"2024-11-08T18:55:51.081210Z","iopub.status.idle":"2024-11-08T18:55:51.116630Z","shell.execute_reply.started":"2024-11-08T18:55:51.081173Z","shell.execute_reply":"2024-11-08T18:55:51.115589Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.imshow(final[0], cmap=\"gray\");","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:51.118102Z","iopub.execute_input":"2024-11-08T18:55:51.118435Z","iopub.status.idle":"2024-11-08T18:55:51.358652Z","shell.execute_reply.started":"2024-11-08T18:55:51.118392Z","shell.execute_reply":"2024-11-08T18:55:51.357783Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# old submision\npredicted2 = [rle_encode(test_mask2) for test_mask2 in preds_test_t]\nlen(predicted2[0])","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:51.360089Z","iopub.execute_input":"2024-11-08T18:55:51.360393Z","iopub.status.idle":"2024-11-08T18:55:51.386868Z","shell.execute_reply.started":"2024-11-08T18:55:51.360352Z","shell.execute_reply":"2024-11-08T18:55:51.386011Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def remove_isolated_points_from_rle(strin):\n    t2 = strin.split(\" \")\n    a = []\n    for i in range(0, len(t2), 2):\n        if t2[i+1]!=\"1\":\n            a.append(t2[i])\n            a.append(t2[i+1])\n    return ' '.join(a)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:51.388281Z","iopub.execute_input":"2024-11-08T18:55:51.388605Z","iopub.status.idle":"2024-11-08T18:55:51.394941Z","shell.execute_reply.started":"2024-11-08T18:55:51.388562Z","shell.execute_reply":"2024-11-08T18:55:51.393986Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predicted_filt = [remove_isolated_points_from_rle(s) for s in predicted2]","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:51.396367Z","iopub.execute_input":"2024-11-08T18:55:51.396707Z","iopub.status.idle":"2024-11-08T18:55:51.409660Z","shell.execute_reply.started":"2024-11-08T18:55:51.396663Z","shell.execute_reply":"2024-11-08T18:55:51.408798Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# new version with the mask nucleus split\npredicted_nucleus = []\ntest_nucleus_image_id = []\n\nfor index, s in enumerate(preds_test_t):\n    nucleus = post_process(s)\n    for nucl in nucleus:\n        predicted_nucleus.append(nucl)\n        test_nucleus_image_id.append(test_images_id[index])","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:51.411078Z","iopub.execute_input":"2024-11-08T18:55:51.411650Z","iopub.status.idle":"2024-11-08T18:55:51.945131Z","shell.execute_reply.started":"2024-11-08T18:55:51.411603Z","shell.execute_reply":"2024-11-08T18:55:51.943809Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.imshow(predicted_nucleus[0], cmap=\"gray\");","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:51.947671Z","iopub.execute_input":"2024-11-08T18:55:51.948021Z","iopub.status.idle":"2024-11-08T18:55:52.185550Z","shell.execute_reply.started":"2024-11-08T18:55:51.947984Z","shell.execute_reply":"2024-11-08T18:55:52.184418Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predicted2 = [rle_encode(test_mask2) for test_mask2 in predicted_nucleus]\nprint(predicted2[0])\npredicted_filt = [remove_isolated_points_from_rle(s) for s in predicted2]\nprint(predicted_filt[0])","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:52.187061Z","iopub.execute_input":"2024-11-08T18:55:52.187338Z","iopub.status.idle":"2024-11-08T18:55:52.493733Z","shell.execute_reply.started":"2024-11-08T18:55:52.187304Z","shell.execute_reply":"2024-11-08T18:55:52.492461Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submit = sample_submission.copy()\n#submit['predicted'] = predicted2\nsubmit = pd.DataFrame({'id':test_nucleus_image_id, 'predicted':predicted_filt})","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:52.495260Z","iopub.execute_input":"2024-11-08T18:55:52.495585Z","iopub.status.idle":"2024-11-08T18:55:52.503690Z","shell.execute_reply.started":"2024-11-08T18:55:52.495540Z","shell.execute_reply":"2024-11-08T18:55:52.502215Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(submit.shape)\nsubmit.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:52.505486Z","iopub.execute_input":"2024-11-08T18:55:52.505774Z","iopub.status.idle":"2024-11-08T18:55:52.533149Z","shell.execute_reply.started":"2024-11-08T18:55:52.505739Z","shell.execute_reply":"2024-11-08T18:55:52.531897Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submit.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T18:55:52.534695Z","iopub.execute_input":"2024-11-08T18:55:52.535470Z","iopub.status.idle":"2024-11-08T18:55:52.547944Z","shell.execute_reply.started":"2024-11-08T18:55:52.535385Z","shell.execute_reply":"2024-11-08T18:55:52.546651Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null}]}