{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":52279,"databundleVersionId":5822112,"sourceType":"competition"},{"sourceId":8380012,"sourceType":"datasetVersion","datasetId":4983219},{"sourceId":8576648,"sourceType":"datasetVersion","datasetId":4935313},{"sourceId":178264778,"sourceType":"kernelVersion"},{"sourceId":177132206,"sourceType":"kernelVersion"},{"sourceId":180621380,"sourceType":"kernelVersion"}],"dockerImageVersionId":30665,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"!pip install -U segmentation-models","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:34:03.195504Z","iopub.execute_input":"2024-06-01T11:34:03.196271Z","iopub.status.idle":"2024-06-01T11:34:15.387715Z","shell.execute_reply.started":"2024-06-01T11:34:03.196234Z","shell.execute_reply":"2024-06-01T11:34:15.386732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport pickle\nimport json, math\nimport numpy as np\nimport random, time\nimport pandas as pd\nfrom tqdm import tqdm\nfrom PIL import Image\nimport seaborn as sns\nimport shutil, sys, os, gc\nfrom datetime import datetime\nimport matplotlib.pyplot as plt\n\n\nimport tensorflow as tf\nfrom tensorflow.keras.utils import Sequence\nfrom sklearn.model_selection import train_test_split\n\n\nos.environ['SM_FRAMEWORK'] = 'tf.keras'\nimport segmentation_models as sm","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:23:57.178194Z","iopub.execute_input":"2024-06-01T11:23:57.178944Z","iopub.status.idle":"2024-06-01T11:24:09.682599Z","shell.execute_reply.started":"2024-06-01T11:23:57.178900Z","shell.execute_reply":"2024-06-01T11:24:09.681680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configs","metadata":{}},{"cell_type":"code","source":"class CFG:\n    # PATHS\n    tile_fpath = \"/kaggle/input/hubmap-human-vasculature-dataset-512512/HuPMap/kidney_tiles.csv\"\n    masks_dir = \"/kaggle/input/hubmap-human-vasculature-dataset-512512/HuPMap/masks/\"\n    images_dir = \"/kaggle/input/hubmap-human-vasculature-dataset-512512/HuPMap/images/\"\n\n    # Seeding for reproducibility\n    seed = 42\n\n    # Image Size\n    _shape = 512\n    image_size = (_shape, _shape)\n    \n    # optimizer\n    lr=2e-3\n    \n    # Batch Size & Epochs\n    epochs = 60\n    batch_size = 8\n\n    # Image prediction prob cutoff\n    cutoff = 0.6    \n    \n    # Model data\n    base_model = \"Linknet\"\n    encoder = \"efficientnetb5\" \n    model_name = \"{}_{}\".format(base_model, encoder)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:24:09.684358Z","iopub.execute_input":"2024-06-01T11:24:09.685051Z","iopub.status.idle":"2024-06-01T11:24:09.691101Z","shell.execute_reply.started":"2024-06-01T11:24:09.685014Z","shell.execute_reply":"2024-06-01T11:24:09.689947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utility Functions","metadata":{}},{"cell_type":"markdown","source":"## Kaggle directory","metadata":{}},{"cell_type":"code","source":"# to remove a file or a folder in kaggle directory\ndef delete_file_or_directory(path):\n    \"\"\"\n    Delete a file or directory if it exists.\n\n    Parameters:\n        path (str): Path to the file or directory.\n    \"\"\"\n    if os.path.exists(path):\n        if os.path.isfile(path):\n            os.remove(path)\n            print(f\"{path} has been deleted.\")\n        elif os.path.isdir(path):\n            shutil.rmtree(path)\n            print(f\"{path} and its contents have been deleted.\")\n    else:\n        print(f\"{path} does not exist.\")","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:24:09.692158Z","iopub.execute_input":"2024-06-01T11:24:09.692420Z","iopub.status.idle":"2024-06-01T11:24:09.717782Z","shell.execute_reply.started":"2024-06-01T11:24:09.692396Z","shell.execute_reply":"2024-06-01T11:24:09.717084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reading data","metadata":{}},{"cell_type":"code","source":"def read_mask(index, channel=2):\n    # loading mask\n    mask = np.load(f\"{CFG.masks_dir}{index}.npy\")\n    \n    # Select the specified channels\n    if channel == 1:\n        mask = mask[:, :, 0] # to consider only blood vessels\n    elif channel == 2:\n        selected_channels = mask[:, :, [0, 2]]\n        mask = np.sum(selected_channels, axis=2)\n    else:\n        pass\n    \n    # expanding dimension\n    if len(mask.shape) != 3:\n        mask = np.expand_dims(mask, axis=-1)\n        mask = np.where(mask > 0, 1, 0).astype(np.uint8)\n\n    return mask\n\n# to read data\ndef read_image_mask(df):\n    # Loading data\n    indexes = df['id'].to_numpy()\n    x = np.array([np.load(CFG.images_dir + index + \".npy\") for index in tqdm(indexes, desc=\"Loading images\")])\n    y = np.array([read_mask(index) for index in tqdm(indexes, desc=\"Loading masks\")])\n    gc.collect()\n    \n    return x, y","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:24:09.719851Z","iopub.execute_input":"2024-06-01T11:24:09.720180Z","iopub.status.idle":"2024-06-01T11:24:09.729392Z","shell.execute_reply.started":"2024-06-01T11:24:09.720154Z","shell.execute_reply":"2024-06-01T11:24:09.728541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model history","metadata":{}},{"cell_type":"code","source":"# Usage example save_history(HISTORY, 'history.json')\ndef save_history(history, file_path):\n    with open(file_path, 'w') as file:\n        json.dump(history, file)\n\n# Usage example loaded_history = load_history('history.json') \ndef load_history(file_path):\n    with open(file_path, 'r') as file:\n        history = json.load(file)\n    return history\n        \n# to plot model hsitory\ndef plot_history(history):\n    fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(15, 5))\n\n    # Plot loss\n    ax1.plot(history[\"loss\"], label=\"train loss\", color=\"tab:blue\")\n    ax1.plot(history[\"val_loss\"], label=\"val loss\", color=\"tab:orange\")\n    ax1.set_xlabel('Epoch')\n    ax1.set_ylabel('Loss')\n    ax1.tick_params(axis='y')\n    ax1.legend(loc=\"upper left\")\n\n    # Plot IoU\n    ax2.plot(history[\"iou_score\"], label=\"train IoU\", color=\"tab:green\")\n    ax2.plot(history[\"val_iou_score\"], label=\"val IoU\", color=\"tab:red\")\n    ax2.set_ylabel('IoU')\n    ax2.tick_params(axis='y')\n    ax2.legend(loc=\"upper right\")\n\n    fig.tight_layout()  # Ensure the labels do not overlap\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:24:09.730506Z","iopub.execute_input":"2024-06-01T11:24:09.730825Z","iopub.status.idle":"2024-06-01T11:24:09.743852Z","shell.execute_reply.started":"2024-06-01T11:24:09.730794Z","shell.execute_reply":"2024-06-01T11:24:09.743056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model data generation","metadata":{}},{"cell_type":"code","source":"def augment(X, Y):    \n    # 1. random flip--------------\n    # 1.1 horizontal\n    if np.random.uniform() > 0.5:\n        X = np.fliplr(X)\n        Y = np.fliplr(Y)\n        \n    # 1.2 vertical\n    if np.random.uniform() > 0.5:\n        X = np.flipud(X)\n        Y = np.flipud(Y)\n    \n    # 2. rotation------------------\n    # 2.1 set angle \n    angle = np.random.randint(4)\n    X = np.rot90(X, k=angle)\n    Y = np.rot90(Y, k=angle)\n    \n    # 3. Translatiion --------------\n    if np.random.uniform() > 0.5:\n        max_translation=(30, 30)\n        dx = np.random.randint(-max_translation[0], max_translation[0] + 1)\n        dy = np.random.randint(-max_translation[1], max_translation[1] + 1)\n\n        # Translate the image and mask using OpenCV\n        rows, cols = X.shape[:2]\n        M = np.float32([[1, 0, dx], [0, 1, dy]])\n        X = cv2.warpAffine(X, M, (cols, rows))\n        Y = cv2.warpAffine(Y, M, (cols, rows))\n\n    \n    if len(Y.shape) == 2:\n        Y = np.expand_dims(Y, axis=-1)\n    \n    return X, Y\n\ndef augment_batch(X_batch, Y_batch, augment):\n    augmented_X_batch, augmented_Y_batch = [], []\n    for X, Y in zip(X_batch, Y_batch):\n        augmented_X, augmented_Y = augment(X, Y)\n        augmented_X_batch.append(augmented_X)\n        augmented_Y_batch.append(augmented_Y)\n        \n    return np.array(augmented_X_batch), np.array(augmented_Y_batch)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:24:09.744992Z","iopub.execute_input":"2024-06-01T11:24:09.745309Z","iopub.status.idle":"2024-06-01T11:24:09.756345Z","shell.execute_reply.started":"2024-06-01T11:24:09.745274Z","shell.execute_reply":"2024-06-01T11:24:09.755455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generates data on which model trains on\nclass DataGenerator(Sequence):\n    def __init__(self, df, normalization, transform=None, batch_size=CFG.batch_size,\n                 images_dir=CFG.images_dir, masks_dir=CFG.masks_dir, **kwargs):\n        super().__init__(**kwargs)\n        self.df = df\n        self.masks_dir = masks_dir\n        self.images_dir = images_dir\n        self.batch_size = batch_size\n        self.transform = transform\n        self.preprocess_feature = normalization\n\n    def __len__(self):\n        return self.df.shape[0] // self.batch_size\n\n    def read_mask(self, index):\n        # loading mask\n        mask = np.load(self.masks_dir + index + \".npy\")\n\n        # Select the specified channels\n    #     mask = mask[:, :, 0] # to consider only blood vessels\n        selected_channels = mask[:, :, [0, 2]]\n        mask = np.sum(selected_channels, axis=2)\n\n        # expanding dimension\n        mask = np.expand_dims(mask, axis=-1)\n        mask = np.where(mask > 0, 1, 0).astype(np.uint8)\n\n        return mask\n    \n    def __getitem__(self, index):\n        # Load images, and masks\n        batch_df = self.df[index * self.batch_size: (index + 1) * self.batch_size]\n        X = [np.load(self.images_dir + row['id'] + \".npy\") for i, row in batch_df.iterrows()]\n        y = [self.read_mask(row['id']) for i, row in batch_df.iterrows()]\n\n        # preprocessing\n        X = self.preprocess_feature(np.array(X))\n        y = np.array(y, dtype=np.float32)\n\n        # augmentation\n        if self.transform:\n            X, y = self.transform[1](X, y, self.transform[0])\n            \n        gc.collect()\n        return X, y","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:24:09.757542Z","iopub.execute_input":"2024-06-01T11:24:09.758081Z","iopub.status.idle":"2024-06-01T11:24:09.770294Z","shell.execute_reply.started":"2024-06-01T11:24:09.758048Z","shell.execute_reply":"2024-06-01T11:24:09.769391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load DataFrame","metadata":{}},{"cell_type":"code","source":"# read csv file\ntile_df = pd.read_csv(CFG.tile_fpath)\n\n# ignore blank masks, and select some features\ntile_df = tile_df[tile_df['annotated'] == 1]\ntile_df = tile_df[(tile_df['blood_vessel'] > 0) | (tile_df['unsure'] > 0)]\ntile_df = tile_df[['id', 'source_wsi', 'dataset', 'dataset_wsi', 'blood_vessel', 'glomerulus', 'unsure']]\n\nprint(tile_df.shape)\ntile_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:24:09.771431Z","iopub.execute_input":"2024-06-01T11:24:09.772177Z","iopub.status.idle":"2024-06-01T11:24:09.837458Z","shell.execute_reply.started":"2024-06-01T11:24:09.772145Z","shell.execute_reply":"2024-06-01T11:24:09.836649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plotting samples","metadata":{}},{"cell_type":"code","source":" # To overlay mask onto its image\ndef overlay_mask(image, mask, opacity=0.80):\n    if np.max(mask) == 0:\n        return image.astype(np.uint8)  # Return the original image if the mask is blank (all zeros)\n    \n    alpha = mask[:, :, 0] * opacity  # Extract the single channel from the mask & Adjust the opacity by multiplying with a factor\n    alpha = alpha[:, :, np.newaxis]   # Add a third dimension to make it compatible with the image\n    result = alpha * mask + (1 - alpha) * image\n    \n    return result.astype(np.uint8)\n\n# Displaying image, mask, and mask overlayed onto image.\ndef show_random_sample(df):\n    fig, ax = plt.subplots(1,3, figsize = (10, 10))\n    \n    #getting random image     \n    _idx = df['id'].to_numpy()[random.randint(0, df.shape[0] - 1)]\n    _img =  np.load(CFG.images_dir + _idx + \".npy\")\n    _mask = read_mask(_idx)\n    _mask[_mask > 1] = 0  \n    _overlay = overlay_mask(_img, _mask)\n\n    ax[0].imshow(_img)\n    ax[1].imshow(_overlay)\n    ax[2].imshow(_mask, cmap='gray')\n\n    ax[0].set_title(\"Image ({})\".format(_idx))\n    ax[1].set_title(\"Overlayed Image\")\n    ax[2].set_title(\"pixel wise label\")","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:24:09.838472Z","iopub.execute_input":"2024-06-01T11:24:09.838733Z","iopub.status.idle":"2024-06-01T11:24:09.847420Z","shell.execute_reply.started":"2024-06-01T11:24:09.838709Z","shell.execute_reply":"2024-06-01T11:24:09.846614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Displaying random image\nshow_random_sample(tile_df)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:24:09.849496Z","iopub.execute_input":"2024-06-01T11:24:09.849770Z","iopub.status.idle":"2024-06-01T11:24:10.664660Z","shell.execute_reply.started":"2024-06-01T11:24:09.849748Z","shell.execute_reply":"2024-06-01T11:24:10.663736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Displaying random image\nshow_random_sample(tile_df)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:24:10.666021Z","iopub.execute_input":"2024-06-01T11:24:10.666284Z","iopub.status.idle":"2024-06-01T11:24:11.459434Z","shell.execute_reply.started":"2024-06-01T11:24:10.666260Z","shell.execute_reply":"2024-06-01T11:24:11.458480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratory data analysis. (E.D.A.)","metadata":{}},{"cell_type":"code","source":"# To dispaly data distribution\nselect = [\"dataset\", \"source_wsi\", \"dataset_wsi\"]\n\nsum_data_df = tile_df[select].groupby(select[-1]).count()\nsum_data_df.plot(kind = \"bar\", title = \"count of (dataset, source wsi)\", color=[\"#4682B4\", \"#4682B4\"])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:24:11.461249Z","iopub.execute_input":"2024-06-01T11:24:11.461535Z","iopub.status.idle":"2024-06-01T11:24:11.764983Z","shell.execute_reply.started":"2024-06-01T11:24:11.461510Z","shell.execute_reply":"2024-06-01T11:24:11.764037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# To dispaly data distribution\nselect = [\"id\", \"dataset\"]\n\nsum_data_df = tile_df[select].groupby(select[-1]).count()\nsum_data_df.plot(kind = \"bar\", title = \"count of dataset\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:24:11.765966Z","iopub.execute_input":"2024-06-01T11:24:11.766228Z","iopub.status.idle":"2024-06-01T11:24:11.968682Z","shell.execute_reply.started":"2024-06-01T11:24:11.766204Z","shell.execute_reply":"2024-06-01T11:24:11.967944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data preparation","metadata":{}},{"cell_type":"code","source":"# model backbone\npreprocess_input = sm.get_preprocessing(CFG.encoder)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:24:12.776940Z","iopub.execute_input":"2024-06-01T11:24:12.777762Z","iopub.status.idle":"2024-06-01T11:24:12.781776Z","shell.execute_reply.started":"2024-06-01T11:24:12.777726Z","shell.execute_reply":"2024-06-01T11:24:12.780833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# spliting data and perserving the same ratio of each class\nX_train, X_val = train_test_split(tile_df, test_size=0.2, random_state=CFG.seed, stratify=tile_df['dataset_wsi'])\nprint(X_train.shape, X_val.shape)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:24:12.942681Z","iopub.execute_input":"2024-06-01T11:24:12.943021Z","iopub.status.idle":"2024-06-01T11:24:12.953932Z","shell.execute_reply.started":"2024-06-01T11:24:12.942993Z","shell.execute_reply":"2024-06-01T11:24:12.952907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating data generator instances\ntrain_generator = DataGenerator(X_train, transform=(augment, augment_batch), normalization=preprocess_input)\nval_generator = DataGenerator(X_val, normalization=preprocess_input)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:24:16.424573Z","iopub.execute_input":"2024-06-01T11:24:16.424939Z","iopub.status.idle":"2024-06-01T11:24:16.429663Z","shell.execute_reply.started":"2024-06-01T11:24:16.424910Z","shell.execute_reply":"2024-06-01T11:24:16.428653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display samples of the DataGenerator images\nx, y = train_generator[0]\nprint(x.shape, y.shape)\nplt.imshow(x[0])","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:24:16.760698Z","iopub.execute_input":"2024-06-01T11:24:16.761073Z","iopub.status.idle":"2024-06-01T11:24:17.627012Z","shell.execute_reply.started":"2024-06-01T11:24:16.761044Z","shell.execute_reply":"2024-06-01T11:24:17.626150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Linknet Model","metadata":{}},{"cell_type":"code","source":"# Constants\nHISTORY = {}\nBEST_MODEL_PATH=F\"best_{CFG.model_name}_weights.keras\"","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:24:18.732708Z","iopub.execute_input":"2024-06-01T11:24:18.733363Z","iopub.status.idle":"2024-06-01T11:24:18.737656Z","shell.execute_reply.started":"2024-06-01T11:24:18.733329Z","shell.execute_reply":"2024-06-01T11:24:18.736654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model history\ntargets = ['iou_score', 'loss', 'val_iou_score', 'val_loss']\n\nfor target in targets:\n    HISTORY[target] = []","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:26:21.646571Z","iopub.execute_input":"2024-06-01T11:26:21.647177Z","iopub.status.idle":"2024-06-01T11:26:21.651618Z","shell.execute_reply.started":"2024-06-01T11:26:21.647141Z","shell.execute_reply":"2024-06-01T11:26:21.650706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ModelCheckpoint\n\n# checkpoint callback\ncheckpoint = ModelCheckpoint(BEST_MODEL_PATH, monitor='val_loss', verbose=0, save_best_only=True, mode='min')\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:26:23.187225Z","iopub.execute_input":"2024-06-01T11:26:23.187936Z","iopub.status.idle":"2024-06-01T11:26:23.395862Z","shell.execute_reply.started":"2024-06-01T11:26:23.187901Z","shell.execute_reply":"2024-06-01T11:26:23.394933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define model\nmodel = sm.Linknet(CFG.encoder, encoder_weights='imagenet', decoder_use_batchnorm=True, classes=1, activation='sigmoid')\nmodel.compile(\n    tf.keras.optimizers.AdamW(learning_rate=CFG.lr),\n    loss=sm.losses.bce_jaccard_loss,\n    metrics=[sm.metrics.iou_score],\n)\n\n# Loading weights (pretrained on similar data)\nmodel.load_weights(\"/kaggle/input/hubmap-linkn-segmentation-no-transfer/best_Linknet_efficientnetb5_weights.keras\")\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:58:47.443269Z","iopub.execute_input":"2024-06-01T11:58:47.443673Z","iopub.status.idle":"2024-06-01T11:59:22.327144Z","shell.execute_reply.started":"2024-06-01T11:58:47.443641Z","shell.execute_reply":"2024-06-01T11:59:22.326149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Training","metadata":{}},{"cell_type":"code","source":"HISTORY = load_history('/kaggle/input/hubmap-linkn-segmentation-no-transfer/training_history.json')","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:59:22.329106Z","iopub.execute_input":"2024-06-01T11:59:22.329572Z","iopub.status.idle":"2024-06-01T11:59:22.338710Z","shell.execute_reply.started":"2024-06-01T11:59:22.329536Z","shell.execute_reply":"2024-06-01T11:59:22.337780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # fit model\n# model.fit(\n#    train_generator,\n#    batch_size=CFG.batch_size,\n#    epochs=CFG.epochs,\n#    validation_data=val_generator,\n#     callbacks=[checkpoint]\n# )","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:27:08.657361Z","iopub.execute_input":"2024-06-01T11:27:08.657608Z","iopub.status.idle":"2024-06-01T11:27:08.661789Z","shell.execute_reply.started":"2024-06-01T11:27:08.657585Z","shell.execute_reply":"2024-06-01T11:27:08.660901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # saving model history\n# for key in ['iou_score', 'loss', 'val_loss', 'val_iou_score']:\n#     HISTORY[key].extend(model.history.history[key])\n\n# save_history(HISTORY, 'training_history.json')","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:27:59.083992Z","iopub.execute_input":"2024-06-01T11:27:59.084837Z","iopub.status.idle":"2024-06-01T11:27:59.122927Z","shell.execute_reply.started":"2024-06-01T11:27:59.084801Z","shell.execute_reply":"2024-06-01T11:27:59.121682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plotting model history loss, and metric\nplot_history(HISTORY)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:59:33.122529Z","iopub.execute_input":"2024-06-01T11:59:33.123212Z","iopub.status.idle":"2024-06-01T11:59:33.548987Z","shell.execute_reply.started":"2024-06-01T11:59:33.123181Z","shell.execute_reply":"2024-06-01T11:59:33.548025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # saving model weights\n# model.save_weights(\"{}_{}.weights.h5\".format(CFG.model_name, \"_\"))","metadata":{"execution":{"iopub.status.busy":"2024-05-11T15:49:33.378955Z","iopub.execute_input":"2024-05-11T15:49:33.379361Z","iopub.status.idle":"2024-05-11T15:49:35.288729Z","shell.execute_reply.started":"2024-05-11T15:49:33.379329Z","shell.execute_reply":"2024-05-11T15:49:35.287722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Evaluation","metadata":{}},{"cell_type":"code","source":"# model.predict method\ndef predict_masks(model, images):\n    # Ensure the input image has the right shape for prediction\n    if len(images.shape) == 3:\n        images = np.expand_dims(images, axis=0)  # Add batch dimension if needed\n\n    # Predict the probabilities for the input image\n    images = preprocess_input(images)\n    prob = model.predict(images)\n\n    # Return the predicted mask\n    return prob\n\n# model.predict method\ndef predict_masks(model, images):\n    # Ensure the input image has the right shape for prediction\n    if len(images.shape) == 3:\n        images = np.expand_dims(images, axis=0)  # Add batch dimension if needed\n\n    # Predict the probabilities for the input image\n    images = preprocess_input(images)\n    prob = model.predict(images)\n\n    # Return the predicted mask\n    return prob\n\ndef calculate_metrics(y_true, y_pred, threshold):\n    y_pred_binary = (y_pred > threshold).astype(np.uint8)\n\n    # True Positives, False Positives, False Negatives, True Negatives\n    TP = np.sum((y_true == 1) & (y_pred_binary == 1))\n    FP = np.sum((y_true == 0) & (y_pred_binary == 1))\n    TN = np.sum((y_true == 0) & (y_pred_binary == 0))\n    FN = np.sum((y_true == 1) & (y_pred_binary == 0))\n\n    # Dice coefficient\n    dice_denominator = 2 * TP + FP + FN\n    dice = (2 * TP) / dice_denominator if dice_denominator != 0 else 1\n\n    # Intersection over Union (IoU)\n    iou_denominator = TP + FP + FN\n    iou = TP / iou_denominator if iou_denominator != 0 else 1\n\n    # Precision\n    precision = TP / (TP + FP) if (TP + FP) != 0 else 1\n\n    # Recall\n    recall = TP / (TP + FN) if (TP + FN) != 0 else 1\n    \n    # F1 Score\n    f1_score = 2 * ((precision * recall) / (precision + recall)) if (precision + recall) != 0 else 1\n\n    # Confidence\n    binary_mask = y_pred > threshold\n    confidence_scores = y_pred.flatten()\n    binary_mask_flat = binary_mask.flatten()\n    blood_vessel_confidences = confidence_scores[binary_mask_flat]\n    confidence = np.mean(blood_vessel_confidences)\n\n    return dice, iou, precision, recall, f1_score, confidence\n\n\ndef metrics_dataframe(Y, Y_hat, threshold=CFG.cutoff):\n    n_val = len(Y)\n    df_object = {}\n    df_object['dice'] = []\n    df_object['iou'] = []\n    df_object['precision'] = []\n    df_object['recall'] = []    \n    df_object['f1_score'] = []\n    df_object['confidence'] = []\n    df_object['threshold'] = threshold\n\n    for i in tqdm(range(n_val), total=n_val):\n        metrics = calculate_metrics(Y[i], Y_hat[i], threshold)\n        df_object['dice'].append(metrics[0])\n        df_object['iou'].append(metrics[1])\n        df_object['precision'].append(metrics[2])\n        df_object['recall'].append(metrics[3])        \n        df_object['f1_score'].append(metrics[4])\n        df_object['confidence'].append(metrics[5])\n        \n    return pd.DataFrame(df_object)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:59:37.916005Z","iopub.execute_input":"2024-06-01T11:59:37.916929Z","iopub.status.idle":"2024-06-01T11:59:37.932822Z","shell.execute_reply.started":"2024-06-01T11:59:37.916896Z","shell.execute_reply":"2024-06-01T11:59:37.931967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # load best model weights\n# model.load_weights(f\"/kaggle/working/{BEST_MODEL_PATH}\")","metadata":{"execution":{"iopub.status.busy":"2024-05-11T15:49:39.63716Z","iopub.execute_input":"2024-05-11T15:49:39.637839Z","iopub.status.idle":"2024-05-11T15:50:20.364783Z","shell.execute_reply.started":"2024-05-11T15:49:39.637807Z","shell.execute_reply":"2024-05-11T15:50:20.363901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# loading validation data\nX, Y = read_image_mask(X_val)\n\n# making prediction\nY_hat = predict_masks(model, X)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T11:59:56.620476Z","iopub.execute_input":"2024-06-01T11:59:56.620855Z","iopub.status.idle":"2024-06-01T12:00:29.423316Z","shell.execute_reply.started":"2024-06-01T11:59:56.620818Z","shell.execute_reply":"2024-06-01T12:00:29.422402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def print_metrics(threshold=CFG.cutoff):\n    n = len(Y)\n    iou, dice = [], []\n    \n    for i in tqdm(range(n)):\n        metrics = calculate_metrics(Y[i], Y_hat[i], threshold)\n        iou.append(metrics[1])\n        dice.append(metrics[0])\n    \n    iou = np.round(np.mean(iou ) * 100, 4)\n    dice = np.round(np.mean(dice) * 100, 4)\n    \n    print(\"threshold {}% - IoU score {}% - Dice coefficient {}%\"\n          .format(threshold*100, np.mean(iou),np.mean(dice)))   \n    \n    return threshold, dice\n    \nmax_dice = -1\nbest_threshold = 0\nfor i in range(50, 100, 5):\n    threshold, dice = print_metrics(i / 100)\n    if dice > max_dice:\n        max_dice = dice\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2024-06-01T12:00:29.425570Z","iopub.execute_input":"2024-06-01T12:00:29.425863Z","iopub.status.idle":"2024-06-01T12:00:33.805114Z","shell.execute_reply.started":"2024-06-01T12:00:29.425837Z","shell.execute_reply":"2024-06-01T12:00:33.804142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metrics_df = metrics_dataframe(Y, Y_hat, best_threshold)\nmetrics_df.to_csv('metrics_dataframe.csv', index=False)\nmetrics_df.mean()","metadata":{"execution":{"iopub.status.busy":"2024-06-01T12:00:33.806324Z","iopub.execute_input":"2024-06-01T12:00:33.806621Z","iopub.status.idle":"2024-06-01T12:00:34.265337Z","shell.execute_reply.started":"2024-06-01T12:00:33.806595Z","shell.execute_reply":"2024-06-01T12:00:34.264458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# IOU boxplot using Seaborn\nsns.boxplot(x=metrics_df['iou'])\n\n# Show the plot\nplt.title('IOU Scores Boxplot')\nplt.xlabel('IOU Scores')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-01T12:00:34.267017Z","iopub.execute_input":"2024-06-01T12:00:34.267304Z","iopub.status.idle":"2024-06-01T12:00:34.393971Z","shell.execute_reply.started":"2024-06-01T12:00:34.267279Z","shell.execute_reply":"2024-06-01T12:00:34.393017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dice boxplot using Seaborn\nsns.boxplot(x=metrics_df['dice'])\n\n# Show the plot\nplt.title('Dice Scores Boxplot')\nplt.xlabel('Dice Scores')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-01T12:02:21.737800Z","iopub.execute_input":"2024-06-01T12:02:21.738212Z","iopub.status.idle":"2024-06-01T12:02:21.868418Z","shell.execute_reply.started":"2024-06-01T12:02:21.738177Z","shell.execute_reply":"2024-06-01T12:02:21.867470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plotting Predictions","metadata":{}},{"cell_type":"code","source":"def plot_result(X, Y_true, Y_pred, cutoff = CFG.cutoff):\n    \n    N = X.shape[0]\n    \n    fig, ax = plt.subplots(N,4, figsize = (10, 3*N))\n    \n    for k in range(N):\n        \n        cutoff_img1 = (Y_pred[k,:,:,0] > cutoff).astype(int)\n\n        true_img = np.zeros((512, 512, 3), dtype = np.uint8)\n        true_img[:,:,1] = Y_true[k,:,:,0]*200\n        \n        cutoff1 = np.zeros((512, 512, 3), dtype = np.uint8)\n        \n        cutoff1[:,:,0] = cutoff_img1*230\n        \n        cutoff1[:,:,1] = cutoff_img1*50\n        cutoff1[:,:,2] = cutoff_img1*50\n        \n        diff_photo1 = cutoff1.copy()\n        diff_photo1[:,:,1] += (Y_true[k,:,:,0]*200).astype(np.uint8)\n        \n        ax[k, 0].imshow(X[k])\n        ax[k, 1].imshow(true_img, cmap = \"gray\")\n        ax[k, 2].imshow(cutoff1, cmap = \"gray\")\n        ax[k, 3].imshow(diff_photo1)\n        \n        for j in range(4):\n            ax[k,j].set_xticks([])\n            ax[k,j].set_yticks([])\n    \n        if k == 0:\n            ax[k, 0].set_title(\"val img\")\n            ax[k, 1].set_title(\"true label\")\n            ax[k, 2].set_title(\"model (cutoff at {})\".format(cutoff))\n            ax[k, 3].set_title(\"Compare (Y:tp)\")","metadata":{"execution":{"iopub.status.busy":"2024-06-01T12:02:24.628590Z","iopub.execute_input":"2024-06-01T12:02:24.629270Z","iopub.status.idle":"2024-06-01T12:02:24.640336Z","shell.execute_reply.started":"2024-06-01T12:02:24.629232Z","shell.execute_reply":"2024-06-01T12:02:24.639418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_val = len(X)\nval_sample = np.random.choice(n_val, 20)\nplot_result(X[val_sample], Y[val_sample], Y_hat[val_sample], cutoff=CFG.cutoff)","metadata":{"execution":{"iopub.status.busy":"2024-06-01T12:03:17.223590Z","iopub.execute_input":"2024-06-01T12:03:17.224449Z","iopub.status.idle":"2024-06-01T12:03:24.329691Z","shell.execute_reply.started":"2024-06-01T12:03:17.224412Z","shell.execute_reply":"2024-06-01T12:03:24.328719Z"},"trusted":true},"execution_count":null,"outputs":[]}]}