{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Deep Learning - Brain Tumor Classification and Segmentation using ResNet50 and ResUnet\n##### data Source: https://www.kaggle.com/mateuszbuda/lgg-mri-segmentation\n\n\nIn this project, I followed a structured approach to analyze medical images for brain tumor classification and segmentation:\n1. The first step was data pre-processing, which involved normalizing pixel values to prepare them for modeling. \n\n2. Next, I utilized transfer learning by leveraging a pre-trained ResNet50 model to train a binary classifier with a remarkable accuracy of 92% in predicting the presence of cancer in patients. \n\n3. Since brain tumor segmentation is challenging due to the irregular shape and texture of tumors and their proximity to other brain structures, I developed a ResUnet model for tumor localization within the medical image. \n\n4. Finally, I evaluated both models to ensure their reliability and usability for future deployment.\n\n\n\n\n\n\n\n\n","metadata":{}},{"cell_type":"markdown","source":"# Importing necessary libraries and datasets\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd  # used for data manipulation and analysis\nimport numpy as np  # used for numerical computation\nimport seaborn as sns  # used for data visualization\nimport matplotlib.pyplot as plt  # used for plotting graphs and charts\nimport plotly.graph_objects as go  # use plotly to plot interactive bar chart\nimport zipfile  # used for working with zip files\nimport cv2  # used for computer vision tasks\nfrom skimage import io  # used for image processing\nimport tensorflow as tf  # used for building and training machine learning models\nfrom tensorflow.python.keras import Sequential  # used for building sequential models\n\nfrom sklearn.model_selection import (\n    train_test_split,\n)  # used to split the data into train and test data\n\nfrom tensorflow.keras import (\n    layers,\n    optimizers,\n)  # used for defining layers and optimizers\n\nfrom tensorflow.keras.applications import DenseNet121  # pre-trained deep learning model\nfrom tensorflow.keras.applications.resnet50 import (\n    ResNet50,\n)  # pre-trained deep learning model\n\nfrom tensorflow.keras.layers import *  # importing all layers from Keras\nfrom tensorflow.keras.models import (\n    Model,\n    load_model,\n)  # modules for creating and loading models\n\nfrom tensorflow.keras.initializers import (\n    glorot_uniform,\n)  # initializing weights of neural network\n\nfrom tensorflow.keras.utils import plot_model  # module for plotting model architecture\nfrom tensorflow.keras.callbacks import (\n    ReduceLROnPlateau,\n    EarlyStopping,\n    ModelCheckpoint,\n    LearningRateScheduler,\n)  # callbacks for model training\n\nfrom IPython.display import display  # module for displaying output in notebook\nfrom tensorflow.keras import (\n    backend as KAA,\n)  # module for managing backend operations in Keras\n\nfrom sklearn.preprocessing import (\n    StandardScaler,\n    normalize,\n)  # modules for scaling and normalization\n\nfrom utilities import (\n    focal_tversky,\n    tversky_loss,\n    tversky,\n    DataGenerator,\n)  # custom utilities for loss function\n\nimport os  # used for interacting with operating system\nimport glob  # used for pattern matching files\nimport random  # used for generating random numbers","metadata":{"execution":{"iopub.status.busy":"2023-04-25T20:03:15.207450Z","iopub.execute_input":"2023-04-25T20:03:15.207904Z","iopub.status.idle":"2023-04-25T20:03:15.218924Z","shell.execute_reply.started":"2023-04-25T20:03:15.207863Z","shell.execute_reply":"2023-04-25T20:03:15.217966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data EDA","metadata":{}},{"cell_type":"markdown","source":"##### \n- installation of packages for image preprocessing utilities and automatic formatting of code cells in Jupyter ","metadata":{}},{"cell_type":"code","source":"!pip install keras_preprocessing","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:28:09.038390Z","iopub.execute_input":"2023-04-25T14:28:09.038862Z","iopub.status.idle":"2023-04-25T14:28:21.418222Z","shell.execute_reply.started":"2023-04-25T14:28:09.038821Z","shell.execute_reply":"2023-04-25T14:28:21.416982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras_preprocessing.image import (\n    ImageDataGenerator,\n)  # utility for generating batches of augmented image data in real-time during model training","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Install nb_black for autoformating\n!pip install nb_black --quiet\n%load_ext lab_black","metadata":{"execution":{"iopub.status.busy":"2023-04-24T15:18:01.242032Z","iopub.execute_input":"2023-04-24T15:18:01.243232Z","iopub.status.idle":"2023-04-24T15:18:16.845088Z","shell.execute_reply.started":"2023-04-24T15:18:01.243183Z","shell.execute_reply":"2023-04-24T15:18:16.843574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"brain_df = pd.read_csv(\"../input/lgg-mri-segmentation/kaggle_3m/data.csv\")\nbrain_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:28:42.891079Z","iopub.execute_input":"2023-04-25T14:28:42.891612Z","iopub.status.idle":"2023-04-25T14:28:42.938078Z","shell.execute_reply.started":"2023-04-25T14:28:42.891563Z","shell.execute_reply":"2023-04-25T14:28:42.936732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"brain_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:28:54.956766Z","iopub.execute_input":"2023-04-25T14:28:54.957721Z","iopub.status.idle":"2023-04-25T14:28:55.025220Z","shell.execute_reply.started":"2023-04-25T14:28:54.957650Z","shell.execute_reply":"2023-04-25T14:28:55.023859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_map = []\nfor sub_dir_path in glob.glob(\"/kaggle/input/lgg-mri-segmentation/kaggle_3m/\"+\"*\"):\n    #if os.path.isdir(sub_path_dir):\n    try:\n        dir_name = sub_dir_path.split('/')[-1]\n        for filename in os.listdir(sub_dir_path):\n            image_path = sub_dir_path + '/' + filename\n            data_map.extend([dir_name, image_path])\n    except Exception as e:\n        print(e)\n\ndf = pd.DataFrame({\"patient_id\": data_map[::2], \"path\": data_map[1::2]})\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:29:34.159949Z","iopub.execute_input":"2023-04-25T14:29:34.160370Z","iopub.status.idle":"2023-04-25T14:29:35.340447Z","shell.execute_reply.started":"2023-04-25T14:29:34.160334Z","shell.execute_reply":"2023-04-25T14:29:35.339115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_imgs = df[~df[\"path\"].str.contains(\"mask\")]\ndf_masks = df[df[\"path\"].str.contains(\"mask\")]\n\n# File path line length images for later sorting\nBASE_LEN = 89  # len(/kaggle/input/lgg-mri-segmentation/kaggle_3m/TCGA_DU_6404_19850629/TCGA_DU_6404_19850629_ <-!!!43.tif)\nEND_IMG_LEN = 4  # len(/kaggle/input/lgg-mri-segmentation/kaggle_3m/TCGA_DU_6404_19850629/TCGA_DU_6404_19850629_43 !!!->.tif)\nEND_MASK_LEN = 9  # (/kaggle/input/lgg-mri-segmentation/kaggle_3m/TCGA_DU_6404_19850629/TCGA_DU_6404_19850629_43 !!!->_mask.tif)\n\n# Data sorting\nimgs = sorted(df_imgs[\"path\"].values, key=lambda x: int(x[BASE_LEN:-END_IMG_LEN]))\nmasks = sorted(df_masks[\"path\"].values, key=lambda x: int(x[BASE_LEN:-END_MASK_LEN]))\n\n# Sorting check\nidx = random.randint(0, len(imgs) - 1)\nprint(\"Path to the Image:\", imgs[idx], \"\\nPath to the Mask:\", masks[idx])","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:29:55.636212Z","iopub.execute_input":"2023-04-25T14:29:55.636663Z","iopub.status.idle":"2023-04-25T14:29:55.664274Z","shell.execute_reply.started":"2023-04-25T14:29:55.636623Z","shell.execute_reply":"2023-04-25T14:29:55.662784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Final dataframe\nbrain_df = pd.DataFrame(\n    {\"patient_id\": df_imgs.patient_id.values, \"image_path\": imgs, \"mask_path\": masks}\n)\n\n\ndef pos_neg_diagnosis(mask_path):\n    value = np.max(cv2.imread(mask_path))\n    if value > 0:\n        return 1\n    else:\n        return 0\n\n\nbrain_df[\"mask\"] = brain_df[\"mask_path\"].apply(lambda x: pos_neg_diagnosis(x))\nbrain_df","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:30:34.596155Z","iopub.execute_input":"2023-04-25T14:30:34.597003Z","iopub.status.idle":"2023-04-25T14:30:39.240945Z","shell.execute_reply.started":"2023-04-25T14:30:34.596968Z","shell.execute_reply":"2023-04-25T14:30:39.239876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PERFORM DATA VISUALIZATION","metadata":{}},{"cell_type":"markdown","source":"## Plot interactive bar chart:\n\n- x-axis values are the unique values in the 'mask' column of the 'brain_df' dataframe\n\n- y-axis values are the number of occurrences of each unique value in the 'mask' column.","metadata":{}},{"cell_type":"code","source":"fig = go.Figure(\n    [go.Bar(x=brain_df[\"mask\"].value_counts().index, y=brain_df[\"mask\"].value_counts())]\n)\nfig.update_traces(\n    marker_color=\"rgb(0,200,0)\",\n    marker_line_color=\"rgb(0,255,0)\",\n    marker_line_width=7,\n    opacity=0.6,\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:30:46.341169Z","iopub.execute_input":"2023-04-25T14:30:46.341627Z","iopub.status.idle":"2023-04-25T14:30:46.634950Z","shell.execute_reply.started":"2023-04-25T14:30:46.341586Z","shell.execute_reply":"2023-04-25T14:30:46.633975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Plot randomly selected (1) MRI scan images from only sick patients followed by corresponding mask","metadata":{}},{"cell_type":"code","source":"# Visualize the images (MRI - Mask)\nimport random\n\nfig, axs = plt.subplots(6, 2, figsize=(16, 32))\ncount = 0\nfor x in range(6):\n    i = random.randint(0, len(brain_df))\n    axs[count][0].title.set_text(\"Brain MRI\")\n    axs[count][0].imshow(cv2.imread(brain_df.image_path[i]))\n    axs[count][1].title.set_text(\"Mask - \" + str(brain_df[\"mask\"][i]))\n    axs[count][1].imshow(cv2.imread(brain_df.mask_path[i]))\n    count += 1\n\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:31:12.023279Z","iopub.execute_input":"2023-04-25T14:31:12.024067Z","iopub.status.idle":"2023-04-25T14:31:15.255856Z","shell.execute_reply.started":"2023-04-25T14:31:12.024017Z","shell.execute_reply":"2023-04-25T14:31:15.254783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize overlap of the images (MRI - Mask)\ncount = 0\nfig, axs = plt.subplots(12, 3, figsize = (20, 50))\nfor i in range(len(brain_df)):\n    if brain_df['mask'][i] ==1 and count <12:\n        img = io.imread(brain_df.image_path[i])\n        axs[count][0].title.set_text('Brain MRI')\n        axs[count][0].imshow(img)\n\n        mask = io.imread(brain_df.mask_path[i])\n        axs[count][1].title.set_text('Mask')\n        axs[count][1].imshow(mask, cmap = 'gray')\n\n\n        img[mask == 255] = (255, 0, 0)\n        axs[count][2].title.set_text('MRI with Mask')\n        axs[count][2].imshow(img)\n        count+=1\n\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:32:50.294137Z","iopub.execute_input":"2023-04-25T14:32:50.294599Z","iopub.status.idle":"2023-04-25T14:32:58.022120Z","shell.execute_reply.started":"2023-04-25T14:32:50.294543Z","shell.execute_reply":"2023-04-25T14:32:58.020649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train A Classifier Model To Detect If Tumor Exists Or Not","metadata":{}},{"cell_type":"code","source":"# Drop the patient id column\nbrain_df_train = brain_df.drop(columns = ['patient_id'])\nbrain_df_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:33:52.282706Z","iopub.execute_input":"2023-04-25T14:33:52.283140Z","iopub.status.idle":"2023-04-25T14:33:52.292831Z","shell.execute_reply.started":"2023-04-25T14:33:52.283104Z","shell.execute_reply":"2023-04-25T14:33:52.291471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert the data in mask column to string format, to use categorical mode in flow_from_dataframe\nbrain_df_train['mask'] = brain_df_train['mask'].apply(lambda x: str(x))\nbrain_df_train.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:34:45.167173Z","iopub.execute_input":"2023-04-25T14:34:45.167638Z","iopub.status.idle":"2023-04-25T14:34:45.188925Z","shell.execute_reply.started":"2023-04-25T14:34:45.167599Z","shell.execute_reply":"2023-04-25T14:34:45.187543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split the data into train and test data\ntrain, test = train_test_split(brain_df_train, test_size = 0.15)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:35:22.071143Z","iopub.execute_input":"2023-04-25T14:35:22.071900Z","iopub.status.idle":"2023-04-25T14:35:22.081025Z","shell.execute_reply.started":"2023-04-25T14:35:22.071854Z","shell.execute_reply":"2023-04-25T14:35:22.079627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create a image generator\n\n\n- initialization of an instance of ImageDataGenerator from Keras, which which will be used to perform data augmentation on image data.\n\n- pixel values in the images will be rescaled to the range of [0,1], making it easier for the model to learn from the data.","metadata":{}},{"cell_type":"code","source":"VAL_SPLIT = 0.15\nIMAGE_SIZE = 256\nIMAGE_CHAN = 3\nBATCH_SIZE = 16\n\n# data generator which scales the data from 0 to 1 and makes validation split of 0.15\ndatagen = ImageDataGenerator(rescale=1.0 / 255.0, validation_split=VAL_SPLIT)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:36:33.822963Z","iopub.execute_input":"2023-04-25T14:36:33.823394Z","iopub.status.idle":"2023-04-25T14:36:33.829794Z","shell.execute_reply.started":"2023-04-25T14:36:33.823357Z","shell.execute_reply":"2023-04-25T14:36:33.828522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# generators for training, validation, and testing sets of images and masks\n# (train validation and test data generators)\ntrain_generator = datagen.flow_from_dataframe(\n    dataframe=train,\n    directory=\"./\",\n    x_col=\"image_path\",\n    y_col=\"mask\",\n    subset=\"training\",\n    batch_size=BATCH_SIZE,\n    shuffle=True,\n    class_mode=\"categorical\",\n    target_size=(IMAGE_SIZE, IMAGE_SIZE),\n)\n\n\nvalid_generator = datagen.flow_from_dataframe(\n    dataframe=train,\n    directory=\"./\",\n    x_col=\"image_path\",\n    y_col=\"mask\",\n    subset=\"validation\",\n    batch_size=BATCH_SIZE,\n    shuffle=True,\n    class_mode=\"categorical\",\n    target_size=(IMAGE_SIZE, IMAGE_SIZE),\n)\n\ntest_datagen = ImageDataGenerator(rescale=1.0 / 255.0)\n\ntest_generator = test_datagen.flow_from_dataframe(\n    dataframe=test,\n    directory=\"./\",\n    x_col=\"image_path\",\n    y_col=\"mask\",\n    batch_size=BATCH_SIZE,\n    shuffle=False,\n    class_mode=\"categorical\",\n    target_size=(IMAGE_SIZE, IMAGE_SIZE),\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:36:39.987913Z","iopub.execute_input":"2023-04-25T14:36:39.988700Z","iopub.status.idle":"2023-04-25T14:36:41.399963Z","shell.execute_reply.started":"2023-04-25T14:36:39.988658Z","shell.execute_reply":"2023-04-25T14:36:41.398634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Developing a Classifier Model To Detect Brain Tumor:\n\n#### Understanding Intuition behind Resnets\n- As CNNs grow deeper, vanishing gradient tend to occur which negatively impact network performance.\n- Vanishing gradient problem occurs when the gradient is back-propagated to earlier layers which results in a very small gradient.\n- Residual Neural Network includes “skip connection” feature which enables training of 152 layers without vanishing gradient issues.\n- Resnet works by adding “identity mappings” on top of the CNN.\n- ImageNet contains 11 million images and 11,000 categories\n\n\n\n#### Get the ResNet50 model (Pretrained on imagenet data)\n- Transfer learning is a machine learning technique in which a network that has been trained to perform a specific task is being reused (repurposed) as a starting point for another similar task.\n\n- We'll freeze the trained CNN network weights from the first layersand only train the newly added dense layers (with randomly initialized weights).\n","metadata":{}},{"cell_type":"code","source":"# Get the ResNet50 base model\nbasemodel = ResNet50(\n    weights=\"imagenet\",\n    include_top=False,\n    input_tensor=Input(shape=(IMAGE_SIZE, IMAGE_SIZE, IMAGE_CHAN)),\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:37:05.753899Z","iopub.execute_input":"2023-04-25T14:37:05.754307Z","iopub.status.idle":"2023-04-25T14:37:11.972381Z","shell.execute_reply.started":"2023-04-25T14:37:05.754274Z","shell.execute_reply":"2023-04-25T14:37:11.970922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"basemodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:37:18.080933Z","iopub.execute_input":"2023-04-25T14:37:18.082213Z","iopub.status.idle":"2023-04-25T14:37:18.634304Z","shell.execute_reply.started":"2023-04-25T14:37:18.082149Z","shell.execute_reply":"2023-04-25T14:37:18.633120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# freeze the model weights\nfor layer in basemodel.layers:\n    layers.trainable = False","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:37:36.963826Z","iopub.execute_input":"2023-04-25T14:37:36.964297Z","iopub.status.idle":"2023-04-25T14:37:36.970920Z","shell.execute_reply.started":"2023-04-25T14:37:36.964254Z","shell.execute_reply":"2023-04-25T14:37:36.969431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Add classification head to the base model","metadata":{}},{"cell_type":"code","source":"headmodel = basemodel.output\nheadmodel = AveragePooling2D(pool_size=(4, 4))(headmodel)\nheadmodel = Flatten(name=\"flatten\")(headmodel)\nheadmodel = Dense(256, activation=\"relu\")(headmodel)\nheadmodel = Dropout(0.3)(headmodel)  \nheadmodel = Dense(256, activation=\"relu\")(headmodel)\nheadmodel = Dropout(0.3)(headmodel)\nheadmodel = Dense(2, activation=\"sigmoid\")(headmodel)\n\nmodel = Model(inputs=basemodel.input, outputs=headmodel)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:38:06.093814Z","iopub.execute_input":"2023-04-25T14:38:06.094904Z","iopub.status.idle":"2023-04-25T14:38:06.707005Z","shell.execute_reply.started":"2023-04-25T14:38:06.094846Z","shell.execute_reply":"2023-04-25T14:38:06.706093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# compile the model\nmodel.compile(loss=\"categorical_crossentropy\", optimizer=\"adam\", metrics=[\"accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:41:43.516627Z","iopub.execute_input":"2023-04-25T14:41:43.517038Z","iopub.status.idle":"2023-04-25T14:41:43.543599Z","shell.execute_reply.started":"2023-04-25T14:41:43.517005Z","shell.execute_reply":"2023-04-25T14:41:43.542418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# use early stopping to exit training if validation loss is not decreasing even after certain epochs (patience)\nearlystopping = EarlyStopping(monitor='val_loss', mode='min', verbose=1, patience=20)\n\n# save the best model with least validation loss\ncheckpointer = ModelCheckpoint(\n    filepath=\"classifier-resnet-weights.hdf5\", verbose=1, save_best_only=True\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:41:46.500900Z","iopub.execute_input":"2023-04-25T14:41:46.501586Z","iopub.status.idle":"2023-04-25T14:41:46.507141Z","shell.execute_reply.started":"2023-04-25T14:41:46.501535Z","shell.execute_reply":"2023-04-25T14:41:46.505867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_generator, steps_per_epoch= train_generator.n // 16, epochs = 10, validation_data= valid_generator, validation_steps= valid_generator.n // 16, callbacks=[checkpointer, earlystopping])","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:41:50.117248Z","iopub.execute_input":"2023-04-25T14:41:50.117712Z","iopub.status.idle":"2023-04-25T18:29:11.739367Z","shell.execute_reply.started":"2023-04-25T14:41:50.117654Z","shell.execute_reply":"2023-04-25T18:29:11.738313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save the model architecture to json file for future use\nmodel_json = model.to_json()\nwith open(\"classifier-resnet-model.json\", \"w\") as json_file:\n    json_file.write(model_json)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T18:32:03.198224Z","iopub.execute_input":"2023-04-25T18:32:03.198703Z","iopub.status.idle":"2023-04-25T18:32:03.260286Z","shell.execute_reply.started":"2023-04-25T18:32:03.198659Z","shell.execute_reply":"2023-04-25T18:32:03.259053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history.history.keys()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T18:32:57.915677Z","iopub.execute_input":"2023-04-25T18:32:57.916125Z","iopub.status.idle":"2023-04-25T18:32:57.923901Z","shell.execute_reply.started":"2023-04-25T18:32:57.916083Z","shell.execute_reply":"2023-04-25T18:32:57.922467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2)\ntrain_acc = history.history[\"accuracy\"]\ntrain_loss = history.history[\"loss\"]\nfig.set_size_inches(12, 4)\n\nax[0].plot(history.history[\"accuracy\"])\nax[0].plot(history.history[\"val_accuracy\"])\nax[0].set_title(\"Training Accuracy vs Validation Accuracy\")\nax[0].set_ylabel(\"Accuracy\")\nax[0].set_xlabel(\"Epoch\")\nax[0].legend([\"Train\", \"Validation\"], loc=\"upper left\")\n\nax[1].plot(history.history[\"loss\"])\nax[1].plot(history.history[\"val_loss\"])\nax[1].set_title(\"Training Loss vs Validation Loss\")\nax[1].set_ylabel(\"Loss\")\nax[1].set_xlabel(\"Epoch\")\nax[1].legend([\"Train\", \"Validation\"], loc=\"upper left\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T18:33:04.716979Z","iopub.execute_input":"2023-04-25T18:33:04.717369Z","iopub.status.idle":"2023-04-25T18:33:05.048275Z","shell.execute_reply.started":"2023-04-25T18:33:04.717335Z","shell.execute_reply":"2023-04-25T18:33:05.046879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Assess Trained Model Performance","metadata":{}},{"cell_type":"code","source":"prediction = model.predict(test_generator)\n\npred = np.argmax(prediction, axis=1)\n#pred = np.asarray(pred).astype('str')\noriginal = np.asarray(test['mask']).astype('int')\n\nfrom sklearn.metrics import accuracy_score, confusion_matrix, classification_report\naccuracy = accuracy_score(original, pred)\nprint(accuracy)\n\ncm = confusion_matrix(original, pred)\n\nreport = classification_report(original, pred, labels = [0,1])\nprint(report)\nplt.figure(figsize = (5,5))\nsns.heatmap(cm, annot=True);","metadata":{"execution":{"iopub.status.busy":"2023-04-25T18:34:05.914887Z","iopub.execute_input":"2023-04-25T18:34:05.915346Z","iopub.status.idle":"2023-04-25T18:35:09.998218Z","shell.execute_reply.started":"2023-04-25T18:34:05.915307Z","shell.execute_reply":"2023-04-25T18:35:09.997159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- The model performs really well\n","metadata":{}},{"cell_type":"markdown","source":"# Build Segmentation Model to Localize Tumor\n","metadata":{}},{"cell_type":"code","source":"# Get the dataframe containing MRIs which have masks associated with them.\nbrain_df_mask = brain_df[brain_df['mask'] == 1]\nbrain_df_mask.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-25T18:40:43.864711Z","iopub.execute_input":"2023-04-25T18:40:43.865170Z","iopub.status.idle":"2023-04-25T18:40:43.876937Z","shell.execute_reply.started":"2023-04-25T18:40:43.865131Z","shell.execute_reply":"2023-04-25T18:40:43.875435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating test, train and val sets\nX_train, X_val = train_test_split(brain_df_mask, test_size=0.15)\nX_test, X_val = train_test_split(X_val, test_size=0.5)\nprint(\"Train size is {}, valid size is {} & test size is {}\".format(len(X_train), len(X_val), len(X_test)))\n\ntrain_ids = list(X_train.image_path)\ntrain_mask = list(X_train.mask_path)\n\nval_ids = list(X_val.image_path)\nval_mask= list(X_val.mask_path)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T18:43:55.486403Z","iopub.execute_input":"2023-04-25T18:43:55.486847Z","iopub.status.idle":"2023-04-25T18:43:55.499959Z","shell.execute_reply.started":"2023-04-25T18:43:55.486807Z","shell.execute_reply":"2023-04-25T18:43:55.498585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = DataGenerator(train_ids, train_mask)\nval_data = DataGenerator(val_ids, val_mask)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T19:46:55.100929Z","iopub.execute_input":"2023-04-25T19:46:55.101367Z","iopub.status.idle":"2023-04-25T19:46:55.106924Z","shell.execute_reply.started":"2023-04-25T19:46:55.101328Z","shell.execute_reply":"2023-04-25T19:46:55.105756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resblock(X, f):\n  \n\n    # make a copy of input\n    X_copy = X\n\n    # main path\n    # he_normal: https://medium.com/@prateekvishnu/xavier-and-he-normal-he-et-al-initialization-8e3d7a087528\n\n    X = Conv2D(f, kernel_size = (1,1) ,strides = (1,1),kernel_initializer ='he_normal')(X)\n    X = BatchNormalization()(X)\n    X = Activation('relu')(X) \n\n    X = Conv2D(f, kernel_size = (3,3), strides =(1,1), padding = 'same', kernel_initializer ='he_normal')(X)\n    X = BatchNormalization()(X)\n\n    # Short path\n    # https://towardsdatascience.com/understanding-and-coding-a-resnet-in-keras-446d7ff84d33\n\n    X_copy = Conv2D(f, kernel_size = (1,1), strides =(1,1), kernel_initializer ='he_normal')(X_copy)\n    X_copy = BatchNormalization()(X_copy)\n\n    # Adding the output from main path and short path together\n\n    X = Add()([X,X_copy])\n    X = Activation('relu')(X)\n\n    return X","metadata":{"execution":{"iopub.status.busy":"2023-04-25T19:46:58.917681Z","iopub.execute_input":"2023-04-25T19:46:58.918705Z","iopub.status.idle":"2023-04-25T19:46:58.927162Z","shell.execute_reply.started":"2023-04-25T19:46:58.918663Z","shell.execute_reply":"2023-04-25T19:46:58.926064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function to upscale and concatenate the values passsed\ndef upsample_concat(x, skip):\n    x = UpSampling2D((2,2))(x)\n    merge = Concatenate()([x, skip])\n\n    return merge","metadata":{"execution":{"iopub.status.busy":"2023-04-25T19:47:01.767886Z","iopub.execute_input":"2023-04-25T19:47:01.768654Z","iopub.status.idle":"2023-04-25T19:47:01.774473Z","shell.execute_reply.started":"2023-04-25T19:47:01.768611Z","shell.execute_reply":"2023-04-25T19:47:01.773416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ResUnet Localization Model\ninput_shape = (256, 256, 3)\n\n# Input tensor shape\nX_input = Input(input_shape)\n\n# Stage 1\nconv1_in = Conv2D(\n    16, 3, activation=\"relu\", padding=\"same\", kernel_initializer=\"he_normal\"\n)(X_input)\nconv1_in = BatchNormalization()(conv1_in)\nconv1_in = Conv2D(\n    16, 3, activation=\"relu\", padding=\"same\", kernel_initializer=\"he_normal\"\n)(conv1_in)\nconv1_in = BatchNormalization()(conv1_in)\npool_1 = MaxPool2D(pool_size=(2, 2))(conv1_in)\n\n# Stage 2\nconv2_in = resblock(pool_1, 32)\npool_2 = MaxPool2D(pool_size=(2, 2))(conv2_in)\n\n# Stage 3\nconv3_in = resblock(pool_2, 64)\npool_3 = MaxPool2D(pool_size=(2, 2))(conv3_in)\n\n# Stage 4\nconv4_in = resblock(pool_3, 128)\npool_4 = MaxPool2D(pool_size=(2, 2))(conv4_in)\n\n# Stage 5 (Bottle Neck)\nconv5_in = resblock(pool_4, 256)\n\n# Upscale stage 1\nup_1 = upsample_concat(conv5_in, conv4_in)\nup_1 = resblock(up_1, 128)\n\n# Upscale stage 2\nup_2 = upsample_concat(up_1, conv3_in)\nup_2 = resblock(up_2, 64)\n\n# Upscale stage 3\nup_3 = upsample_concat(up_2, conv2_in)\nup_3 = resblock(up_3, 32)\n\n# Upscale stage 4\nup_4 = upsample_concat(up_3, conv1_in)\nup_4 = resblock(up_4, 16)\n\n# Final Output\noutput = Conv2D(1, (1, 1), padding=\"same\", activation=\"sigmoid\")(up_4)\n\nseg_model = Model(inputs=X_input, outputs=output)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T19:47:03.996036Z","iopub.execute_input":"2023-04-25T19:47:03.997169Z","iopub.status.idle":"2023-04-25T19:47:04.743222Z","shell.execute_reply.started":"2023-04-25T19:47:03.997122Z","shell.execute_reply":"2023-04-25T19:47:04.742230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seg_model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T19:47:06.546751Z","iopub.execute_input":"2023-04-25T19:47:06.547519Z","iopub.status.idle":"2023-04-25T19:47:06.798639Z","shell.execute_reply.started":"2023-04-25T19:47:06.547477Z","shell.execute_reply":"2023-04-25T19:47:06.797241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Understand The Intuition Behind Resunet Models\n\n- ResUNet architecture combines UNet backbone architecture with residual blocks to overcome the vanishing gradients problems present in deep architectures.\n- Unet architecture is based on Fully Convolutional Networks and modified in a way that it performs well on segmentation tasks.\n\n\n#### Resunet consists of three parts:\n\n1. Encoder or contracting path.\n- First block consists of 3x3 convolution layer + Relu + Batch-Normalization\n- Remaining three blocks consist of Res-blocks followed by Max-pooling 2x2.\n\n2. Bottleneck.\n- It is in- in-between the contracting and expanding path.\n- It consist of Res-block followed by up sampling conv layer 2x2.\n\n3. Decoder or expansive path.\n- 3 blocks following bottleneck consist of Res-blocks followed by up-sampling conv layer 2 x 2\n- Final block consist of Res- Res-block followed by 1x1 conv layer","metadata":{}},{"cell_type":"markdown","source":"## Train A Segmentation Resunet Model To Localize Tumor","metadata":{}},{"cell_type":"code","source":"# compling model and callbacks functions\nadam = tf.keras.optimizers.Adam(lr = 0.05, epsilon = 0.1)\nseg_model.compile(optimizer = adam, \n                  loss = focal_tversky, \n                  metrics = [tversky]\n                 )\n#callbacks\nearlystopping = EarlyStopping(monitor='val_loss',\n                              mode='min', \n                              verbose=1, \n                              patience=20\n                             )\n# save the best model with lower validation loss\ncheckpointer = ModelCheckpoint(filepath=\"/kaggle/input/modelresunet/ResUNet-segModel-weights.hdf5\", \n                               verbose=1, \n                               save_best_only=True\n                              )\nreduce_lr = ReduceLROnPlateau(monitor='val_loss',\n                              mode='min',\n                              verbose=1,\n                              patience=10,\n                              min_delta=0.0001,\n                              factor=0.2\n                             )","metadata":{"execution":{"iopub.status.busy":"2023-04-25T19:51:15.535938Z","iopub.execute_input":"2023-04-25T19:51:15.536356Z","iopub.status.idle":"2023-04-25T19:51:15.559989Z","shell.execute_reply.started":"2023-04-25T19:51:15.536320Z","shell.execute_reply":"2023-04-25T19:51:15.558720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# hist = seg_model.fit(train_data, \n#                   epochs = 60, \n#                   validation_data = val_data,\n#                   callbacks = [checkpointer, earlystopping, reduce_lr]\n#                  )","metadata":{"execution":{"iopub.status.busy":"2023-04-25T19:42:49.050120Z","iopub.execute_input":"2023-04-25T19:42:49.051266Z","iopub.status.idle":"2023-04-25T19:42:49.055769Z","shell.execute_reply.started":"2023-04-25T19:42:49.051220Z","shell.execute_reply":"2023-04-25T19:42:49.054584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # save the model architecture to json file for future use\n\n# model_json = model_seg.to_json()\n# with open(\"ResUNet-model.json\",\"w\") as json_file:\n#     json_file.write(model_json)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T19:47:53.482408Z","iopub.execute_input":"2023-04-25T19:47:53.482868Z","iopub.status.idle":"2023-04-25T19:47:53.487766Z","shell.execute_reply.started":"2023-04-25T19:47:53.482827Z","shell.execute_reply":"2023-04-25T19:47:53.486565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# hist.history.keys()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T19:48:35.945052Z","iopub.execute_input":"2023-04-25T19:48:35.946187Z","iopub.status.idle":"2023-04-25T19:48:35.949865Z","shell.execute_reply.started":"2023-04-25T19:48:35.946141Z","shell.execute_reply":"2023-04-25T19:48:35.948850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # use early stopping to exit training if validation loss is not decreasing even after certain epochs (patience)\n# earlystopping = EarlyStopping(monitor=\"val_loss\", mode=\"min\", verbose=1, patience=20)\n\n# # save the best model with lower validation loss\n# checkpointer = ModelCheckpoint(\n#     filepath=\"ResUNet-weights.hdf5\", verbose=1, save_best_only=True\n# )","metadata":{"execution":{"iopub.status.busy":"2023-04-24T21:57:07.620826Z","iopub.status.idle":"2023-04-24T21:57:07.621573Z","shell.execute_reply.started":"2023-04-24T21:57:07.621196Z","shell.execute_reply":"2023-04-24T21:57:07.621234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plt.figure(figsize=(12,5))\n# plt.subplot(1,2,1)\n# plt.plot(hist.history['loss']);\n# plt.plot(hist.history['val_loss']);\n# plt.title(\"SEG Model focal tversky Loss\");\n# plt.ylabel(\"focal tversky loss\");\n# plt.xlabel(\"Epochs\");\n# plt.legend(['train', 'val']);\n\n# plt.subplot(1,2,2)\n# plt.plot(hist.history['tversky']);\n# plt.plot(hist.history['val_tversky']);\n# plt.title(\"SEG Model tversky score\");\n# plt.ylabel(\"tversky Accuracy\");\n# plt.xlabel(\"Epochs\");\n# plt.legend(['train', 'val']);","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_ids = list(X_test.image_path)\n# test_mask = list(X_test.mask_path)\n# test_data = DataGenerator(test_ids, test_mask)\n# _, tv = seg_model.evaluate(test_data)\n# print(\"Segmentation tversky is {:.2f}%\".format(tv*100))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Assess Trained Segmentation Resunet Model Performance\n","metadata":{}},{"cell_type":"code","source":"# # utilities file contains the code for custom loss function and custom data generator\n# from utilities import prediction\n\n# # making prediction\n# df_pred = prediction(test, model, seg_model)\n# df_pred","metadata":{"execution":{"iopub.status.busy":"2023-04-25T19:49:54.827593Z","iopub.execute_input":"2023-04-25T19:49:54.828009Z","iopub.status.idle":"2023-04-25T19:49:54.832856Z","shell.execute_reply.started":"2023-04-25T19:49:54.827974Z","shell.execute_reply":"2023-04-25T19:49:54.831617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # merging original and prediction df\n# df_pred = test.merge(df_pred, on='image_path')\n# df_pred.head(10)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #Visually verify that model predictions make sense\n# count = 0\n# fig, axs = plt.subplots(10, 5, figsize=(30, 50))\n# for i in range(len(df_pred)):\n#   if df_pred['has_mask'][i] == 1 and count < 10:\n#     # read the images and convert them to RGB format\n#     img = io.imread(df_pred.image_path[i])\n#     img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n#     axs[count][0].title.set_text(\"Brain MRI\")\n#     axs[count][0].imshow(img)\n\n#     # Obtain the mask for the image\n#     mask = io.imread(df_pred.mask_path[i])\n#     axs[count][1].title.set_text(\"Original Mask\")\n#     axs[count][1].imshow(mask)\n\n#     # Obtain the predicted mask for the image\n#     predicted_mask = np.asarray(df_pred.predicted_mask[i])[0].squeeze().round()\n#     axs[count][2].title.set_text(\"AI Predicted Mask\")\n#     axs[count][2].imshow(predicted_mask)\n\n#     # Apply the mask to the image 'mask==255'\n#     img[mask == 255] = (255, 0, 0)\n#     axs[count][3].title.set_text(\"MRI with Original Mask (Ground Truth)\")\n#     axs[count][3].imshow(img)\n\n#     img_ = io.imread(df_pred.image_path[i])\n#     img_ = cv2.cvtColor(img_, cv2.COLOR_BGR2RGB)\n#     img_[predicted_mask == 1] = (0, 255, 0)\n#     axs[count][4].title.set_text(\"MRI with AI Predicted Mask\")\n#     axs[count][4].imshow(img_)\n#     count += 1\n\n# fig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T19:50:51.279407Z","iopub.execute_input":"2023-04-25T19:50:51.279835Z","iopub.status.idle":"2023-04-25T19:50:51.286635Z","shell.execute_reply.started":"2023-04-25T19:50:51.279797Z","shell.execute_reply":"2023-04-25T19:50:51.284958Z"},"trusted":true},"execution_count":null,"outputs":[]}]}