{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"sourceType":"competition"}],"dockerImageVersionId":30302,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Note: \n\nAfter having run some simple codes in our system (which is not GPU enabled, and also Kaggle GPU seems more appropriate) we found out that the data given to us as a Google Drive file [https://drive.google.com/drive/folders/1GYAe8hZB8Si5YSW0akNXBpsELRicE4Hp] is EXACTLY the same as that is here: [https://www.kaggle.com/competitions/rsna-pneumonia-detection-challenge].","metadata":{}},{"cell_type":"markdown","source":"## Libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd                                                 # for panel data analysis\npd.set_option('display.max_columns', None)                          # Unfolding hidden features if the cardinality is high      \npd.set_option('display.max_colwidth', None)                         # Unfolding the max feature width for better clrity      \npd.set_option('display.max_rows', None)                             # Unfolding hidden data points if the cardinality is high\npd.set_option('display.float_format', lambda x: '%.3f' % x)         # To suppress scientific notation over exponential values\n#-------------------------------------------------------------------------------------------------------------------------------\nimport numpy as np                                                  # Importing package numpys (For Numerical Python)\n#-------------------------------------------------------------------------------------------------------------------------------\nimport matplotlib.pyplot as plt                                     # Importing pyplot interface using matplotlib\nfrom matplotlib.pyplot import figure\nfigure(figsize=(15, 12), dpi=120)\nfrom matplotlib.pylab import rcParams                               # Backend used for rendering and GUI integration                                               \nimport seaborn as sns                                               # Importing seaborn library for interactive visualization\n#set seaborn plotting aesthetics\nsns.set(style='whitegrid')\n%matplotlib inline\n#-------------------------------------------------------------------------------------------------------------------------------\nimport os\nimport csv\nimport random\nimport pydicom\n#-------------------------------------------------------------------------------------------------------------------------------\n\nfrom itertools import chain\nfrom skimage.io import imread, imshow, concatenate_images\nfrom skimage.transform import resize\nfrom skimage.morphology import label\nfrom sklearn.model_selection import train_test_split\n\nimport tensorflow as tf\nfrom tensorflow import keras\n\nfrom tensorflow.keras.models import Model, load_model\nfrom tensorflow.keras.losses import BinaryCrossentropy\nfrom tensorflow.keras.backend import flatten, sum, clear_session\nfrom tensorflow.keras.layers import Input, BatchNormalization, Activation, Dense, Dropout, SeparableConv2D, Conv2DTranspose, UpSampling2D\nfrom keras.layers.core import Lambda, RepeatVector, Reshape\nfrom keras.layers.convolutional import Conv2D, Conv2DTranspose\nfrom keras.layers.pooling import MaxPooling2D, GlobalMaxPool2D\nfrom keras.layers.merge import concatenate, add\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras.optimizers import Adam\n\nfrom tensorflow.keras.preprocessing.image import array_to_img, img_to_array, load_img\nfrom tensorflow.keras.metrics import MeanIoU\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.callbacks import CSVLogger\nimport datetime","metadata":{"execution":{"iopub.status.busy":"2024-02-06T23:24:23.889789Z","iopub.execute_input":"2024-02-06T23:24:23.890589Z","iopub.status.idle":"2024-02-06T23:24:23.913564Z","shell.execute_reply.started":"2024-02-06T23:24:23.890543Z","shell.execute_reply":"2024-02-06T23:24:23.912631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Parse train data as dictionary","metadata":{}},{"cell_type":"code","source":"# empty dictionary\nparsed_dict = {}\n# load table\nwith open(os.path.join('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv'), mode='r') as infile:\n    # open reader\n    reader = csv.reader(infile)\n    # skip header\n    next(reader, None)\n    # loop through rows\n    for rows in reader:\n        # retrieve information\n        filename = rows[0]\n        location = rows[1:5]\n        pneumonia = rows[5]\n        # if row contains pneumonia add label to dictionary\n        # which contains a list of pneumonia locations per filename\n        if pneumonia == '1':\n            # convert string to float to int\n            location = [int(float(i)) for i in location]\n            \n            # save pneumonia location in dictionary\n            if filename in parsed_dict:\n                parsed_dict[filename].append(location)\n            else:\n                parsed_dict[filename] = [location]","metadata":{"_uuid":"e08496a85ef9b0823595c3745d2677c6e84b6a3a","execution":{"iopub.status.busy":"2024-02-06T23:24:24.302827Z","iopub.execute_input":"2024-02-06T23:24:24.303813Z","iopub.status.idle":"2024-02-06T23:24:24.427168Z","shell.execute_reply.started":"2024-02-06T23:24:24.303775Z","shell.execute_reply":"2024-02-06T23:24:24.426139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create a helper list of Training Images. And split it into train and val\n#### We take random 2560 images in out validation set. 0.10 split.","metadata":{}},{"cell_type":"code","source":"# load and shuffle img_list\nfolder = '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images'\nimg_list = os.listdir(folder)\nrandom.shuffle(img_list)\n\n\n# split into train and validation img_list\nn_valid_samples = 2560 # 0.10 of 26684\ntrain_img_list = img_list[n_valid_samples:]\nvalid_img_list = img_list[:n_valid_samples]\nprint('n train samples', len(train_img_list))\nprint('n valid samples', len(valid_img_list))\nn_train_samples = len(img_list) - n_valid_samples","metadata":{"_uuid":"ccd0b0d52cafd125558ed5560a9cc8fa15760bc5","execution":{"iopub.status.busy":"2024-02-06T23:24:24.702304Z","iopub.execute_input":"2024-02-06T23:24:24.702714Z","iopub.status.idle":"2024-02-06T23:24:25.354796Z","shell.execute_reply.started":"2024-02-06T23:24:24.702677Z","shell.execute_reply":"2024-02-06T23:24:25.353723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loss Functions","metadata":{}},{"cell_type":"code","source":"# define iou or jaccard loss function\ndef iou_loss(y_true, y_pred):\n    y_true = tf.reshape(tf.cast(y_true, tf.float32), [-1])\n    y_pred = tf.reshape(tf.cast(y_pred, tf.float32), [-1])\n    intersection = tf.reduce_sum(y_true * y_pred)\n    total = tf.reduce_sum(y_true) + tf.reduce_sum(y_pred)\n    union = total - intersection\n    IoU = (intersection + 1.) / (union + 1.)\n    return 1 - IoU\n\n# def iou_loss(y_true, y_pred):\n#     y_true = tf.reshape(y_true, [-1]) \n#     y_pred = tf.reshape(y_pred, [-1])\n#     intersection = tf.reduce_sum(y_true * y_pred)\n#     score = (intersection + 1.) / (tf.reduce_sum(y_true) + tf.reduce_sum(y_pred) - intersection + 1.)\n#     return 1 - score\n\n# combine bce loss and iou loss\ndef iou_bce_loss(y_true, y_pred):\n    return 0.5 * keras.losses.binary_crossentropy(y_true, y_pred) + 0.5 * iou_loss(y_true, y_pred)\n\n# mean iou as a metric\ndef mean_iou(y_true, y_pred):\n    y_pred = tf.round(y_pred)\n    intersect = tf.reduce_sum(y_true * y_pred, axis=[1, 2, 3])\n    union = tf.reduce_sum(y_true, axis=[1, 2, 3]) + tf.reduce_sum(y_pred, axis=[1, 2, 3])\n    smooth = tf.ones(tf.shape(intersect))\n    return tf.reduce_mean((intersect + smooth) / (union - intersect + smooth))","metadata":{"execution":{"iopub.status.busy":"2024-02-06T23:24:25.35625Z","iopub.execute_input":"2024-02-06T23:24:25.35653Z","iopub.status.idle":"2024-02-06T23:24:25.367025Z","shell.execute_reply.started":"2024-02-06T23:24:25.356504Z","shell.execute_reply":"2024-02-06T23:24:25.366132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Learning Rate Annealing Function","metadata":{"execution":{"iopub.status.busy":"2022-11-19T23:20:49.857847Z","iopub.execute_input":"2022-11-19T23:20:49.858407Z","iopub.status.idle":"2022-11-19T23:20:49.877456Z","shell.execute_reply.started":"2022-11-19T23:20:49.858299Z","shell.execute_reply":"2022-11-19T23:20:49.876617Z"}}},{"cell_type":"code","source":"# cosine learning rate annealing\ndef cosine_annealing(x):\n    lr = 0.001\n    epochs = 25\n    return lr*(np.cos(np.pi*x/epochs)+1.)/2\nlearning_rate = tf.keras.callbacks.LearningRateScheduler(cosine_annealing)","metadata":{"execution":{"iopub.status.busy":"2024-02-06T23:24:25.513442Z","iopub.execute_input":"2024-02-06T23:24:25.513827Z","iopub.status.idle":"2024-02-06T23:24:25.519488Z","shell.execute_reply.started":"2024-02-06T23:24:25.513793Z","shell.execute_reply":"2024-02-06T23:24:25.518544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Hyperparameters","metadata":{}},{"cell_type":"code","source":"BATCH_SIZE = 32\nIMG_SIZE = 256\nSEED = 42","metadata":{"execution":{"iopub.status.busy":"2024-02-06T23:24:25.910238Z","iopub.execute_input":"2024-02-06T23:24:25.910878Z","iopub.status.idle":"2024-02-06T23:24:25.915333Z","shell.execute_reply.started":"2024-02-06T23:24:25.910841Z","shell.execute_reply":"2024-02-06T23:24:25.914366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data generator function","metadata":{}},{"cell_type":"code","source":"class generator(keras.utils.Sequence):\n    \n    def __init__(self, folder, img_list, parsed_dict=None, batch_size=32, image_size=256, shuffle=True, augment=False, predict=False):\n        self.folder = folder\n        self.img_list = img_list\n        self.parsed_dict = parsed_dict\n        self.batch_size = batch_size\n        self.image_size = image_size\n        self.shuffle = shuffle\n        self.augment = augment\n        self.predict = predict\n        self.on_epoch_end()\n        \n    def __load__(self, filename):\n        # load dicom file as numpy array\n        img = pydicom.dcmread(os.path.join(self.folder, filename)).pixel_array\n        # create empty mask\n        msk = np.zeros(img.shape)\n        # get filename without extension\n        filename = filename.split('.')[0]\n        # if image contains pneumonia\n        if filename in self.parsed_dict:\n            # loop through pneumonia\n            for location in self.parsed_dict[filename]:\n                # add 1's at the location of the pneumonia\n                x, y, w, h = location\n                msk[y:y+h, x:x+w] = 1\n        # resize both image and mask\n        img = resize(img, (self.image_size, self.image_size), mode='reflect')\n        msk = resize(msk, (self.image_size, self.image_size), mode='reflect') > 0.5\n        # if augment then horizontal flip half the time\n        if self.augment and random.random() > 0.5: #image augmentation parameters\n            img = np.fliplr(img)\n            msk = np.fliplr(msk)\n        # add trailing channel dimension\n        img = np.expand_dims(img, -1)\n        msk = np.expand_dims(msk, -1)\n        return img, msk\n    \n    def __loadpredict__(self, filename):\n        # load dicom file as numpy array\n        img = pydicom.dcmread(os.path.join(self.folder, filename)).pixel_array\n        # resize image\n        img = resize(img, (self.image_size, self.image_size), mode='reflect')\n        # add trailing channel dimension\n        img = np.expand_dims(img, -1)\n        return img\n        \n    def __getitem__(self, index):\n        # select batch\n        img_list = self.img_list[index*self.batch_size:(index+1)*self.batch_size]\n        # predict mode: return images and img_list\n        if self.predict:\n            # load files\n            imgs = [self.__loadpredict__(filename) for filename in img_list]\n            # create numpy batch\n            imgs = np.array(imgs)\n            return imgs, img_list\n        # train mode: return images and masks\n        else:\n            # load files\n            items = [self.__load__(filename) for filename in img_list]\n            # unzip images and masks\n            imgs, msks = zip(*items)\n            # create numpy batch\n            imgs = np.array(imgs)\n            msks = np.array(msks)\n            return imgs, msks\n        \n    def on_epoch_end(self):\n        if self.shuffle:\n            random.shuffle(self.img_list)\n        \n    def __len__(self):\n        if self.predict:\n            # return everything\n            return int(np.ceil(len(self.img_list) / self.batch_size))\n        else:\n            # return full batches only\n            return int(len(self.img_list) / self.batch_size)","metadata":{"_uuid":"86b3f780a03cddda78c6adfde461d6ff8dad5672","execution":{"iopub.status.busy":"2024-02-06T23:24:26.310305Z","iopub.execute_input":"2024-02-06T23:24:26.311199Z","iopub.status.idle":"2024-02-06T23:24:26.333569Z","shell.execute_reply.started":"2024-02-06T23:24:26.311163Z","shell.execute_reply":"2024-02-06T23:24:26.332475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Network","metadata":{"_uuid":"ebc622a4b406354cc4ef28801eab72346a724d8b"}},{"cell_type":"code","source":"def downsample(channels, inputs):\n    x = keras.layers.BatchNormalization(momentum=0.9)(inputs)\n    x = keras.layers.LeakyReLU(0)(x)\n    x = keras.layers.Conv2D(channels, 1, padding='same', use_bias=False)(x)\n    x = keras.layers.MaxPool2D(2)(x)\n    return x\n\ndef skip_conn(channels, inputs):\n    x = keras.layers.BatchNormalization(momentum=0.9)(inputs)\n    x = keras.layers.LeakyReLU(0)(x)\n    x = keras.layers.Conv2D(channels, 3, padding='same', use_bias=False)(x)\n    x = keras.layers.BatchNormalization(momentum=0.9)(x)\n    x = keras.layers.LeakyReLU(0)(x)\n    x = keras.layers.Conv2D(channels, 3, padding='same', use_bias=False)(x)\n    return keras.layers.add([x, inputs])\n\ndef architecture(input_size, channels, n_blocks=2, depth=4):\n    # input\n    inputs = keras.Input(shape=(input_size, input_size, 1))\n    x = keras.layers.Conv2D(channels, 3, padding='same', use_bias=False)(inputs)\n    # residual blocks\n    for d in range(depth):\n        channels = channels * 2\n        x = downsample(channels, x)\n        for b in range(n_blocks):\n            x = skip_conn(channels, x)\n    # output\n    x = keras.layers.BatchNormalization(momentum=0.9)(x)\n    x = keras.layers.LeakyReLU(0)(x)\n    x = keras.layers.Conv2D(1, 1, activation='sigmoid')(x)\n    outputs = keras.layers.UpSampling2D(2**depth)(x)\n    model = keras.Model(inputs=inputs, outputs=outputs)\n    return model","metadata":{"_uuid":"9fc2b108689637a6037b48ebab3f7659b8704bf9","execution":{"iopub.status.busy":"2024-02-06T23:24:26.726909Z","iopub.execute_input":"2024-02-06T23:24:26.727286Z","iopub.status.idle":"2024-02-06T23:24:26.741603Z","shell.execute_reply.started":"2024-02-06T23:24:26.727254Z","shell.execute_reply":"2024-02-06T23:24:26.74065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Compiling and training network","metadata":{}},{"cell_type":"code","source":"# create train and validation generators\nfolder = '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images'\ntrain_ds = generator(folder, train_img_list, parsed_dict, batch_size=32, image_size=256, shuffle=True, augment=True, predict=False)\nval_ds = generator(folder, valid_img_list, parsed_dict, batch_size=32, image_size=256, shuffle=False, predict=False)\n\n# create network and compiler\nmodel = architecture(input_size=256, channels=32, n_blocks=2, depth=4)\nmodel.compile(optimizer='adam',\n              loss=iou_bce_loss,\n              metrics=['accuracy', mean_iou])\n\nhistory = model.fit_generator(train_ds, validation_data=val_ds, callbacks=[learning_rate], epochs=25, workers=4, use_multiprocessing=True)","metadata":{"_uuid":"4369be30f61440eb6858d57829fa541c4ee893bf","execution":{"iopub.status.busy":"2024-02-06T23:24:27.182537Z","iopub.execute_input":"2024-02-06T23:24:27.182923Z","iopub.status.idle":"2024-02-07T05:42:23.608444Z","shell.execute_reply.started":"2024-02-06T23:24:27.182893Z","shell.execute_reply":"2024-02-07T05:42:23.60646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evolution of loss, accuracy and IoU","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12,4))\nplt.subplot(131)\nplt.plot(history.epoch, history.history[\"loss\"], label=\"Train loss\")\nplt.plot(history.epoch, history.history[\"val_loss\"], label=\"Valid loss\")\nplt.legend()\nplt.subplot(132)\nplt.plot(history.epoch, history.history[\"acc\"], label=\"Train accuracy\")\nplt.plot(history.epoch, history.history[\"val_acc\"], label=\"Valid accuracy\")\nplt.legend()\nplt.subplot(133)\nplt.plot(history.epoch, history.history[\"mean_iou\"], label=\"Train iou\")\nplt.plot(history.epoch, history.history[\"val_mean_iou\"], label=\"Valid iou\")\nplt.legend()\nplt.show()","metadata":{"_uuid":"3666ba4cac9ed2c3029b824af220404bfcc16f23","execution":{"iopub.status.busy":"2024-02-07T05:42:24.788396Z","iopub.status.idle":"2024-02-07T05:42:24.788799Z","shell.execute_reply.started":"2024-02-07T05:42:24.788602Z","shell.execute_reply":"2024-02-07T05:42:24.78863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Plotting predictions in one batch","metadata":{}},{"cell_type":"code","source":"for imgs, msks in val_ds:\n    # predict batch of images\n    preds = model.predict(imgs)\n    # create figure\n    f, axarr = plt.subplots(4, 8, figsize=(20,15))\n    axarr = axarr.ravel()\n    axidx = 0\n    # loop through batch\n    for img, msk, pred in zip(imgs, msks, preds):\n        # plot image\n        axarr[axidx].imshow(img[:, :, 0])\n        # threshold true mask\n        comp = msk[:, :, 0] > 0.5\n        # apply connected components\n        comp = measure.label(comp)\n        # apply bounding boxes\n        predictionString = ''\n        for region in measure.regionprops(comp):\n            # retrieve x, y, height and width\n            y, x, y2, x2 = region.bbox\n            height = y2 - y\n            width = x2 - x\n            axarr[axidx].add_patch(patches.Rectangle((x,y),width,height,linewidth=2,edgecolor='b',facecolor='none'))\n        # threshold predicted mask\n        comp = pred[:, :, 0] > 0.5\n        # apply connected components\n        comp = measure.label(comp)\n        # apply bounding boxes\n        predictionString = ''\n        for region in measure.regionprops(comp):\n            # retrieve x, y, height and width\n            y, x, y2, x2 = region.bbox\n            height = y2 - y\n            width = x2 - x\n            axarr[axidx].add_patch(patches.Rectangle((x,y),width,height,linewidth=2,edgecolor='r',facecolor='none'))\n        axidx += 1\n    plt.show()\n    # only plot one batch\n    break","metadata":{"_uuid":"fbbf7546d396560d21b7af17d4c959713f425d8e","execution":{"iopub.status.busy":"2024-02-07T05:42:24.790168Z","iopub.status.idle":"2024-02-07T05:42:24.790502Z","shell.execute_reply.started":"2024-02-07T05:42:24.790334Z","shell.execute_reply":"2024-02-07T05:42:24.790349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_save_path = '/kaggle/working/my_model.h5'  # Specify the path and filename for saving the model\nmodel.save(model_save_path)  # Save the model","metadata":{"execution":{"iopub.status.busy":"2024-02-07T05:48:29.937572Z","iopub.execute_input":"2024-02-07T05:48:29.938713Z","iopub.status.idle":"2024-02-07T05:48:30.368412Z","shell.execute_reply.started":"2024-02-07T05:48:29.938673Z","shell.execute_reply":"2024-02-07T05:48:30.367565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-02-07T05:51:31.644049Z","iopub.execute_input":"2024-02-07T05:51:31.644457Z","iopub.status.idle":"2024-02-07T05:51:31.694609Z","shell.execute_reply.started":"2024-02-07T05:51:31.644424Z","shell.execute_reply":"2024-02-07T05:51:31.693258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}