{"cells":[{"metadata":{},"cell_type":"markdown","source":"### Import the basic libraries"},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nimport random\nimport numpy as np\nimport pandas as pd\nimport seaborn as sb\nimport skimage.io as io\nimport cv2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport matplotlib.gridspec as gspec","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import random\nrandom.seed(0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import tensorflow\ntensorflow.__version__","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pydicom\nimport glob\nimport pylab","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Read the bounding box & target class dataset"},{"metadata":{"trusted":true},"cell_type":"code","source":"datasetDir = '../input/rsna-pneumonia-detection-challenge/'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bbox_df = pd.read_csv(datasetDir + 'stage_2_train_labels.csv')\ntargets_df = pd.read_csv(datasetDir + 'stage_2_detailed_class_info.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bbox_df.isnull().values.any(), targets_df.isnull().values.any()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bbox_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"targets_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bbox_df.patientId.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bbox_df.sort_values('patientId')\ntargets_df.sort_values('patientId')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bbox_df.shape, targets_df.shape","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"- Total number of records in each dataframe are same as expected"},{"metadata":{"trusted":true},"cell_type":"code","source":"bbox_df['patientId'].value_counts().shape[0]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"- Only 26684 records are unique out of the total 30227 records "},{"metadata":{"trusted":true},"cell_type":"code","source":"bbox_df[bbox_df.Target == 1]['patientId'].value_counts().shape[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bbox_df[bbox_df.Target == 0]['patientId'].value_counts().shape[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bbox_df[bbox_df.Target == 0].shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"targets_df['patientId'].value_counts().shape[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bbox_df.count().T","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bbox_df.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"raw","source":"- There are around 20672 records which are having null values for the bounding coordinates and remaining 9555 records associated with target 1 are having valie bounding box informations"},{"metadata":{},"cell_type":"markdown","source":"### Merge the bounding box details & target class dataset"},{"metadata":{},"cell_type":"markdown","source":"#### Now, as we have the dataset merge the bounding box details & targets dataframe together"},{"metadata":{"trusted":true},"cell_type":"code","source":"bbox_w_targets_df = pd.concat([bbox_df, targets_df.drop('patientId', axis=1)], axis=1)\nbbox_w_targets_df.head(10)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Check the duplicate entries now against each patientId"},{"metadata":{},"cell_type":"markdown","source":"##### Even though the patientId's are same, they are having different/multiple bounding box co-ordinates. The x,y co-ordinates of the bounding boxes are not duplicated, so we can't remove them.\n\n- Multiple bounding boxes may indicate multiple area of pneumonia detected for a single patient"},{"metadata":{"trusted":true},"cell_type":"code","source":"bbox_w_targets_df.groupby(['class', 'Target']).size().reset_index(name='Count By Class')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"duplicateCounts_df = bbox_w_targets_df.groupby('patientId').size().reset_index(name='counts')\nduplicateCounts_df.groupby('counts').size().reset_index(name='Count By Duplicates')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"- 1 indicates that there are no duplicates rows for 23286 records.\n- 2 : Two entries associated for a single patient we have 3266 such patient records.\n- 3 : Three bounding boxes associated for 119 patient records.\n- 4 : Four bounding boxes are present fo very minimal number of patients. 13 exactly"},{"metadata":{"trusted":true},"cell_type":"code","source":"duplicateCounts_df[duplicateCounts_df.counts == 4].sample(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Just refering a sample of one of the patient Id with 4 bounding boxes\nbbox_w_targets_df[bbox_w_targets_df.patientId == \n                  duplicateCounts_df[duplicateCounts_df.counts == 4].iloc[0].patientId]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Merge the above row count entries also in our actual bbox dataframe"},{"metadata":{"trusted":true},"cell_type":"code","source":"bbox_w_counts = pd.merge(bbox_w_targets_df, duplicateCounts_df, on='patientId')\nbbox_w_counts.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Non Null Values Count: ', bbox_w_targets_df.x.notnull().sum())\nprint('Null Values Count: ', bbox_w_targets_df.x.isnull().sum())\nprint('Total Count: ', bbox_w_targets_df.x.notnull().sum() + bbox_w_targets_df.x.isnull().sum())\nprint(\"Null Values in % : {0} ({1:2.2f}%)\".format(bbox_w_targets_df.x.isnull().sum(), \n                                                  (bbox_w_targets_df.x.isnull().sum()/len(bbox_w_targets_df))*100))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Data Visualization"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(18,5))\nfig.add_subplot(1, 3, 1)\np1 = sb.countplot(bbox_w_targets_df['Target'])\nfig.add_subplot(1, 3, 2)\np2 = sb.countplot(bbox_w_targets_df['class'])\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ax = sb.countplot(bbox_w_targets_df['class'])\nax.set(title = 'Class Distribution')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_disc = bbox_w_targets_df['class'].value_counts()\nprint('Percentage of patients with No Long opacity/Not Normal : {:5d} or {:.2f}%'.format(class_disc[0],(class_disc[0]/bbox_w_targets_df['class'].count())*100))\nprint('Percentage of patients with Long opacity : {:5d} or {:.2f}% '.format(class_disc[1],(class_disc[1]/bbox_w_targets_df['class'].count())*100))\nprint('Percentage of patients with Normal : {:5d} or {:.2f}% '.format(class_disc[2],(class_disc[2]/bbox_w_targets_df['class'].count())*100))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Read the CPR image data provided in dcm format"},{"metadata":{"trusted":true},"cell_type":"code","source":"dicom_img_dir = os.path.join(datasetDir, 'stage_2_train_images')\ndicom_img_dir","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"filenames = os.listdir(dicom_img_dir)\nlen(filenames)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"- Total of 26684 images are present which is exactly equivalent to the unique patient id provided in the labels dataset"},{"metadata":{},"cell_type":"markdown","source":"### Read the dcm image metadata of a single patient file at first"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_images_dir = os.path.join(datasetDir,'stage_2_train_images')\ntest_images_dir = os.path.join(datasetDir,'stage_2_test_images')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Total number of Training images available are : {:5d}'.format(len(list(glob.iglob(train_images_dir + \"/*.dcm\", recursive=True)))))\nprint('Total number of Test images available are : {:5d}'.format(len(list(glob.iglob(test_images_dir + \"/*.dcm\", recursive=False)))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dcm_file = os.path.join(dicom_img_dir, filenames[0])\ndcm_file","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pydicom.read_file(dcm_file)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"- The actual image of the CPR report is present in the last element tagged as Pixel data which is of array format.\n- All the remaining tags or elements are metadata providing additional details"},{"metadata":{"trusted":true},"cell_type":"code","source":"dcm_img = pydicom.read_file(dcm_file).pixel_array\ndcm_img.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dcm_img","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dcm_img_3ch = np.stack([dcm_img]*3, -1)\ndcm_img_3ch.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,20))\nfig.add_subplot(1, 3, 1)\nplt.imshow(dcm_img)\nfig.add_subplot(1, 3, 2)\nplt.imshow(dcm_img_3ch)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Let's plot the bounding box on top of the CPR data"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Extract Bounding box from data sst and visualize the bounding box in image\ndef extract_data(dataset):\n    extract_bbox = lambda row: [row['y'], row['x'], row['height'], row['width']]\n    datacol = {}\n    for n, row in dataset.iterrows():        \n        pid = row['patientId']\n        if pid not in datacol:\n            datacol[pid] = {\n                'dicom': train_images_dir + '/%s.dcm' % pid,\n                'label': row['Target'],\n                'boxes': []}\n        \n        if datacol[pid]['label'] == 1:\n            datacol[pid]['boxes'].append(extract_bbox(row))\n    return datacol","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bbox_data_dict = extract_data(bbox_w_targets_df)\nbbox_data_dict","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Select a patient Id with postive target value '1'\nsamplePatientId = bbox_df[bbox_df.Target == 1]['patientId'].iloc[0]\nsamplePatientId","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plotBoundingBoxes(imgdata):\n    d = pydicom.read_file(imgdata['dicom'])\n    im = d.pixel_array    \n    im = np.stack([im] * 3, axis=2)\n    #Add boxes with random color if present\n    for box in imgdata['boxes']:\n        rgb = np.floor(np.random.rand(3) * 256).astype('int')\n        im = overlayBoudingBox(im=im, box=box, rgb=rgb, stroke=6)\n\n    pylab.imshow(im, cmap=pylab.cm.gist_gray)\n    pylab.axis('off')\n\ndef overlayBoudingBox(im, box, rgb, stroke=1):\n    #Convert coordinates to integers\n    box = [int(b) for b in box]\n    \n    #Extract coordinates\n    y1, x1, height, width = box\n    y2 = y1 + height\n    x2 = x1 + width\n\n    im[y1:y1 + stroke, x1:x2] = rgb\n    im[y2:y2 + stroke, x1:x2] = rgb\n    im[y1:y2, x1:x1 + stroke] = rgb\n    im[y1:y2, x2:x2 + stroke] = rgb\n    return im","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plotBoundingBoxes(bbox_data_dict[samplePatientId])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"imgdata = bbox_data_dict[samplePatientId]\nd = pydicom.read_file(imgdata['dicom'])\nim = d.pixel_array    \nim = np.stack([im] * 3, axis=2)\n#Add boxes with random color if present\nfor box in imgdata['boxes']:\n    rgb = np.floor(np.random.rand(3) * 256).astype('int')\n    im = overlayBoudingBox(im=im, box=box, rgb=rgb, stroke=6)\npylab.imshow(im, cmap=pylab.cm.gist_gray)\npylab.axis('off')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bbox_df[bbox_df.Target == 1].index","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Plot some 5 random images from the dataset which are identified as Pneumonia positive"},{"metadata":{"trusted":true},"cell_type":"code","source":"rand_positive_patientIds = random.sample(list(bbox_df[bbox_df.Target == 1].index), 1)\nrand_positive_patientIds\nfor i in rand_positive_patientIds:\n    print(bbox_df[bbox_df.index == i].patientId.iloc[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"rand_positive_index = random.sample(list(bbox_df[bbox_df.Target == 1].index), 5)\nfig = plt.figure()\nfig.set_figheight(25)\nfig.set_figwidth(25)\npltloc = 0\nfor randIndex in rand_positive_index:\n    pltloc += 1\n    patientId = bbox_df[bbox_df.index == randIndex].patientId.iloc[0]\n    a = fig.add_subplot(1, 5, pltloc)\n    a.set(title = 'Index:' + str(randIndex))\n    plotBoundingBoxes(bbox_data_dict[patientId])\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Pre-Process the training & test dataset"},{"metadata":{"trusted":true},"cell_type":"code","source":"IMAGE_WIDTH = 224\nIMAGE_HEIGHT = 224","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(bbox_data_dict)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class DataGenerator(keras.utils.Sequence):\n    'Generates data for Keras'\n    def __init__(self, list_IDs,parsed,batch_size=32, dim=(256,256), n_channels=3,\n                 n_classes=2, shuffle=True):\n        'Initialization'\n        self.dim = dim\n        self.batch_size = batch_size\n        self.parsed = parsed\n        self.list_IDs = list_IDs\n        self.n_channels = n_channels\n        self.n_classes = n_classes\n        self.shuffle = shuffle\n        self.on_epoch_end()\n\n    def __len__(self):\n        'Denotes the number of batches per epoch'\n        return int(np.floor(len(self.list_IDs) / self.batch_size))\n\n    def __getitem__(self, index):\n        'Generate one batch of data'\n        # Generate indexes of the batch\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        # Find list of IDs\n        list_IDs_temp = [self.list_IDs[k] for k in indexes]\n        # Generate data\n        X, y = self.__data_generation(list_IDs_temp)\n        return X,y\n\n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        self.indexes = np.arange(len(self.list_IDs))\n        if self.shuffle == True:\n            np.random.shuffle(self.indexes)\n\n    def __data_generation(self, list_IDs_temp):\n        'Generates data containing batch_size samples' # X : (n_samples, *dim, n_channels)\n        # Initialization\n        X = np.empty((self.batch_size, *self.dim, self.n_channels))\n        y = np.zeros((self.batch_size, *self.dim))\n        # Generate data\n        for i, ID in enumerate(list_IDs_temp):\n            if ID in parsed:\n                path =  dicom_img_dir +'/'+ID+'.dcm'\n                dcm_data = pydicom.read_file(path)\n                im = dcm_data.pixel_array\n                msk = np.zeros(im.shape)\n                im = cv2.resize(im, dsize=(256, 256), interpolation=cv2.INTER_CUBIC)\n                im =  cv2.cvtColor(im,cv2.COLOR_GRAY2RGB)\n                X[i] = preprocess_input(np.array(im, dtype=np.float32))\n                for boxes in parsed[ID]['boxes']:\n                    # add 1's at the location of the pneumonia\n                    x1, y1,w, h = boxes\n                    y2 = y1 + h\n                    x2 = x1 + w\n                    msk[int(x1):int(x2), int(y1):int(y2)] = 1\n                msk = cv2.resize(msk, dsize=(256, 256))\n                y[i] = msk\n        return X,y","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Model Creation"},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.layers import Concatenate, UpSampling2D, Conv2D, Reshape\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.applications.mobilenet import MobileNet","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def create_model(trainable=True):\n    model = MobileNet(input_shape=(IMAGE_HEIGHT, IMAGE_WIDTH, 3), \n                      include_top=False, alpha=ALPHA, weights=\"imagenet\")\n\n    for layer in model.layers:\n        layer.trainable = trainable\n\n    block1 = model.get_layer(\"conv_pw_1_relu\").output\n    block2 = model.get_layer(\"conv_pw_3_relu\").output\n    block3 = model.get_layer(\"conv_pw_5_relu\").output\n    block4 = model.get_layer(\"conv_pw_11_relu\").output\n    block5 = model.get_layer(\"conv_pw_13_relu\").output\n\n    x = Concatenate()([UpSampling2D()(block5), block4])\n    x = Concatenate()([UpSampling2D()(x), block3])\n    x = Concatenate()([UpSampling2D()(x), block2])\n    x = Concatenate()([UpSampling2D()(x), block1])\n\n    x = Conv2D(1, kernel_size=1, activation=\"sigmoid\")(x)\n    x = Reshape((HEIGHT_CELLS, WIDTH_CELLS))(x)\n    \n    return Model(inputs=model.input, outputs=x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Give trainable=False as argument, if you want to freeze lower layers for fast training (but low accuracy)\nmodel = create_model()\n\n# Print summary\nprint(model.summary())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Print the layer details of mobilenet and improve the Upsampling\nfor layer in model.layers:\n    if('conv_pw' in layer.name and 'relu' in layer.name):\n        print(layer.name, layer.output_shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# define iou or jaccard loss function\ndef iou_loss(y_true, y_pred):\n    y_true = tf.reshape(y_true, [-1])\n    y_pred = tf.reshape(y_pred, [-1])\n    intersection = tf.reduce_sum(y_true * y_pred)\n    score = (intersection + 1.) / (tf.reduce_sum(y_true) + tf.reduce_sum(y_pred) - intersection + ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# mean iou as a metric\ndef mean_iou(y_true, y_pred):\n    y_pred = tf.round(y_pred)    \n    intersect = tf.reduce_sum(y_true * y_pred, axis=[1])\n    union = tf.reduce_sum(y_true, axis=[1]) + tf.reduce_sum(y_pred, axis=[1])\n    smooth = tf.ones(tf.shape(intersect))\n    return tf.reduce_mean((intersect + smooth) / (union - intersect + smooth))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def loss(y_true, y_pred):\n    return binary_crossentropy(y_true, y_pred) - log(dice_coefficient(y_true, y_pred) + epsilon())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def dice_coefficient(y_true, y_pred):\n    numerator = 2 * tensorflow.reduce_sum(y_true * y_pred)\n    denominator = tensorflow.reduce_sum(y_true + y_pred)\n    return numerator / (denominator + epsilon())","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}