{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import os,shutil\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom PIL import Image\nimport pydicom as dicom\nfrom skimage.transform import resize\nimport cv2\nimport seaborn as sns\nsns.set_style('darkgrid')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"os.listdir('/kaggle/input/rsna-pneumonia-detection-challenge')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df=pd.read_csv('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv')\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['path']='/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/'+df['patientId'].astype(str)+'.dcm'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['path'][0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"negative=df[df['Target']==0]\nprint(len(negative))\nnegative.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"positive=df[df['Target']==1]\nunique_positive=positive[['path','patientId']]\npath=unique_positive['path'].unique()\npatientId=unique_positive['patientId'].unique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"unique_positive=pd.DataFrame({'path':path,'patientId':patientId})\nlen(unique_positive)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"os.mkdir('/kaggle/working/data')\n\nos.mkdir('/kaggle/working/data/positive')\n\nos.mkdir('/kaggle/working/data/negative')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"os.chdir('/kaggle/working')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for _,row in tqdm(unique_positive.iterrows()):\n    img=dicom.read_file(row['path']).pixel_array\n    img=resize(img,(256,256))\n    plt.imsave('data/positive/'+row['patientId']+'.jpg',img,cmap='gray')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for _,row in tqdm(negative.iterrows()):\n    img=dicom.read_file(row['path']).pixel_array\n    img=resize(img,(256,256))\n    plt.imsave('data/negative/'+row['patientId']+'.jpg',img,cmap='gray')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(30,20))\nfor j,img in enumerate(os.listdir('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images')):\n    path=os.path.join('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images',img)\n    tar=df[df['path']==path]['Target'].values[0]\n    img=dicom.read_file(path).pixel_array\n    plt.subplot(4,4,j+1)\n    plt.axis('off')\n    if tar==0:\n        plt.title('Negative')\n    else:\n        plt.title('Positive')\n        \n        s=df[df['path']==path]\n        \n        for _,row in s.iterrows():\n            x=int(row['x'])\n            y=int(row['y'])\n            w=int(row['width'])\n            h=int(row['height'])\n            cv2.rectangle(img,(x,y),(x+h,y+h),(255,255,0),5)\n    plt.imshow(img,cmap='gray')\n    if(j==15):\n        break","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.applications.vgg16 import VGG16,preprocess_input\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"datagen=ImageDataGenerator(samplewise_center=True,samplewise_std_normalization=True,horizontal_flip=True,\n                          width_shift_range=0.05,rescale=1/255,fill_mode='nearest',height_shift_range=0.05,\n                           preprocessing_function=preprocess_input,validation_split=0.3,\n                          )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train=datagen.flow_from_directory('data',color_mode='rgb',batch_size=128,class_mode='binary',subset='training')\ntest=datagen.flow_from_directory('data',color_mode='rgb',batch_size=32,class_mode='binary',subset='validation')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.class_indices","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pre_trained_model = VGG16(input_shape = (256,256,3), \n                                include_top = False, \n                                weights = 'imagenet')\n\nfor layer in pre_trained_model.layers:\n  layer.trainable = False\n\n# pre_trained_model.summary()\n\nlast_layer = pre_trained_model.get_layer('block5_pool')\nprint('last layer output shape: ', last_layer.output_shape)\nlast_output = last_layer.output","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.layers import Flatten,Dense,Dropout,BatchNormalization,ReLU,GaussianDropout\n\nmodel = Flatten()(last_output)\nmodel = Dense(1024)(model)\nmodel=ReLU(0.1)(model)\nmodel=Dropout(0.25)(model)\nmodel=BatchNormalization()(model)\nmodel = Dense(1024)(model)\nmodel=ReLU(0.1)(model)\nmodel=Dropout(0.25)(model)\nmodel=BatchNormalization()(model)\nmodel = Dense(1, activation='sigmoid')(model)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.models import Model\n\n\nfmodel = Model( pre_trained_model.input, model) \n\nfmodel.compile(optimizer = 'adam', \n              loss = 'binary_crossentropy', \n              metrics = ['accuracy'])\nfmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping,ReduceLROnPlateau\n\n\nearly=EarlyStopping(monitor='accuracy',patience=3,mode='auto')\nreduce_lr = ReduceLROnPlateau(monitor='accuracy', factor=0.5, patience=2, verbose=1,cooldown=0, mode='auto',min_delta=0.0001, min_lr=1e-5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_weight={0:1,1:3.3}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fmodel.fit(train,epochs=20,callbacks=[reduce_lr],steps_per_epoch=100,validation_data=test,class_weight=class_weight)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fmodel.save('/kaggle/working/model_vgg16.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Plot Accuracy\nplt.figure(figsize=(30,20))\nval_acc=np.asarray(fmodel.history.history['val_accuracy'])*100\nacc=np.asarray(fmodel.history.history['accuracy'])*100\nacc=pd.DataFrame({'val_acc':val_acc,'acc':acc})\nacc.plot(figsize=(20,10),yticks=range(50,100,5))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Plot loss\nloss=fmodel.history.history['loss']\nval_loss=fmodel.history.history['val_loss']\nloss=pd.DataFrame({'val_loss':val_loss,'loss':loss})\nloss.plot(figsize=(20,10))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y=[]\n\ntest.reset()\n\nfor i in tqdm(range(84)):\n    _,tar=test.__getitem__(i)\n    for j in tar:\n        y.append(j)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.reset()\ny_pred=fmodel.predict(test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# define iou or jaccard loss function\ndef iou_loss(y_true, y_pred):\n    #print(y_true)\n    y_true=tf.cast(y_true, tf.float32)\n    y_pred=tf.cast(y_pred, tf.float32)\n    y_true = tf.reshape(y_true, [-1])\n    y_pred = tf.reshape(y_pred, [-1])\n   \n    intersection = tf.reduce_sum(y_true * y_pred)\n    score = (intersection + 1.) / (tf.reduce_sum(y_true) + tf.reduce_sum(y_pred) - intersection + 1.)\n    return 1 - score\n\n# combine bce loss and iou loss\ndef iou_bce_loss(y_true, y_pred):\n    return 0.5 * keras.losses.binary_crossentropy(y_true, y_pred) + 0.5 * iou_loss(y_true, y_pred)\n\n# mean iou as a metric\ndef mean_iou(y_true, y_pred):\n    y_pred = tf.round(y_pred)\n    intersect = tf.reduce_sum(y_true * y_pred, axis=[1, 2, 3])\n    union = tf.reduce_sum(y_true, axis=[1, 2, 3]) + tf.reduce_sum(y_pred, axis=[1, 2, 3])\n    smooth = tf.ones(tf.shape(intersect))\n    return tf.reduce_mean((intersect + smooth) / (union - intersect + smooth))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred=[]\nfor i in y_pred:\n    if i[0]>=0.5:\n        pred.append(1)\n    else:\n        pred.append(0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.metrics import roc_curve,auc,precision_recall_curve,classification_report","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Classification Report\n# print(classification_report(y,pred))  --> error for Found input variables with inconsistent numbers of samples: [2688, 8004]\nprint(classification_report(y,pred[:len(y)]))\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We can see that the F1-score for normal category is very high - 0.72, whereas for pneumonia,\nit’s just 0.52. This clearly shows the effect of imbalanced dataset. Let’s have a look at the area\nunder the ROC curve to assess the model performance.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(30,20))\nfpr,tpr,_=roc_curve(y,y_pred[:len(y)])\narea_under_curve=auc(fpr,tpr)\nprint('The area under the curve is:',area_under_curve)\n# Plot area under curve\nplt.plot(fpr,tpr,'b.-')\nplt.xlabel('false positive rate')\nplt.ylabel('true positive rate')\nplt.plot(fpr,fpr,linestyle='--',color='black')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}