{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import random\n\nimport numpy as np\nimport pandas as pd\n\nimport matplotlib.pyplot as plt\nfrom matplotlib.pyplot import figure\nfigure(figsize=(15, 12), dpi=120)\nimport seaborn as sns\nsns.set(style='whitegrid') #set seaborn plotting aesthetics\n%matplotlib inline\n\nimport pydicom as dcm\nfrom pathlib import Path\nimport os\nfrom tqdm.notebook import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-11-25T20:29:03.890146Z","iopub.execute_input":"2022-11-25T20:29:03.890493Z","iopub.status.idle":"2022-11-25T20:29:03.907623Z","shell.execute_reply.started":"2022-11-25T20:29:03.890461Z","shell.execute_reply":"2022-11-25T20:29:03.906678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_class = pd.read_csv('../input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv')\ntrain_labels = pd.read_csv('../input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv')\n\ntrain_path = Path('../input/rsna-pneumonia-detection-challenge/stage_2_train_images')\ntest_path = Path('../input/rsna-pneumonia-detection-challenge/stage_2_test_images')\n\ntrain_meta = pd.concat([train_labels, \n                        train_class.drop(columns=['patientId'])], axis=1)\n\nbox_df = train_meta.groupby('patientId').size().reset_index(name='boxes')\n\ntrain_ds = pd.merge(train_meta, box_df, on='patientId')\n\nbox_df = box_df.groupby('boxes').size().reset_index(name='patients')\n\n\n# List of information we needs with us\nvars = ['PatientAge','PatientSex','ImagePath']\n\n\ndef process_dicom_data(df, path):\n    \n    # adding new columns to the imported DataFrame with Null values\n    for var in vars:\n        df[var] = None\n        \n    images = os.listdir(path)\n    \n    #looping through each dicom image, extract the information from it, and \n    # add it to the DataFrame\n    \n    for i, img_name in tqdm(enumerate(images)):\n        \n        imagePath = os.path.join(path,img_name)\n        img_data = dcm.read_file(imagePath)\n        \n        idx = (df['patientId']==img_data.PatientID)\n        df.loc[idx,'PatientAge'] = pd.to_numeric(img_data.PatientAge)\n        df.loc[idx,'PatientSex'] = img_data.PatientSex\n        df.loc[idx, 'ImagePath'] = str.format(imagePath)\n\nprocess_dicom_data(train_ds,'../input/rsna-pneumonia-detection-challenge/stage_2_train_images')","metadata":{"execution":{"iopub.status.busy":"2022-11-25T20:29:07.830365Z","iopub.execute_input":"2022-11-25T20:29:07.831047Z","iopub.status.idle":"2022-11-25T20:35:18.652579Z","shell.execute_reply.started":"2022-11-25T20:29:07.831010Z","shell.execute_reply":"2022-11-25T20:35:18.651560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds.to_csv('pneumonia_ds', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-11-25T20:43:46.053473Z","iopub.execute_input":"2022-11-25T20:43:46.053878Z","iopub.status.idle":"2022-11-25T20:43:46.226223Z","shell.execute_reply.started":"2022-11-25T20:43:46.053847Z","shell.execute_reply":"2022-11-25T20:43:46.225011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Pre Processing the image","metadata":{}},{"cell_type":"code","source":"import cv2","metadata":{"execution":{"iopub.status.busy":"2022-11-25T20:46:38.930354Z","iopub.execute_input":"2022-11-25T20:46:38.930742Z","iopub.status.idle":"2022-11-25T20:46:38.963478Z","shell.execute_reply.started":"2022-11-25T20:46:38.930708Z","shell.execute_reply":"2022-11-25T20:46:38.962582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images = []\nADJUSTED_IMAGE_SIZE = 128\nimageList = []\nclassLabels = []\nlabels = []\noriginalImage = []\n\ndef readAndReshapeImage(image):\n    img = np.array(image).astype(np.uint8)\n    res = cv2.resize(img,(ADJUSTED_IMAGE_SIZE,ADJUSTED_IMAGE_SIZE), interpolation = cv2.INTER_LINEAR)\n    return res\n\ndef populateImage(rowData):\n    for index, row in rowData.iterrows():\n        patientId = row.patientId\n        classlabel = row[\"class\"]\n        dcm_file = '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/'+'{}.dcm'.format(patientId)\n        dcm_data = dcm.read_file(dcm_file)\n        img = dcm_data.pixel_array\n        \n        ## Converting the image to 3 channels as the dicom image pixel does not have colour classes wiht it\n        if len(img.shape) != 3 or img.shape[2] != 3:\n            img = np.stack((img,) * 3, -1)\n        \n        imageList.append(readAndReshapeImage(img))\n#         originalImage.append(img)\n        classLabels.append(classlabel)\n    tmpImages = np.array(imageList)\n    tmpLabels = np.array(classLabels)\n#     originalImages = np.array(originalImage)\n    return tmpImages,tmpLabels\n\nimages, labels = populateImage(train_meta)\nprint(images.shape , labels.shape)","metadata":{"execution":{"iopub.status.busy":"2022-11-25T20:46:41.150382Z","iopub.execute_input":"2022-11-25T20:46:41.151071Z","iopub.status.idle":"2022-11-25T20:54:16.921668Z","shell.execute_reply.started":"2022-11-25T20:46:41.151035Z","shell.execute_reply":"2022-11-25T20:54:16.920592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Scikitlearn suggests using OneHotEncoder for X matrix i.e. the features you feed in a model, and to use a LabelBinarizer for the y labels.\n\nThey are quite similar, except that OneHotEncoder could return a sparse matrix that saves a lot of memory and you won't really need that in y labels.\n\nEven if you have a multi-label multi-class problem, you can use MultiLabelBinarizer for your y labels rather than switching to OneHotEncoder for multi hot encoding.","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelBinarizer\nencode = LabelBinarizer()\ny = encode.fit_transform(labels)\n\n## splitting into train ,test and validation data\n\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(images, y, test_size=0.3, random_state=50)\nX_test, X_val, y_test, y_val = train_test_split(X_test,y_test, test_size = 0.5, random_state=50)","metadata":{"id":"ZGhxjLPljrXK","execution":{"iopub.status.busy":"2022-11-25T20:58:40.722067Z","iopub.execute_input":"2022-11-25T20:58:40.722495Z","iopub.status.idle":"2022-11-25T20:58:41.509563Z","shell.execute_reply.started":"2022-11-25T20:58:40.722460Z","shell.execute_reply":"2022-11-25T20:58:41.508484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CNN Model without transfer learning","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.layers import Layer, Convolution2D, Flatten, Dense\nfrom tensorflow.keras.layers import Concatenate, UpSampling2D, Conv2D, Reshape, GlobalAveragePooling2D, GlobalMaxPooling2D\nfrom tensorflow.keras.layers import Dense, Activation,Flatten,Dropout,MaxPooling2D,BatchNormalization\n\nfrom tensorflow.keras.models import Model, Sequential\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras import losses,optimizers","metadata":{"execution":{"iopub.status.busy":"2022-11-25T21:02:38.361903Z","iopub.execute_input":"2022-11-25T21:02:38.362253Z","iopub.status.idle":"2022-11-25T21:02:38.369678Z","shell.execute_reply.started":"2022-11-25T21:02:38.362224Z","shell.execute_reply":"2022-11-25T21:02:38.368495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## We start with 32 filters with 3,3 kernal and no padding , then 64 and 128 wiht drop layers in between \n## And softmax activaation as the last layer\n\ndef cnn_model(height, width, num_channels, num_classes, loss='categorical_crossentropy', metrics=['accuracy']):\n  batch_size = None\n\n  model = Sequential()\n\n  model.add(Conv2D(filters = 32, kernel_size = (3,3),padding = 'Same', \n                  activation ='relu', batch_input_shape = (batch_size,height, width, num_channels)))\n\n\n  model.add(Conv2D(filters = 32, kernel_size = (3,3),padding = 'Same', \n                  activation ='relu'))\n  model.add(MaxPooling2D(pool_size=(2,2)))\n  model.add(Dropout(0.2))\n\n\n  model.add(Conv2D(filters = 64, kernel_size = (3,3),padding = 'Same', \n                  activation ='relu'))\n  model.add(Conv2D(filters = 64, kernel_size = (3,3),padding = 'same', \n                  activation ='relu'))\n  model.add(MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n  model.add(Dropout(0.3))\n\n  model.add(Conv2D(filters = 128, kernel_size = (3,3),padding = 'Same', \n                  activation ='relu'))\n  model.add(Conv2D(filters = 128, kernel_size = (3,3),padding = 'Same', \n                  activation ='relu'))\n  model.add(MaxPooling2D(pool_size=(2,2), strides=(2,2)))\n  model.add(Dropout(0.4))\n\n  model.add(GlobalMaxPooling2D())\n  model.add(Dense(256, activation = \"relu\"))\n  model.add(Dropout(0.5))\n  model.add(Dense(num_classes, activation = \"softmax\"))\n\n  optimizer = Adam(lr=0.001)\n  model.compile(optimizer = optimizer, loss = loss, metrics = metrics)\n  model.summary()\n  return model","metadata":{"id":"6wcRGVMyjrXL","execution":{"iopub.status.busy":"2022-11-25T21:12:10.114458Z","iopub.execute_input":"2022-11-25T21:12:10.115463Z","iopub.status.idle":"2022-11-25T21:12:10.126770Z","shell.execute_reply.started":"2022-11-25T21:12:10.115422Z","shell.execute_reply":"2022-11-25T21:12:10.125724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model Summary\ncnn = cnn_model(ADJUSTED_IMAGE_SIZE,ADJUSTED_IMAGE_SIZE,3,3)","metadata":{"id":"UMv_bWwJjrXL","outputId":"e7579dfc-5622-4cbf-d252-bbef2577c763","execution":{"iopub.status.busy":"2022-11-25T21:12:15.514487Z","iopub.execute_input":"2022-11-25T21:12:15.515458Z","iopub.status.idle":"2022-11-25T21:12:15.619228Z","shell.execute_reply.started":"2022-11-25T21:12:15.515405Z","shell.execute_reply":"2022-11-25T21:12:15.618164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Training for 20 epocs with batch size of 16\nhistory = cnn.fit(X_train, \n                  y_train, \n                  epochs = 20, \n                  validation_data = (X_val,y_val),\n                  batch_size = 16)","metadata":{"id":"MkmcMUzFjrXL","outputId":"b202fbba-6dcc-47c7-ee79-9e6089618c3d","execution":{"iopub.status.busy":"2022-11-25T21:04:00.848256Z","iopub.execute_input":"2022-11-25T21:04:00.848695Z","iopub.status.idle":"2022-11-25T21:06:32.683021Z","shell.execute_reply.started":"2022-11-25T21:04:00.848661Z","shell.execute_reply":"2022-11-25T21:06:32.681428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The Model got stuck at validation accuracy of 40%. Learning rate is too high. Let us also make use of some callbacks.","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\n\ncallbacks = [\n    ReduceLROnPlateau(monitor='val_loss', factor=0.1, patience=4),\n    EarlyStopping(monitor='val_loss', patience=3, restore_best_weights=True)\n]\n","metadata":{"execution":{"iopub.status.busy":"2022-11-25T21:10:07.564951Z","iopub.execute_input":"2022-11-25T21:10:07.565322Z","iopub.status.idle":"2022-11-25T21:10:07.572201Z","shell.execute_reply.started":"2022-11-25T21:10:07.565291Z","shell.execute_reply":"2022-11-25T21:10:07.570842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = cnn.fit(X_train, \n                  y_train, \n                  epochs = 20, \n                  validation_data = (X_val,y_val),\n                  batch_size = 16,\n                 callbacks = callbacks)","metadata":{"execution":{"iopub.status.busy":"2022-11-25T21:12:23.699828Z","iopub.execute_input":"2022-11-25T21:12:23.700214Z","iopub.status.idle":"2022-11-25T21:17:21.809388Z","shell.execute_reply.started":"2022-11-25T21:12:23.700181Z","shell.execute_reply":"2022-11-25T21:17:21.808437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Training accuracy is around 60 percent whereas validation accuracy is round 58 percent. We have avoided overfitting, but it seems to be clear that a normal CNN will not help us. ","metadata":{}},{"cell_type":"code","source":"fcl_loss, fcl_accuracy = cnn.evaluate(X_test, y_test, verbose=1)\nprint('Test loss:', fcl_loss)\nprint('Test accuracy:', fcl_accuracy)","metadata":{"id":"SwNDgKvVjrXM","outputId":"1fa9456d-87b5-49a8-8dc2-03df4511d8ba","execution":{"iopub.status.busy":"2022-11-25T21:21:21.604695Z","iopub.execute_input":"2022-11-25T21:21:21.605145Z","iopub.status.idle":"2022-11-25T21:21:24.897001Z","shell.execute_reply.started":"2022-11-25T21:21:21.605090Z","shell.execute_reply":"2022-11-25T21:21:24.896093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Test accuracy is also 61 percent is. At least our model is consistent. Hehe. ","metadata":{}},{"cell_type":"code","source":"## PLottting the accuracy vs loss graph\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\nepochs_range = range(11)\n\nplt.figure(figsize=(15, 15))\nplt.subplot(2, 2, 1)\nplt.plot(epochs_range, acc, label='Training Accuracy')\nplt.plot(epochs_range, val_acc, label='Validation Accuracy')\nplt.legend(loc='lower right')\nplt.title('Training and Validation Accuracy')\n\nplt.subplot(2, 2, 2)\nplt.plot(epochs_range, loss, label='Training Loss')\nplt.plot(epochs_range, val_loss, label='Validation Loss')\nplt.legend(loc='upper right')\nplt.title('Training and Validation Loss')\nplt.show()\n\n#The training and Validation loss is almost same, but for the training and validation accuracy chart, \n#the validation accuracy falls down in the later epochs, this could be because we have only taken 200 images for processing.","metadata":{"id":"JuOIpAcBjrXM","outputId":"3e832382-3838-4fba-a70a-3865244eebe8","execution":{"iopub.status.busy":"2022-11-25T21:22:33.383166Z","iopub.execute_input":"2022-11-25T21:22:33.383513Z","iopub.status.idle":"2022-11-25T21:22:33.906610Z","shell.execute_reply.started":"2022-11-25T21:22:33.383483Z","shell.execute_reply":"2022-11-25T21:22:33.905587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## FUnction to create a dataframe for results\ndef createResultDf(name,accuracy,testscore):\n    result = pd.DataFrame({'Method':[name], 'accuracy': [accuracy] ,'Test Score':[testscore]})\n    return result","metadata":{"execution":{"iopub.status.busy":"2022-11-25T21:22:42.666762Z","iopub.execute_input":"2022-11-25T21:22:42.667165Z","iopub.status.idle":"2022-11-25T21:22:42.673202Z","shell.execute_reply.started":"2022-11-25T21:22:42.667110Z","shell.execute_reply":"2022-11-25T21:22:42.671861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resultDF = createResultDf(\"CNN\",acc[-1],fcl_accuracy)","metadata":{"execution":{"iopub.status.busy":"2022-11-25T21:22:45.948869Z","iopub.execute_input":"2022-11-25T21:22:45.949225Z","iopub.status.idle":"2022-11-25T21:22:45.955089Z","shell.execute_reply.started":"2022-11-25T21:22:45.949194Z","shell.execute_reply":"2022-11-25T21:22:45.953947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resultDF","metadata":{"execution":{"iopub.status.busy":"2022-11-25T21:22:57.260088Z","iopub.execute_input":"2022-11-25T21:22:57.260609Z","iopub.status.idle":"2022-11-25T21:22:57.279420Z","shell.execute_reply.started":"2022-11-25T21:22:57.260563Z","shell.execute_reply":"2022-11-25T21:22:57.278385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport itertools\nplt.subplots(figsize=(22,7)) #set the size of the plot \n\ndef plot_confusion_matrix(cm, classes,\n                          normalize=False,\n                          title='Confusion matrix',\n                          cmap=plt.cm.Blues):\n    plt.imshow(cm, interpolation='nearest', cmap=cmap)\n    plt.title(title)\n    plt.colorbar()\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation=45)\n    plt.yticks(tick_marks, classes)\n\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n\n    thresh = cm.max() / 2.\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n        plt.text(j, i, cm[i, j],\n                 horizontalalignment=\"center\",\n                 color=\"white\" if cm[i, j] > thresh else \"black\")\n\n    plt.tight_layout()\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n\n# Predict the values from the validation dataset\nY_pred = cnn.predict(X_test)\n# Convert predictions classes to one hot vectors \nY_pred_classes = np.argmax(Y_pred,axis = 1) \n# Convert validation observations to one hot vectors\nY_true = np.argmax(y_test,axis = 1) \n# compute the confusion matrix\nconfusion_mtx = confusion_matrix(Y_true, Y_pred_classes) \n# plot the confusion matrix\nplot_confusion_matrix(confusion_mtx, classes = range(3))\n\n#Class 0 ,1 and 2\n#Class 0 is Lung Opacity\n#Class 1 is No Lung Opacity/Normal, the model has predicted mostly wrong in this case to the Target 0. Type 2 error\n#Class 2 is Normal\n","metadata":{"id":"64Vf2LX3jrXM","outputId":"42ab677f-85e4-49fd-e3bf-243c904719a2","execution":{"iopub.status.busy":"2022-11-25T21:23:06.782542Z","iopub.execute_input":"2022-11-25T21:23:06.782918Z","iopub.status.idle":"2022-11-25T21:23:08.853585Z","shell.execute_reply.started":"2022-11-25T21:23:06.782887Z","shell.execute_reply":"2022-11-25T21:23:08.852412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import recall_score, confusion_matrix, precision_score, f1_score, accuracy_score, roc_auc_score,classification_report\nfrom sklearn.metrics import classification_report\n\nY_truepred = np.argmax(y_test,axis = 1) \n\nY_testPred = cnn.predict(X_test)\n# Convert predictions classes to one hot vectors \nY_pred_classes = np.argmax(Y_pred,axis = 1) \n\nreportData = classification_report(Y_truepred, Y_pred_classes,output_dict=True)\n\nfor data in reportData:\n    if(data == '-1' or data == '1'):\n        if(type(reportData[data]) is dict):\n            for subData in reportData[data]:\n                resultDF[data+\"_\"+subData] = reportData[data][subData]\n\nresultDF","metadata":{"execution":{"iopub.status.busy":"2022-11-25T21:23:17.772648Z","iopub.execute_input":"2022-11-25T21:23:17.773001Z","iopub.status.idle":"2022-11-25T21:23:19.524689Z","shell.execute_reply.started":"2022-11-25T21:23:17.772971Z","shell.execute_reply":"2022-11-25T21:23:19.523645Z"},"trusted":true},"execution_count":null,"outputs":[]}]}