{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nfrom glob import glob\nfrom tqdm.notebook import tqdm\nfrom ast import literal_eval\nimport os\nimport matplotlib.pyplot as plt\nimport matplotlib\nmatplotlib.rcParams.update({'font.size': 22})\nfrom sklearn.metrics import accuracy_score\nfrom skimage import exposure\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.utils import plot_model\nimport cv2\nfrom matplotlib.patches import Rectangle\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import ResNet50, DenseNet121, Xception\nfrom tensorflow.keras.layers import Dense, Flatten, Conv2D, MaxPooling2D, Dropout, GlobalAveragePooling2D\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras import models\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, ModelCheckpoint, EarlyStopping\nimport tensorflow.keras.backend as K\nfrom tensorflow.math import confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:07:00.276130Z","iopub.execute_input":"2021-06-01T07:07:00.276542Z","iopub.status.idle":"2021-06-01T07:07:06.313539Z","shell.execute_reply.started":"2021-06-01T07:07:00.276461Z","shell.execute_reply":"2021-06-01T07:07:06.312640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data manipulations","metadata":{}},{"cell_type":"code","source":"df_image = pd.read_csv('../input/siim-covid19-detection/train_image_level.csv')\ndisplay(df_image.head(3))\nprint(df_image.shape)","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:07:06.315023Z","iopub.execute_input":"2021-06-01T07:07:06.315393Z","iopub.status.idle":"2021-06-01T07:07:06.375833Z","shell.execute_reply.started":"2021-06-01T07:07:06.315353Z","shell.execute_reply":"2021-06-01T07:07:06.374882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_study = pd.read_csv('../input/siim-covid19-detection/train_study_level.csv')\ndisplay(df_study.head(3))\nprint(df_study.shape)","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:07:06.377715Z","iopub.execute_input":"2021-06-01T07:07:06.378079Z","iopub.status.idle":"2021-06-01T07:07:06.397710Z","shell.execute_reply.started":"2021-06-01T07:07:06.378017Z","shell.execute_reply":"2021-06-01T07:07:06.396782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sampleSub = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\ndisplay(df_sampleSub.head(3))\nprint(df_sampleSub.shape)","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:07:06.399481Z","iopub.execute_input":"2021-06-01T07:07:06.399838Z","iopub.status.idle":"2021-06-01T07:07:06.420319Z","shell.execute_reply.started":"2021-06-01T07:07:06.399801Z","shell.execute_reply":"2021-06-01T07:07:06.419578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_study['id'] = df_study['id'].str.replace('_study',\"\")\ndf_study.rename({'id': 'StudyInstanceUID'},axis=1, inplace=True)\ndf_study.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:07:06.422617Z","iopub.execute_input":"2021-06-01T07:07:06.422897Z","iopub.status.idle":"2021-06-01T07:07:06.442320Z","shell.execute_reply.started":"2021-06-01T07:07:06.422872Z","shell.execute_reply":"2021-06-01T07:07:06.439703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_image.merge(df_study, on='StudyInstanceUID')\ndf_train.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:07:06.443542Z","iopub.execute_input":"2021-06-01T07:07:06.443912Z","iopub.status.idle":"2021-06-01T07:07:06.465960Z","shell.execute_reply.started":"2021-06-01T07:07:06.443884Z","shell.execute_reply":"2021-06-01T07:07:06.465154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_dir_jpg = '../input/covid-jpg-512/train'\n# train_dir_origin ='../input/siim-covid19-detection/train'\n# paths_original = []\n# paths_jpg = []\n# for _, row in tqdm(df_train.iterrows()):\n#     image_id = row['id'].split('_')[0]\n#     study_id = row['StudyInstanceUID']\n#     image_path_jpg = glob(f'{train_dir_jpg}/{image_id}.jpg')\n#     image_path_original = glob(f'{train_dir_origin}/{study_id}/*/{image_id}.dcm')\n#     paths_jpg.append(image_path_jpg)\n#     paths_original.append(image_path_original)","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:07:06.467240Z","iopub.execute_input":"2021-06-01T07:07:06.467581Z","iopub.status.idle":"2021-06-01T07:07:06.471279Z","shell.execute_reply.started":"2021-06-01T07:07:06.467546Z","shell.execute_reply":"2021-06-01T07:07:06.470333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_train['path'] = paths_jpg\n# df_train['origin'] = paths_original\n# df_train.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:07:06.473420Z","iopub.execute_input":"2021-06-01T07:07:06.474024Z","iopub.status.idle":"2021-06-01T07:07:06.484592Z","shell.execute_reply.started":"2021-06-01T07:07:06.473986Z","shell.execute_reply":"2021-06-01T07:07:06.483718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.loc[df_train['Negative for Pneumonia']==1, 'study_label'] = 'negative'\ndf_train.loc[df_train['Typical Appearance']==1, 'study_label'] = 'typical'\ndf_train.loc[df_train['Indeterminate Appearance']==1, 'study_label'] = 'indeterminate'\ndf_train.loc[df_train['Atypical Appearance']==1, 'study_label'] = 'atypical'\ndf_train.drop(['Negative for Pneumonia','Typical Appearance', 'Indeterminate Appearance', 'Atypical Appearance'], axis=1, inplace=True)\ndf_train['id'] = df_train['id'].str.replace('_image', '.jpg')\ndf_train['image_label'] = df_train['label'].str.split().apply(lambda x : x[0])\ndf_train.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:07:06.486992Z","iopub.execute_input":"2021-06-01T07:07:06.488086Z","iopub.status.idle":"2021-06-01T07:07:06.687299Z","shell.execute_reply.started":"2021-06-01T07:07:06.488059Z","shell.execute_reply":"2021-06-01T07:07:06.686337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_size = pd.read_csv('../input/covid-jpg-512/size.csv')\ndf_size.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:07:06.688643Z","iopub.execute_input":"2021-06-01T07:07:06.688975Z","iopub.status.idle":"2021-06-01T07:07:06.713003Z","shell.execute_reply.started":"2021-06-01T07:07:06.688939Z","shell.execute_reply":"2021-06-01T07:07:06.712088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.merge(df_size, on='id')\ndf_train.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:07:07.651880Z","iopub.execute_input":"2021-06-01T07:07:07.652323Z","iopub.status.idle":"2021-06-01T07:07:07.688413Z","shell.execute_reply.started":"2021-06-01T07:07:07.652281Z","shell.execute_reply":"2021-06-01T07:07:07.687605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualization","metadata":{}},{"cell_type":"code","source":"n = 20\ntrain_dir = '../input/covid-jpg-512/train'\nfig, axs = plt.subplots(4, 5, figsize=(20,20))\nfig.subplots_adjust(hspace=.2, wspace=.2)\naxs = axs.ravel()\nfor i in range(n):\n    img = cv2.imread(os.path.join(train_dir, df_train['id'][i]))\n    axs[i].imshow(img)\n    if type(df_train['boxes'][i])==str:\n        boxes = literal_eval(df_train['boxes'][i])\n        for box in boxes:\n            axs[i].add_patch(Rectangle((box['x']*(512/df_train['dim1'][i]), box['y']*(512/df_train['dim0'][i])), box['width']*(512/df_train['dim1'][i]), box['height']*(512/df_train['dim0'][i]), fill=0, color='y', linewidth=2))\n            axs[i].set_title(df_train['study_label'][i])\n    else:\n        axs[i].set_title(df_train['study_label'][i])","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:07:08.724693Z","iopub.execute_input":"2021-06-01T07:07:08.725038Z","iopub.status.idle":"2021-06-01T07:07:11.047892Z","shell.execute_reply.started":"2021-06-01T07:07:08.725001Z","shell.execute_reply":"2021-06-01T07:07:11.047081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PreProcessing","metadata":{}},{"cell_type":"code","source":"def preprocess_image(img):\n    equ_img = exposure.equalize_hist(img)\n    return equ_img\n\nim= cv2.imread('../input/covid-jpg-512/train/007cf31356c6.jpg')\nim2 = preprocess_image(im)\nres = np.concatenate((im/255, im2), axis=1)\nplt.imshow(res)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:07:11.049292Z","iopub.execute_input":"2021-06-01T07:07:11.049658Z","iopub.status.idle":"2021-06-01T07:07:11.276870Z","shell.execute_reply.started":"2021-06-01T07:07:11.049617Z","shell.execute_reply":"2021-06-01T07:07:11.275913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ImageGenerators and Augmentations","metadata":{}},{"cell_type":"code","source":"img_size = 299\nbatch_size = 16\n\nimage_generator = ImageDataGenerator(\n        validation_split=0.2,\n        #rotation_range=20,\n        horizontal_flip = True,\n        zoom_range = 0.1,\n        #shear_range = 0.1,\n        brightness_range = [0.8, 1.1],\n        fill_mode='nearest',\n        preprocessing_function=preprocess_image\n)\n\nimage_generator_valid = ImageDataGenerator(validation_split=0.2,preprocessing_function=preprocess_image)\n\ntrain_generator = image_generator.flow_from_dataframe(\n        dataframe = df_train,\n        directory='../input/covid-jpg-512/train',\n        x_col = 'id',\n        y_col =  'study_label',  \n        target_size=(img_size, img_size),\n        batch_size=batch_size,\n        subset='training', seed = 23) \n\nvalid_generator=image_generator_valid.flow_from_dataframe(\n    dataframe = df_train,\n    directory='../input/covid-jpg-512/train',\n    x_col = 'id',\n    y_col = 'study_label',\n    target_size=(img_size, img_size),\n    batch_size=batch_size,\n    subset='validation', shuffle=False,  seed=23) ","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:07:11.319319Z","iopub.execute_input":"2021-06-01T07:07:11.319602Z","iopub.status.idle":"2021-06-01T07:07:17.439082Z","shell.execute_reply.started":"2021-06-01T07:07:11.319575Z","shell.execute_reply":"2021-06-01T07:07:17.437423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for j in range(2):\n    aug_images = [train_generator[0][0][j] for i in range(5)]\n    fig, axes = plt.subplots(1, 5, figsize=(24,24))\n    axes = axes.flatten()\n    for img, ax in zip(aug_images, axes):\n        ax.imshow(img)\n        ax.axis('off')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:07:17.440657Z","iopub.execute_input":"2021-06-01T07:07:17.441012Z","iopub.status.idle":"2021-06-01T07:07:22.459876Z","shell.execute_reply.started":"2021-06-01T07:07:17.440975Z","shell.execute_reply":"2021-06-01T07:07:22.458858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Architecture","metadata":{}},{"cell_type":"code","source":"pre_model = Xception(weights='imagenet', \n                  include_top = False, \n                  input_shape=(img_size, img_size, 3))\npre_model.trainable=True","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:07:22.461781Z","iopub.execute_input":"2021-06-01T07:07:22.462150Z","iopub.status.idle":"2021-06-01T07:07:26.227394Z","shell.execute_reply.started":"2021-06-01T07:07:22.462090Z","shell.execute_reply":"2021-06-01T07:07:26.226511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = pre_model.output\nx = GlobalAveragePooling2D()(x)\noutput = Dense(4, activation='softmax')(x)\nmodel = models.Model(pre_model.input, output)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:07:37.257161Z","iopub.execute_input":"2021-06-01T07:07:37.257534Z","iopub.status.idle":"2021-06-01T07:07:37.341646Z","shell.execute_reply.started":"2021-06-01T07:07:37.257504Z","shell.execute_reply":"2021-06-01T07:07:37.340792Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_model(model)","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:07:50.424722Z","iopub.execute_input":"2021-06-01T07:07:50.425051Z","iopub.status.idle":"2021-06-01T07:07:52.081543Z","shell.execute_reply.started":"2021-06-01T07:07:50.425021Z","shell.execute_reply":"2021-06-01T07:07:52.073938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(Adam(lr=1e-3),loss='categorical_crossentropy',metrics='categorical_accuracy')","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:07:52.083331Z","iopub.execute_input":"2021-06-01T07:07:52.083622Z","iopub.status.idle":"2021-06-01T07:07:52.108209Z","shell.execute_reply.started":"2021-06-01T07:07:52.083592Z","shell.execute_reply":"2021-06-01T07:07:52.107315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rlr = ReduceLROnPlateau(monitor = 'val_loss', factor = 0.1, patience = 2, verbose = 1, \n                                min_delta = 1e-4, min_lr = 1e-6, mode = 'min')\nes = EarlyStopping(monitor = 'val_loss', min_delta = 1e-4, patience = 5, mode = 'min', \n                    restore_best_weights = True, verbose = 1)\nckp = ModelCheckpoint('model.h5',monitor = 'val_loss',\n                      verbose = 0, save_best_only = True, mode = 'min')\nhistory = model.fit(\n      train_generator,\n      epochs=25,\n      validation_data=valid_generator,\n      callbacks=[es, rlr, ckp],\n      verbose=1)\n\nK.clear_session()","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:07:52.109964Z","iopub.execute_input":"2021-06-01T07:07:52.110358Z","iopub.status.idle":"2021-06-01T07:08:03.810294Z","shell.execute_reply.started":"2021-06-01T07:07:52.110319Z","shell.execute_reply":"2021-06-01T07:08:03.806985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"actual =  valid_generator.labels\npreds = np.argmax(model.predict(valid_generator), axis=1)\ncfmx = confusion_matrix(actual, preds)\nacc = accuracy_score(actual, preds)\nprint ('Test Accuracy:', acc )\nprint('Confusion matrix:', cfmx)","metadata":{"execution":{"iopub.status.busy":"2021-05-29T02:30:03.315428Z","iopub.status.idle":"2021-05-29T02:30:03.320746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hist = pd.DataFrame(history.history)\nfig, (ax1, ax2) = plt.subplots(figsize=(12,12),nrows=2, ncols=1)\nhist['loss'].plot(ax=ax1,c='k',label='training loss')\nhist['val_loss'].plot(ax=ax1,c='r',linestyle='--', label='validation loss')\nax1.legend()\nhist['categorical_accuracy'].plot(ax=ax2,c='k',label='training accuracy')\nhist['val_categorical_accuracy'].plot(ax=ax2,c='r',linestyle='--',label='validation accuracy')\nax2.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-29T02:30:03.322357Z","iopub.status.idle":"2021-05-29T02:30:03.323444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}