{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport shutil\nimport cv2\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras.applications.vgg16 import preprocess_input, VGG16\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\nfrom sklearn.model_selection import train_test_split\nimport yaml\n\nfrom kaggle_secrets import UserSecretsClient\nimport cv2\nimport pydicom\n\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport skimage.io\nimport tqdm\nimport glob\nimport tensorflow \n\nfrom tqdm import tqdm\nfrom sklearn.utils import shuffle\nfrom sklearn.model_selection import train_test_split\nfrom keras.metrics import Recall,Precision\nfrom skimage.io import imread, imshow\nfrom skimage.transform import resize\nfrom skimage.color import grey2rgb\n\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import InputLayer, BatchNormalization, Dropout, Flatten, Dense, Activation, MaxPool2D, Conv2D\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom tensorflow.keras.applications.vgg19 import VGG19\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array","metadata":{"papermill":{"duration":6.264186,"end_time":"2021-07-02T07:19:54.181951","exception":false,"start_time":"2021-07-02T07:19:47.917765","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-09T15:23:54.300153Z","iopub.execute_input":"2021-07-09T15:23:54.300522Z","iopub.status.idle":"2021-07-09T15:24:00.053118Z","shell.execute_reply.started":"2021-07-09T15:23:54.300447Z","shell.execute_reply":"2021-07-09T15:24:00.052156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SIIM_COVID19_DETECTION_DIR = '/kaggle/input/siim-covid19-detection/'\nPART0_RESIZED_DIR = '../input/siim-covid19-resized-to-256px-jpg'\n\n\nTEMP_DIR = '/kaggle/temp/'\n\nINPUT_DIR = PART0_RESIZED_DIR+'/train/'\n\nOUTPUT_DIR = DATASET_DIR = TEMP_DIR+'/train/'\nTRAIN_DIR = DATASET_DIR + 'train/'\nTA_DIR = TRAIN_DIR+'ta/'\nIA_DIR = TRAIN_DIR+'ia/'\nAA_DIR = TRAIN_DIR+'aa/'\nNP_DIR = TRAIN_DIR+'np/'\n\nWORKING_DIR = '/kaggle/working/'\n\nWANDB_PROJECT_NAME = 'project8-kaggle-covid19'\nWANDB_ENTITY_NAME = ''\n\nTRAIN_IMAGE_LEVEL_PATH = SIIM_COVID19_DETECTION_DIR+'train_image_level.csv'\nTRAIN_STUDY_LEVEL_PATH = SIIM_COVID19_DETECTION_DIR+'train_study_level.csv'\nMETA_PATH = PART0_RESIZED_DIR+'meta.csv'\n\nBATCH_SIZE = 32\nEPOCHS = 25\nIMG_SIZE = WIDTH = HEIGHT = 224\nLEARNING_RATE = 0.00008\n\nINTERPOLATION = cv2.INTER_LANCZOS4","metadata":{"papermill":{"duration":0.021199,"end_time":"2021-07-02T07:19:54.216486","exception":false,"start_time":"2021-07-02T07:19:54.195287","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-09T15:24:00.054522Z","iopub.execute_input":"2021-07-09T15:24:00.054828Z","iopub.status.idle":"2021-07-09T15:24:00.064056Z","shell.execute_reply.started":"2021-07-09T15:24:00.054795Z","shell.execute_reply":"2021-07-09T15:24:00.060639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_image_level = pd.read_csv(TRAIN_IMAGE_LEVEL_PATH)\ndf_train_study_level = pd.read_csv(TRAIN_STUDY_LEVEL_PATH)\n\ndf_train_image_level['id'] = df_train_image_level.apply(lambda row: row.id.split('_')[0], axis=1)\ndf_train_image_level['path'] = df_train_image_level.apply(lambda row: INPUT_DIR+row.id+'.jpg', axis=1)\ndf_train_image_level['image_level'] = df_train_image_level.apply(lambda row: row.label.split(' ')[0], axis=1)\n\ndf_train_study_level['id'] = df_train_study_level.apply(lambda row: row.id.split('_')[0], axis=1)\ndf_train_study_level.columns = ['StudyInstanceUID', 'Negative for Pneumonia', 'Typical Appearance', 'Indeterminate Appearance', 'Atypical Appearance']","metadata":{"papermill":{"duration":0.405043,"end_time":"2021-07-02T07:19:54.633984","exception":false,"start_time":"2021-07-02T07:19:54.228941","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-09T15:24:00.067471Z","iopub.execute_input":"2021-07-09T15:24:00.067797Z","iopub.status.idle":"2021-07-09T15:24:00.631703Z","shell.execute_reply.started":"2021-07-09T15:24:00.067772Z","shell.execute_reply":"2021-07-09T15:24:00.630867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_image_level = df_train_image_level.merge(df_train_study_level, on='StudyInstanceUID',how=\"left\")\ndf_train_image_level = df_train_image_level[['id','StudyInstanceUID','path','Negative for Pneumonia','Typical Appearance','Indeterminate Appearance','Atypical Appearance']]\ndf_train_image_level = df_train_image_level.dropna()\ndf_train_image_level = df_train_image_level[~df_train_image_level.duplicated(subset=['StudyInstanceUID'], keep='first')]\ndf_train_image_level = df_train_image_level.reset_index(drop=True)","metadata":{"papermill":{"duration":0.051349,"end_time":"2021-07-02T07:19:54.698351","exception":false,"start_time":"2021-07-02T07:19:54.647002","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-09T15:24:00.634150Z","iopub.execute_input":"2021-07-09T15:24:00.634673Z","iopub.status.idle":"2021-07-09T15:24:00.679026Z","shell.execute_reply.started":"2021-07-09T15:24:00.634633Z","shell.execute_reply":"2021-07-09T15:24:00.678259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"[os.makedirs(dir, exist_ok=True) for dir in [TA_DIR,IA_DIR,AA_DIR,NP_DIR]]\nfor i in tqdm(range(len(df_train_image_level))):\n    row = df_train_image_level.loc[i]\n    if row['Typical Appearance']:\n        shutil.copy(row.path, f'{TA_DIR}{row.id}.jpg')\n    elif row['Indeterminate Appearance']:\n        shutil.copy(row.path, f'{IA_DIR}{row.id}.jpg')\n    elif row['Atypical Appearance']:\n        shutil.copy(row.path, f'{AA_DIR}{row.id}.jpg')\n    elif row['Negative for Pneumonia']:\n        shutil.copy(row.path, f'{NP_DIR}{row.id}.jpg')\n    else:\n        print('Error: check df_train_image_level')","metadata":{"papermill":{"duration":50.48131,"end_time":"2021-07-02T07:20:45.192548","exception":false,"start_time":"2021-07-02T07:19:54.711238","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-09T15:24:00.680267Z","iopub.execute_input":"2021-07-09T15:24:00.680587Z","iopub.status.idle":"2021-07-09T15:24:26.917901Z","shell.execute_reply.started":"2021-07-09T15:24:00.680553Z","shell.execute_reply":"2021-07-09T15:24:26.917068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datagen_kwargs = dict(validation_split=.20,\n                      preprocessing_function=preprocess_input\n                     )\ndataflow_kwargs = dict(target_size=(IMG_SIZE, IMG_SIZE),\n                       batch_size=BATCH_SIZE,\n                       interpolation=\"lanczos\"\n                      )\n\nvalid_datagen = tf.keras.preprocessing.image.ImageDataGenerator(**datagen_kwargs)\nvalid_generator = valid_datagen.flow_from_directory(TRAIN_DIR,\n                                                    subset=\"validation\",\n                                                    shuffle=False,\n                                                    **dataflow_kwargs)\n\ntrain_datagen = tf.keras.preprocessing.image.ImageDataGenerator(\n    rotation_range=40,\n    horizontal_flip=True,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    **datagen_kwargs)\ntrain_generator = train_datagen.flow_from_directory(TRAIN_DIR,\n                                                    subset=\"training\",\n                                                    shuffle=True,\n                                                    **dataflow_kwargs)\n\nprint('classes :', train_generator.class_indices)","metadata":{"papermill":{"duration":0.558579,"end_time":"2021-07-02T07:20:45.889452","exception":false,"start_time":"2021-07-02T07:20:45.330873","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-09T15:24:26.919007Z","iopub.execute_input":"2021-07-09T15:24:26.919357Z","iopub.status.idle":"2021-07-09T15:24:27.247461Z","shell.execute_reply.started":"2021-07-09T15:24:26.919328Z","shell.execute_reply":"2021-07-09T15:24:27.246481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model Initialization\n\nbase_model = VGG19(input_shape=(224,224,3), \n                         include_top=False,\n                         weights=\"imagenet\")","metadata":{"papermill":{"duration":7.237286,"end_time":"2021-07-02T07:20:53.260705","exception":false,"start_time":"2021-07-02T07:20:46.023419","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-09T15:24:27.248710Z","iopub.execute_input":"2021-07-09T15:24:27.249056Z","iopub.status.idle":"2021-07-09T15:24:31.033815Z","shell.execute_reply.started":"2021-07-09T15:24:27.249020Z","shell.execute_reply":"2021-07-09T15:24:31.032839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Freezing Layers\n\nfor layer in base_model.layers:\n    layer.trainable=False","metadata":{"papermill":{"duration":0.155408,"end_time":"2021-07-02T07:20:53.563171","exception":false,"start_time":"2021-07-02T07:20:53.407763","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-09T15:24:31.036342Z","iopub.execute_input":"2021-07-09T15:24:31.036725Z","iopub.status.idle":"2021-07-09T15:24:31.042221Z","shell.execute_reply.started":"2021-07-09T15:24:31.036687Z","shell.execute_reply":"2021-07-09T15:24:31.041196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# model","metadata":{"papermill":{"duration":0.129313,"end_time":"2021-07-02T07:20:53.820442","exception":false,"start_time":"2021-07-02T07:20:53.691129","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Building Model\n\nmodel=Sequential()\nmodel.add(base_model)\nmodel.add(Flatten())\nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dropout(0.25))\nmodel.add(Dense(64, activation='relu'))\nmodel.add(Dropout(0.25))\nmodel.add(Dense(4, activation='sigmoid'))","metadata":{"papermill":{"duration":1.559015,"end_time":"2021-07-02T07:20:55.508907","exception":false,"start_time":"2021-07-02T07:20:53.949892","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-09T15:24:31.043897Z","iopub.execute_input":"2021-07-09T15:24:31.044405Z","iopub.status.idle":"2021-07-09T15:24:31.144244Z","shell.execute_reply.started":"2021-07-09T15:24:31.044368Z","shell.execute_reply":"2021-07-09T15:24:31.143418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Summary\n\nmodel.summary()","metadata":{"papermill":{"duration":0.176682,"end_time":"2021-07-02T07:20:55.815461","exception":false,"start_time":"2021-07-02T07:20:55.638779","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-09T15:24:31.145418Z","iopub.execute_input":"2021-07-09T15:24:31.145828Z","iopub.status.idle":"2021-07-09T15:24:31.154515Z","shell.execute_reply.started":"2021-07-09T15:24:31.145792Z","shell.execute_reply":"2021-07-09T15:24:31.153689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.utils import plot_model\nfrom IPython.display import SVG, Image\nplot_model(model, to_file='model.png', show_shapes=True, show_layer_names=True)\nImage('model.png',width=400, height=200)","metadata":{"papermill":{"duration":0.646628,"end_time":"2021-07-02T07:20:56.594822","exception":false,"start_time":"2021-07-02T07:20:55.948194","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-09T15:24:31.156145Z","iopub.execute_input":"2021-07-09T15:24:31.156748Z","iopub.status.idle":"2021-07-09T15:24:31.599986Z","shell.execute_reply.started":"2021-07-09T15:24:31.156712Z","shell.execute_reply":"2021-07-09T15:24:31.598208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model Compile \n\nOPT    = tensorflow.keras.optimizers.Adam(lr=0.001)\n\nmodel.compile(loss='categorical_crossentropy',\n              metrics=[tensorflow.keras.metrics.AUC(name = 'auc'),Precision(),Recall()],\n              optimizer=OPT)","metadata":{"papermill":{"duration":0.198123,"end_time":"2021-07-02T07:20:57.012522","exception":false,"start_time":"2021-07-02T07:20:56.814399","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-09T15:24:31.602933Z","iopub.execute_input":"2021-07-09T15:24:31.603296Z","iopub.status.idle":"2021-07-09T15:24:31.635587Z","shell.execute_reply.started":"2021-07-09T15:24:31.603266Z","shell.execute_reply":"2021-07-09T15:24:31.634791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining Callbacks\n\nfilepath = './best_weights.hdf5'\n\nearlystopping = EarlyStopping(monitor = 'val_auc', \n                              mode = 'max' , \n                              patience = 15,\n                              verbose = 1)\n\ncheckpoint    = ModelCheckpoint(filepath, \n                                monitor = 'val_auc', \n                                mode='max', \n                                save_best_only=True, \n                                verbose = 1)\n\n\ncallback_list = [earlystopping, checkpoint]","metadata":{"papermill":{"duration":0.140501,"end_time":"2021-07-02T07:20:57.284448","exception":false,"start_time":"2021-07-02T07:20:57.143947","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-09T15:24:31.636821Z","iopub.execute_input":"2021-07-09T15:24:31.637183Z","iopub.status.idle":"2021-07-09T15:24:31.642518Z","shell.execute_reply.started":"2021-07-09T15:24:31.637149Z","shell.execute_reply":"2021-07-09T15:24:31.641584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_history=model.fit(train_generator,\n                        validation_data=valid_generator,\n                        epochs = 25,\n                        callbacks = callback_list,\n                        verbose = 1)","metadata":{"papermill":{"duration":1513.725622,"end_time":"2021-07-02T07:46:11.141247","exception":false,"start_time":"2021-07-02T07:20:57.415625","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-09T15:24:31.643872Z","iopub.execute_input":"2021-07-09T15:24:31.644224Z","iopub.status.idle":"2021-07-09T15:27:00.803711Z","shell.execute_reply.started":"2021-07-09T15:24:31.644189Z","shell.execute_reply":"2021-07-09T15:27:00.802859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import keras\nmodel.save('./best_weights.hdf5')\n#model = keras.models.load_model('./best_weights.hdf5')","metadata":{"papermill":{"duration":2.317546,"end_time":"2021-07-02T07:46:14.217064","exception":false,"start_time":"2021-07-02T07:46:11.899518","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-09T15:27:00.806939Z","iopub.execute_input":"2021-07-09T15:27:00.807238Z","iopub.status.idle":"2021-07-09T15:27:01.074542Z","shell.execute_reply.started":"2021-07-09T15:27:00.807210Z","shell.execute_reply":"2021-07-09T15:27:01.073521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(model_history.history['auc'])\nplt.plot(model_history.history['val_auc'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='lower right')\nplt.show()\n\nplt.plot(model_history.history['loss'])\nplt.plot(model_history.history['val_loss'])\nplt.title('train set loss')\n\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper right')\nplt.show()\n\nplt.plot(model_history.history['precision'])\nplt.plot(model_history.history['val_precision'])\nplt.title(' precision')\nplt.ylabel('precision')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper right')\nplt.show()\n\n\nplt.plot(model_history.history['recall'])\nplt.plot(model_history.history['val_recall'])\nplt.title(' recall')\nplt.ylabel('recall')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper right')\nplt.show()\n","metadata":{"papermill":{"duration":1.316042,"end_time":"2021-07-02T07:46:16.533726","exception":false,"start_time":"2021-07-02T07:46:15.217684","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-09T15:27:01.077312Z","iopub.execute_input":"2021-07-09T15:27:01.077690Z","iopub.status.idle":"2021-07-09T15:27:01.648108Z","shell.execute_reply.started":"2021-07-09T15:27:01.077617Z","shell.execute_reply":"2021-07-09T15:27:01.647130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss, acc , precision,recall=model.evaluate(valid_generator)\nprint('Test Accuracy: %.3f' % acc)\nprint('Test Precision: %.3f' % precision)\nprint('Test Recall: %.3f' % recall)\nprint('Test loss: %.3f' % loss)\n","metadata":{"papermill":{"duration":10.677481,"end_time":"2021-07-02T07:46:28.035352","exception":false,"start_time":"2021-07-02T07:46:17.357871","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-09T15:27:01.649442Z","iopub.execute_input":"2021-07-09T15:27:01.649802Z","iopub.status.idle":"2021-07-09T15:27:07.075527Z","shell.execute_reply.started":"2021-07-09T15:27:01.649763Z","shell.execute_reply":"2021-07-09T15:27:07.074716Z"},"trusted":true},"execution_count":null,"outputs":[]}]}