{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1><center>APTOS 2019 Blindness Detection</center></h1>\n<h2><center>Detect diabetic retinopathy to stop blindness before it's too late</center></h2>\n<center><img src=\"https://raw.githubusercontent.com/dimitreOliveira/MachineLearning/master/Kaggle/APTOS%202019%20Blindness%20Detection/aux_img.png\"></center>\n\nIn this synchronous Kernels-only competition, you'll build a machine learning model to speed up disease detection. You’ll work with thousands of images collected in rural areas to help identify diabetic retinopathy automatically. If successful, you will not only help to prevent lifelong blindness, but these models may be used to detect other sorts of diseases in the future, like glaucoma and macular degeneration.\n\nIn this notebook, I will be using basic deep learning and transfer learning (ResNet50) to create a baseline.\n##### Image source: http://cceyemd.com/diabetes-and-eye-exams/\n\n### Dependencies","metadata":{}},{"cell_type":"code","source":"# !pip install tensorflow==1.15\n# # !pip install tensorflow-gpu==1.15\n","metadata":{"execution":{"iopub.status.busy":"2022-04-17T18:58:41.115475Z","iopub.execute_input":"2022-04-17T18:58:41.115821Z","iopub.status.idle":"2022-04-17T18:58:41.121893Z","shell.execute_reply.started":"2022-04-17T18:58:41.115759Z","shell.execute_reply":"2022-04-17T18:58:41.121028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nprint(tf.__version__)","metadata":{"execution":{"iopub.status.busy":"2022-04-17T18:58:41.131863Z","iopub.execute_input":"2022-04-17T18:58:41.132117Z","iopub.status.idle":"2022-04-17T18:58:42.04974Z","shell.execute_reply.started":"2022-04-17T18:58:41.13207Z","shell.execute_reply":"2022-04-17T18:58:42.048879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport random\nimport warnings\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import confusion_matrix, cohen_kappa_score\nfrom keras.models import Model\nfrom keras import optimizers, applications\nfrom tensorflow.python.keras.preprocessing.image import load_img, img_to_array\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.callbacks import EarlyStopping, ReduceLROnPlateau\nfrom keras.layers import Dense, Dropout, GlobalAveragePooling2D, Input\nfrom IPython.display import clear_output\n\n# Set seeds to make the experiment more reproducible.\nfrom tensorflow import set_random_seed\ndef seed_everything(seed=0):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    set_random_seed(0)\nseed_everything()\n%matplotlib inline\nsns.set(style=\"whitegrid\")\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-04-17T18:58:42.053183Z","iopub.execute_input":"2022-04-17T18:58:42.053416Z","iopub.status.idle":"2022-04-17T18:58:43.128309Z","shell.execute_reply.started":"2022-04-17T18:58:42.053372Z","shell.execute_reply":"2022-04-17T18:58:43.127313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load data","metadata":{"_kg_hide-output":true}},{"cell_type":"code","source":"train = pd.read_csv('../input/aptos2019-blindness-detection/train.csv')\n# test = pd.read_csv('../input/diabetic-retinopathy-resized/trainLabels.csv')","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-04-17T18:58:43.132858Z","iopub.execute_input":"2022-04-17T18:58:43.135069Z","iopub.status.idle":"2022-04-17T18:58:43.15761Z","shell.execute_reply.started":"2022-04-17T18:58:43.135019Z","shell.execute_reply":"2022-04-17T18:58:43.15682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA\n\n## Data overview","metadata":{}},{"cell_type":"code","source":"print('Number of train samples: ', train.shape[0])\n# print('Number of test samples: ', test.shape[0])\ndisplay(train.head())","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-17T18:58:43.161969Z","iopub.execute_input":"2022-04-17T18:58:43.164148Z","iopub.status.idle":"2022-04-17T18:58:43.198086Z","shell.execute_reply.started":"2022-04-17T18:58:43.164098Z","shell.execute_reply":"2022-04-17T18:58:43.197127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain_original, batchtest = train_test_split(train, test_size=0.3,random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-04-17T18:58:43.204232Z","iopub.execute_input":"2022-04-17T18:58:43.204653Z","iopub.status.idle":"2022-04-17T18:58:43.233808Z","shell.execute_reply.started":"2022-04-17T18:58:43.204469Z","shell.execute_reply":"2022-04-17T18:58:43.23299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Label class distribution\n\nAs we can see we have an unbalanced database, we have two times more class 0 than 2, and classes 1, 2 and 4 each have less than half of the class 2 data.","metadata":{}},{"cell_type":"code","source":"f, ax = plt.subplots(figsize=(14, 8.7))\nax = sns.countplot(x=\"diagnosis\", data=train_original, palette=\"GnBu_d\")\nsns.despine()\nplt.show()\n\nx = pd.crosstab(index=train_original['diagnosis'],columns='count')\nprint(x)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-17T18:58:43.239605Z","iopub.execute_input":"2022-04-17T18:58:43.241706Z","iopub.status.idle":"2022-04-17T18:58:43.644046Z","shell.execute_reply.started":"2022-04-17T18:58:43.241656Z","shell.execute_reply":"2022-04-17T18:58:43.642939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(figsize=(14, 8.7))\nax = sns.countplot(x=\"diagnosis\", data=batchtest, palette=\"GnBu_d\")\nsns.despine()\nplt.show()\n\nx = pd.crosstab(index=batchtest['diagnosis'],columns='count')\nprint(x)","metadata":{"execution":{"iopub.status.busy":"2022-04-17T18:58:43.645551Z","iopub.execute_input":"2022-04-17T18:58:43.645869Z","iopub.status.idle":"2022-04-17T18:58:44.217096Z","shell.execute_reply.started":"2022-04-17T18:58:43.645813Z","shell.execute_reply":"2022-04-17T18:58:44.214865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Legend\n- 0 - No DR\n- 1 - Mild\n- 2 - Moderate\n- 3 - Severe\n- 4 - Proliferative DR ","metadata":{}},{"cell_type":"markdown","source":"# Model parameters","metadata":{}},{"cell_type":"code","source":"# Model parameters\nBATCH_SIZE = 8\nEPOCHS = 20\nWARMUP_EPOCHS = 2\nLEARNING_RATE = 1e-4\nWARMUP_LEARNING_RATE = 1e-3\nHEIGHT = 512\nWIDTH = 512\nCANAL = 3\nN_CLASSES = train_original['diagnosis'].nunique()\nES_PATIENCE = 5\nRLROP_PATIENCE = 3\nDECAY_DROP = 0.5","metadata":{"execution":{"iopub.status.busy":"2022-04-17T18:58:44.218579Z","iopub.execute_input":"2022-04-17T18:58:44.218881Z","iopub.status.idle":"2022-04-17T18:58:44.228908Z","shell.execute_reply.started":"2022-04-17T18:58:44.218825Z","shell.execute_reply":"2022-04-17T18:58:44.227826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocecss data\n\ntrain_original[\"id_code\"] = train_original[\"id_code\"].apply(lambda x: x + \".png\")\n# test[\"image\"] = test[\"image\"].apply(lambda x: x + \".jpeg\")\ntrain_original['diagnosis'] = train_original['diagnosis'].astype('str')\nbatchtest[\"id_code\"] = batchtest[\"id_code\"].apply(lambda x: x + \".png\")\nbatchtest['diagnosis'] = batchtest['diagnosis'].astype('str')\ntrain_original.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-17T18:58:44.231092Z","iopub.execute_input":"2022-04-17T18:58:44.231658Z","iopub.status.idle":"2022-04-17T18:58:44.555108Z","shell.execute_reply.started":"2022-04-17T18:58:44.231378Z","shell.execute_reply":"2022-04-17T18:58:44.554431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batchtest.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-17T18:58:44.556617Z","iopub.execute_input":"2022-04-17T18:58:44.556905Z","iopub.status.idle":"2022-04-17T18:58:44.570468Z","shell.execute_reply.started":"2022-04-17T18:58:44.556859Z","shell.execute_reply":"2022-04-17T18:58:44.569398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir /kaggle/working/images","metadata":{"execution":{"iopub.status.busy":"2022-04-17T18:58:44.572244Z","iopub.execute_input":"2022-04-17T18:58:44.572822Z","iopub.status.idle":"2022-04-17T18:58:45.247356Z","shell.execute_reply.started":"2022-04-17T18:58:44.572751Z","shell.execute_reply":"2022-04-17T18:58:45.246429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import os\n# import matplotlib.pyplot as plt\n# import shutil\n# from distutils.dir_util import copy_tree\n\n# files = os.listdir(\"../input/aptos2019-blindness-detection/train_images/\")\n# # print(files)\n# fromDirectory = '../input/aptos2019-blindness-detection/train_images'\n# toDirectory = '../working/images/'\n\n# copy_tree(fromDirectory, toDirectory)\n# clear_output(wait=True)","metadata":{"execution":{"iopub.status.busy":"2022-04-17T18:58:45.249962Z","iopub.execute_input":"2022-04-17T18:58:45.250278Z","iopub.status.idle":"2022-04-17T18:58:45.253946Z","shell.execute_reply.started":"2022-04-17T18:58:45.250229Z","shell.execute_reply":"2022-04-17T18:58:45.253129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for file_name in file_names:\n#     shutil.move(source_dir, target_dir)","metadata":{"execution":{"iopub.status.busy":"2022-04-17T18:58:45.255222Z","iopub.execute_input":"2022-04-17T18:58:45.255697Z","iopub.status.idle":"2022-04-17T18:58:45.264054Z","shell.execute_reply.started":"2022-04-17T18:58:45.255644Z","shell.execute_reply":"2022-04-17T18:58:45.263228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # lets define a ImageDataGenerator object\n# # change the arguments below as per the requirment\n# idg = ImageDataGenerator(horizontal_flip = True,\n#                          vertical_flip = True\n# #                          brightness_range = [1.0, 1.5]\n#                          )\n# # lets define a ImageDataGenerator object\n# # change the arguments below as per the requirment\n# idg1 = ImageDataGenerator(horizontal_flip = True,\n# #                          vertical_flip = True\n#                          brightness_range = [1.0, 1.5]\n#                          )","metadata":{"execution":{"iopub.status.busy":"2022-04-17T18:58:45.265565Z","iopub.execute_input":"2022-04-17T18:58:45.266073Z","iopub.status.idle":"2022-04-17T18:58:45.273397Z","shell.execute_reply.started":"2022-04-17T18:58:45.265822Z","shell.execute_reply":"2022-04-17T18:58:45.272277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train1 = train_original.loc[train_original['diagnosis'] == '1']\n# train1.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-17T18:58:45.276655Z","iopub.execute_input":"2022-04-17T18:58:45.27694Z","iopub.status.idle":"2022-04-17T18:58:45.282076Z","shell.execute_reply.started":"2022-04-17T18:58:45.27688Z","shell.execute_reply":"2022-04-17T18:58:45.281211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# # sample code to check if our agumentation is working for a single image\n# # lets read our image to be processed - change the directory as needed\n\n\n# y = train1['id_code']\n# # for img in y:\n# #     print (y)\n# images_1 =[]\n# line_count = 0\n# for line in y:\n#     images_1.append(line)\n#     line_count += 1\n# # print(images_3)\n\n\n# for img in images_1:\n#     img = load_img(\"../input/aptos2019-blindness-detection/train_images/\" + img)\n#     input_arr = img_to_array(img)\n#     input_arr = input_arr.reshape((1,) + input_arr.shape) \n#     i = 0\n# # keras flow function usually work for batches\n# # chnage the directory and number of iterations as required\n#     for batch in idg1.flow(input_arr, batch_size=1,\n#                               save_to_dir='../working/images/', save_prefix='one', save_format='png'):\n#         i += 1\n#         if i > 2:\n#             break  # need to break the loop otherwise it will run infinite times\n\n\n# a1 = pd.DataFrame(columns = [\"id_code\", \"diagnosis\"])\n# for image in os.listdir(\"../working/images/\"):\n#     if image.startswith(\"one\"):\n#         df = pd.DataFrame([[image, '1']], columns = [\"id_code\", \"diagnosis\"])\n#         a1 = a1.append(df, ignore_index = True)\n        \n# a1.head()\n# train_original = pd.concat([train_original,a1])\n# len(train_original.index)","metadata":{"execution":{"iopub.status.busy":"2022-04-17T18:58:45.283936Z","iopub.execute_input":"2022-04-17T18:58:45.284498Z","iopub.status.idle":"2022-04-17T18:58:45.290906Z","shell.execute_reply.started":"2022-04-17T18:58:45.284197Z","shell.execute_reply":"2022-04-17T18:58:45.29004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # sample code to check if our agumentation is working for a single image\n# # lets read our image to be processed - change the directory as needed\n# train3 = train_original.loc[train_original['diagnosis'] == '3']\n# train3.head()\n\n# y = train3['id_code']\n# # for img in y:\n# #     print (y)\n# images_3 =[]\n# line_count = 0\n# for line in y:\n#     images_3.append(line)\n#     line_count += 1\n# # print(images_3)\n\n\n# for img in images_3:\n#     img = load_img(\"../input/aptos2019-blindness-detection/train_images/\" + img)\n#     input_arr = img_to_array(img)\n#     input_arr = input_arr.reshape((1,) + input_arr.shape) \n#     i = 0\n# # keras flow function usually work for batches\n# # chnage the directory and number of iterations as required\n#     for batch in idg.flow(input_arr, batch_size=1,\n#                               save_to_dir='../working/images/', save_prefix='ger', save_format='png'):\n#         i += 1\n#         if i > 5:\n#             break  # need to break the loop otherwise it will run infinite timesth\n\n\n# a3 = pd.DataFrame(columns = [\"id_code\", \"diagnosis\"])\n# for image in os.listdir(\"../working/images/\"):\n#     if image.startswith(\"ger\"):\n#         df = pd.DataFrame([[image, '3']], columns = [\"id_code\", \"diagnosis\"])\n#         a3 = a3.append(df, ignore_index = True)\n        \n# a3.head()\n# train_original = pd.concat([train_original,a3])\n# len(train_original.index)","metadata":{"execution":{"iopub.status.busy":"2022-04-17T18:58:45.294045Z","iopub.execute_input":"2022-04-17T18:58:45.294636Z","iopub.status.idle":"2022-04-17T18:58:45.314257Z","shell.execute_reply.started":"2022-04-17T18:58:45.294557Z","shell.execute_reply":"2022-04-17T18:58:45.313226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # sample code to check if our agumentation is working for a single image\n# # lets read our image to be processed - change the directory as needed\n# train4 = train_original.loc[train_original['diagnosis'] == '4']\n# train4.head()\n\n# y = train4['id_code']\n# # for img in y:\n# #     print (y)\n# images_4 =[]\n# line_count = 0\n# for line in y:\n#     images_4.append(line)\n#     line_count += 1\n# # print(images_4)\n\n\n# for img in images_4:\n#     img = load_img(\"../input/aptos2019-blindness-detection/train_images/\" + img)\n#     input_arr = img_to_array(img)\n#     input_arr = input_arr.reshape((1,) + input_arr.shape) \n#     i = 0\n# # keras flow function usually work for batches\n# # chnage the directory and number of iterations as required\n#     for batch in idg1.flow(input_arr, batch_size=1,\n#                               save_to_dir='../working/images/', save_prefix='new', save_format='png'):\n#         i += 1\n#         if i > 4:\n#             break  # need to break the loop otherwise it will run infinite times\n\n# a4 = pd.DataFrame(columns = [\"id_code\", \"diagnosis\"])\n# for image in os.listdir(\"../working/images/\"):\n#     if image.startswith(\"new\"):\n#         df = pd.DataFrame([[image, '4']], columns = [\"id_code\", \"diagnosis\"])\n#         a4 = a4.append(df, ignore_index = True)\n        \n# a4.head()\n# train_original = pd.concat([train_original,a4])\n# len(train_original.index)","metadata":{"execution":{"iopub.status.busy":"2022-04-17T18:58:45.315845Z","iopub.execute_input":"2022-04-17T18:58:45.316434Z","iopub.status.idle":"2022-04-17T18:58:45.328803Z","shell.execute_reply.started":"2022-04-17T18:58:45.316376Z","shell.execute_reply":"2022-04-17T18:58:45.327926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# f, ax = plt.subplots(figsize=(14, 8.7))\n# ax = sns.countplot(x=\"diagnosis\", data=train_original, palette=\"GnBu_d\")\n# sns.despine()\n# plt.show()\n\n# x = pd.crosstab(index=train_original['diagnosis'],columns='count')\n# print(x)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-17T18:58:45.330644Z","iopub.execute_input":"2022-04-17T18:58:45.331091Z","iopub.status.idle":"2022-04-17T18:58:45.345807Z","shell.execute_reply.started":"2022-04-17T18:58:45.330919Z","shell.execute_reply":"2022-04-17T18:58:45.344565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_original.to_csv(r'./train_original.csv', index = False, header=True)","metadata":{"execution":{"iopub.status.busy":"2022-04-17T18:58:45.349111Z","iopub.execute_input":"2022-04-17T18:58:45.34963Z","iopub.status.idle":"2022-04-17T18:58:45.358536Z","shell.execute_reply.started":"2022-04-17T18:58:45.349577Z","shell.execute_reply":"2022-04-17T18:58:45.35508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data generator","metadata":{}},{"cell_type":"code","source":"train_datagen=ImageDataGenerator(rescale=1./255, \n                                 validation_split=0.2)\n\ntrain_generator=train_datagen.flow_from_dataframe(\n    dataframe=train_original,\n    directory=\"../input/aptos2019-blindness-detection/train_images\",\n    x_col=\"id_code\",\n    y_col=\"diagnosis\",\n    batch_size=BATCH_SIZE,\n    class_mode=\"categorical\",\n    target_size=(HEIGHT, WIDTH),\n    subset='training')\n\nvalid_generator=train_datagen.flow_from_dataframe(\n    dataframe=train_original,\n    directory=\"../input/aptos2019-blindness-detection/train_images\",\n    x_col=\"id_code\",\n    y_col=\"diagnosis\",\n    batch_size=BATCH_SIZE,\n    class_mode=\"categorical\",    \n    target_size=(HEIGHT, WIDTH),\n    subset='validation')\n\ntest_datagen = ImageDataGenerator(rescale=1./255)\n# batchtest\ntest_generator1 = test_datagen.flow_from_dataframe(  \n        dataframe=batchtest,\n        directory = \"../input/aptos2019-blindness-detection/train_images\",\n        x_col=\"id_code\",\n        target_size=(HEIGHT, WIDTH),\n        batch_size=1,\n        shuffle=False,\n        class_mode=None)\n\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-17T18:58:45.36218Z","iopub.execute_input":"2022-04-17T18:58:45.363018Z","iopub.status.idle":"2022-04-17T18:58:48.734466Z","shell.execute_reply.started":"2022-04-17T18:58:45.362907Z","shell.execute_reply":"2022-04-17T18:58:48.733638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"def create_model(input_shape, n_out):\n    input_tensor = Input(shape=input_shape)\n    base_model = applications.ResNet50(weights=None, \n                                       include_top=False,\n                                       input_tensor=input_tensor)\n#     base_model.load_weights('../input/resnet50/resnet50_weights_tf_dim_ordering_tf_kernels_notop.h5')\n\n    x = GlobalAveragePooling2D()(base_model.output)\n    x = Dropout(0.5)(x)\n    x = Dense(2048, activation='relu')(x)\n    x = Dropout(0.5)(x)\n    final_output = Dense(n_out, activation='softmax', name='final_output')(x)\n    model = Model(input_tensor, final_output)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-04-17T18:58:48.735798Z","iopub.execute_input":"2022-04-17T18:58:48.736065Z","iopub.status.idle":"2022-04-17T18:58:48.744496Z","shell.execute_reply.started":"2022-04-17T18:58:48.736021Z","shell.execute_reply":"2022-04-17T18:58:48.743716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = create_model(input_shape=(HEIGHT, WIDTH, CANAL), n_out=N_CLASSES)\n\n# for layer in model.layers:\n#     layer.trainable = False\n\n# for i in range(-5, 0):\n#     model.layers[i].trainable = True\n\nmetric_list = [\"accuracy\"]\noptimizer = optimizers.Adam(lr=WARMUP_LEARNING_RATE)\nmodel.compile(optimizer=optimizer, loss=\"categorical_crossentropy\",  metrics=metric_list)\n# model.summary()\nclear_output(wait=True)\n# from keras.utils import plot_model\n# graph = plot_model(model,to_file='model.png', show_shapes=True)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-04-17T18:58:48.746429Z","iopub.execute_input":"2022-04-17T18:58:48.747039Z","iopub.status.idle":"2022-04-17T18:58:55.969544Z","shell.execute_reply.started":"2022-04-17T18:58:48.746972Z","shell.execute_reply":"2022-04-17T18:58:55.968847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train top layers","metadata":{}},{"cell_type":"code","source":"STEP_SIZE_TRAIN = train_generator.n//train_generator.batch_size\nSTEP_SIZE_VALID = valid_generator.n//valid_generator.batch_size\n\nhistory_warmup = model.fit_generator(generator=train_generator,\n                                     steps_per_epoch=STEP_SIZE_TRAIN,\n                                     validation_data=valid_generator,\n                                     validation_steps=STEP_SIZE_VALID,\n                                     epochs=20,\n                                     verbose=1).history","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-04-17T18:58:55.970997Z","iopub.execute_input":"2022-04-17T18:58:55.971298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fine-tune the complete model","metadata":{}},{"cell_type":"code","source":"# from keras.metrics import categorical_accuracy\n\n# for layer in model.layers:\n#     layer.trainable = True\n\n# es = EarlyStopping(monitor='val_loss', mode='min', patience=ES_PATIENCE, restore_best_weights=True, verbose=1)\n# rlrop = ReduceLROnPlateau(monitor='val_loss', mode='min', patience=RLROP_PATIENCE, factor=DECAY_DROP, min_lr=1e-6, verbose=1)\n\n# callback_list = [es, rlrop]\n# optimizer = optimizers.Adam(lr=LEARNING_RATE)\n# model.compile(optimizer=optimizer, loss=\"binary_crossentropy\",  metrics=metric_list)\n# # model.summary()\n# clear_output(wait=True)\n# # graph = plot_model(model,to_file='cnn_model.png', show_shapes=True)","metadata":{"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# history_finetunning = model.fit_generator(generator=train_generator,\n#                                           steps_per_epoch=STEP_SIZE_TRAIN,\n#                                           validation_data=valid_generator,\n#                                           validation_steps=STEP_SIZE_VALID,\n#                                           epochs=EPOCHS,\n#                                           callbacks=callback_list,\n#                                           verbose=1).history","metadata":{"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('../working/ResNet50_aug.h5')\nprint('Model Saved')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model loss graph ","metadata":{}},{"cell_type":"code","source":"# history = {'loss': history_warmup['loss'] + history_finetunning['loss'], \n#            'val_loss': history_warmup['val_loss'] + history_finetunning['val_loss'], \n#            'acc': history_warmup['acc'] + history_finetunning['acc'], \n#            'val_acc': history_warmup['val_acc'] + history_finetunning['val_acc']}\n\n# sns.set_style(\"whitegrid\")\n# fig, (ax1, ax2) = plt.subplots(2, 1, sharex='col', figsize=(20, 14))\n\n# ax1.plot(history['loss'], label='Train loss')\n# ax1.plot(history['val_loss'], label='Validation loss')\n# ax1.legend(loc='best')\n# ax1.set_title('Loss')\n\n# ax2.plot(history['acc'], label='Train Accuracy')\n# ax2.plot(history['val_acc'], label='Validation accuracy')\n# ax2.legend(loc='best')\n# ax2.set_title('Accuracy')\n\n# plt.xlabel('Epochs')\n# sns.despine()\n# plt.show()","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Evaluation","metadata":{}},{"cell_type":"code","source":"complete_datagen = ImageDataGenerator(rescale=1./255)\ncomplete_generator = complete_datagen.flow_from_dataframe(  \n        dataframe=train_original,\n        directory = \"../input/aptos2019-blindness-detection/train_images\",\n        x_col=\"id_code\",\n        target_size=(HEIGHT, WIDTH),\n        batch_size=1,\n        shuffle=False,\n        class_mode=None)\n\n\nSTEP_SIZE_COMPLETE = complete_generator.n//complete_generator.batch_size\ntrain_preds = model.predict_generator(complete_generator, steps=STEP_SIZE_COMPLETE)\ntrain_preds = [np.argmax(pred) for pred in train_preds]","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Confusion Matrix","metadata":{}},{"cell_type":"code","source":"labels = ['0 - No DR', '1 - Mild', '2 - Moderate', '3 - Severe', '4 - Proliferative DR']\ncnf_matrix = confusion_matrix(train_original['diagnosis'].astype('int'), train_preds)\ncnf_matrix_norm = cnf_matrix.astype('float') / cnf_matrix.sum(axis=1)[:, np.newaxis]\ndf_cm = pd.DataFrame(cnf_matrix_norm, index=labels, columns=labels)\nplt.figure(figsize=(16, 7))\nsns.heatmap(df_cm, annot=True, fmt='.2f', cmap=\"Blues\")\nplt.show()","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Quadratic Weighted Kappa","metadata":{}},{"cell_type":"code","source":"print(\"Train Cohen Kappa score: %.3f\" % cohen_kappa_score(train_preds, train_original['diagnosis'].astype('int'), weights='quadratic'))","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Apply model to test set and output predictions","metadata":{}},{"cell_type":"code","source":"test_generator1.reset()\nSTEP_SIZE_TEST1 = test_generator1.n//test_generator1.batch_size\npreds = model.predict_generator(test_generator1, steps=STEP_SIZE_TEST1)\npredictions = [np.argmax(pred) for pred in preds]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filenames = test_generator1.filenames\nresults1 = pd.DataFrame({'id_code':filenames, 'diag':predictions})\nresults1['id_code'] = results1['id_code'].map(lambda x: str(x)[:-4])\nresults1.to_csv('submission1.csv',index=False)\nresults1.head(10)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data1 = pd.read_csv('../working/submission1.csv', low_memory = False)\ndata1[\"id_code\"] = data1[\"id_code\"].apply(lambda x: x + \".png\")\noutput1 = pd.merge(batchtest, data1, \n                   on='id_code', \n                   how='outer')\noutput1.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\noutput1[\"diagnosis\"] = output1[\"diagnosis\"].astype('int')\ny_true =  output1[\"diagnosis\"]\noutput1['diag'] = output1['diag'].astype('int')\ny_pred = output1['diag']\nlabels = [0,1,2,3,4]\nconfusion_matrix(y_true.values, y_pred.values, labels=labels)\n\n# print(y_pred.values)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nfrom sklearn.metrics import accuracy_score\n\ntarget_names = ['class 0', 'class 1', 'class 2', 'class 3', 'class 4']\nprint(classification_report(y_true.values, y_pred.values, target_names=target_names, digits=4))\naccuracy_score(y_true.values, y_pred.values)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = ['0 - No DR', '1 - Mild', '2 - Moderate', '3 - Severe', '4 - Proliferative DR']\ncnf_matrix = confusion_matrix(y_true.values, y_pred.values)\ncnf_matrix_norm = cnf_matrix.astype('float') / cnf_matrix.sum(axis=1)[:, np.newaxis]\ndf_cm = pd.DataFrame(cnf_matrix_norm, index=labels, columns=labels)\nplt.figure(figsize=(16, 7))\nsns.heatmap(df_cm, annot=True, fmt='.2f', cmap=\"Blues\")\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}