{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"},{"sourceId":556303,"sourceType":"datasetVersion","datasetId":266957},{"sourceId":556726,"sourceType":"datasetVersion","datasetId":267272},{"sourceId":8272012,"sourceType":"datasetVersion","datasetId":4911444},{"sourceId":8277101,"sourceType":"datasetVersion","datasetId":4915051}],"dockerImageVersionId":29188,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport sys\nimport cv2\nimport shutil\nimport random\nimport warnings\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom tensorflow import set_random_seed\nfrom sklearn.utils import class_weight\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, cohen_kappa_score\nfrom keras import backend as K\nfrom keras.models import Model\nfrom keras.utils import to_categorical\nfrom keras import optimizers, applications\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.layers import Dense, Dropout, GlobalAveragePooling2D, Input\nfrom keras.callbacks import EarlyStopping, ReduceLROnPlateau, Callback, LearningRateScheduler\n\ndef seed_everything(seed=0):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    set_random_seed(0)\n\nseed = 0\nseed_everything(seed)\n%matplotlib inline\nsns.set(style=\"whitegrid\")\nwarnings.filterwarnings(\"ignore\")\nsys.path.append(os.path.abspath('../input/efficientnet/efficientnet-master/efficientnet-master/'))\nfrom efficientnet import *","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-04-30T12:12:52.838625Z","iopub.execute_input":"2024-04-30T12:12:52.838916Z","iopub.status.idle":"2024-04-30T12:12:57.952790Z","shell.execute_reply.started":"2024-04-30T12:12:52.838872Z","shell.execute_reply":"2024-04-30T12:12:57.951905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fold_set = pd.read_csv('/kaggle/input/5-fold/5-fold.csv')\nX_train = fold_set[fold_set['fold_0'] == 'train']\nX_val = fold_set[fold_set['fold_0'] == 'validation']\ntest = pd.read_csv('../input/aptos2019-blindness-detection/test.csv')\nprint('Number of train samples: ', X_train.shape[0])\nprint('Number of validation samples: ', X_val.shape[0])\nprint('Number of test samples: ', test.shape[0])\n\n# Preprocecss data\nX_train[\"id_code\"] = X_train[\"id_code\"].apply(lambda x: x + \".png\")\nX_val[\"id_code\"] = X_val[\"id_code\"].apply(lambda x: x + \".png\")\ntest[\"id_code\"] = test[\"id_code\"].apply(lambda x: x + \".png\")\ndisplay(X_train.head())","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-04-30T12:13:03.296476Z","iopub.execute_input":"2024-04-30T12:13:03.296812Z","iopub.status.idle":"2024-04-30T12:13:03.585765Z","shell.execute_reply.started":"2024-04-30T12:13:03.296754Z","shell.execute_reply":"2024-04-30T12:13:03.584992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model parameters\nFACTOR = 4\nBATCH_SIZE = 8 * FACTOR\nEPOCHS = 20\nWARMUP_EPOCHS = 5\nLEARNING_RATE = 1e-4 * FACTOR\nWARMUP_LEARNING_RATE = 1e-3 * FACTOR\nHEIGHT = 224\nWIDTH = 224\nCHANNELS = 3\nES_PATIENCE = 5\nRLROP_PATIENCE = 3\nDECAY_DROP = 0.5\nLR_WARMUP_EPOCHS_1st = 2\nLR_WARMUP_EPOCHS_2nd = 5\nSTEP_SIZE = len(X_train) // BATCH_SIZE\nTOTAL_STEPS_1st = WARMUP_EPOCHS * STEP_SIZE\nTOTAL_STEPS_2nd = EPOCHS * STEP_SIZE\nWARMUP_STEPS_1st = LR_WARMUP_EPOCHS_1st * STEP_SIZE\nWARMUP_STEPS_2nd = LR_WARMUP_EPOCHS_2nd * STEP_SIZE","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-04-30T12:13:19.078350Z","iopub.execute_input":"2024-04-30T12:13:19.078841Z","iopub.status.idle":"2024-04-30T12:13:19.086495Z","shell.execute_reply.started":"2024-04-30T12:13:19.078727Z","shell.execute_reply":"2024-04-30T12:13:19.085567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_base_path = '../input/aptos2019-blindness-detection/train_images/'\ntest_base_path = '../input/aptos2019-blindness-detection/test_images/'\ntrain_dest_path = 'base_dir/train_images/'\nvalidation_dest_path = 'base_dir/validation_images/'\ntest_dest_path =  'base_dir/test_images/'\n\n# Making sure directories don't exist\nif os.path.exists(train_dest_path):\n    shutil.rmtree(train_dest_path)\nif os.path.exists(validation_dest_path):\n    shutil.rmtree(validation_dest_path)\nif os.path.exists(test_dest_path):\n    shutil.rmtree(test_dest_path)\n    \n# Creating train, validation and test directories\nos.makedirs(train_dest_path)\nos.makedirs(validation_dest_path)\nos.makedirs(test_dest_path)\n\ndef crop_image(img, tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n            img = np.stack([img1,img2,img3],axis=-1)\n            \n        return img\n\ndef circle_crop(img):\n    img = crop_image(img)\n\n    height, width, depth = img.shape\n    largest_side = np.max((height, width))\n    img = cv2.resize(img, (largest_side, largest_side))\n\n    height, width, depth = img.shape\n\n    x = width//2\n    y = height//2\n    r = np.amin((x, y))\n\n    circle_img = np.zeros((height, width), np.uint8)\n    cv2.circle(circle_img, (x, y), int(r), 1, thickness=-1)\n    img = cv2.bitwise_and(img, img, mask=circle_img)\n    img = crop_image(img)\n\n    return img\n    \ndef preprocess_image(base_path, save_path, image_id, HEIGHT, WIDTH, sigmaX=10):\n    image = cv2.imread(base_path + image_id)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    image = circle_crop(image)\n    image = cv2.resize(image, (HEIGHT, WIDTH))\n    image = cv2.addWeighted(image, 4, cv2.GaussianBlur(image, (0,0), sigmaX), -4 , 128)\n    cv2.imwrite(save_path + image_id, image)\n    \n# Pre-procecss train set\nfor i, image_id in enumerate(X_train['id_code']):\n    preprocess_image(train_base_path, train_dest_path, image_id, HEIGHT, WIDTH)\n    \n# Pre-procecss validation set\nfor i, image_id in enumerate(X_val['id_code']):\n    preprocess_image(train_base_path, validation_dest_path, image_id, HEIGHT, WIDTH)\n    \n# Pre-procecss test set\nfor i, image_id in enumerate(test['id_code']):\n    preprocess_image(test_base_path, test_dest_path, image_id, HEIGHT, WIDTH)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T12:13:22.507311Z","iopub.execute_input":"2024-04-30T12:13:22.507669Z","iopub.status.idle":"2024-04-30T12:36:41.023153Z","shell.execute_reply.started":"2024-04-30T12:13:22.507609Z","shell.execute_reply":"2024-04-30T12:36:41.022135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datagen=ImageDataGenerator(rescale=1./255, \n                           rotation_range=360,\n                           horizontal_flip=True,\n                           vertical_flip=True)\n\ntrain_generator=datagen.flow_from_dataframe(\n                        dataframe=X_train,\n                        directory=train_dest_path,\n                        x_col=\"id_code\",\n                        y_col=\"diagnosis\",\n                        class_mode=\"raw\",\n                        batch_size=BATCH_SIZE,\n                        target_size=(HEIGHT, WIDTH),\n                        seed=seed)\n\nvalid_generator=datagen.flow_from_dataframe(\n                        dataframe=X_val,\n                        directory=validation_dest_path,\n                        x_col=\"id_code\",\n                        y_col=\"diagnosis\",\n                        class_mode=\"raw\",\n                        batch_size=BATCH_SIZE,\n                        target_size=(HEIGHT, WIDTH),\n                        seed=seed)\n\ntest_generator=datagen.flow_from_dataframe(  \n                       dataframe=test,\n                       directory=test_dest_path,\n                       x_col=\"id_code\",\n                       batch_size=1,\n                       class_mode=None,\n                       shuffle=False,\n                       target_size=(HEIGHT, WIDTH),\n                       seed=seed)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T12:44:47.047376Z","iopub.execute_input":"2024-04-30T12:44:47.047708Z","iopub.status.idle":"2024-04-30T12:44:47.127906Z","shell.execute_reply.started":"2024-04-30T12:44:47.047654Z","shell.execute_reply":"2024-04-30T12:44:47.127223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cosine_decay_with_warmup(global_step,\n                             learning_rate_base,\n                             total_steps,\n                             warmup_learning_rate=0.0,\n                             warmup_steps=0,\n                             hold_base_rate_steps=0):\n    \"\"\"\n    Cosine decay schedule with warm up period.\n    In this schedule, the learning rate grows linearly from warmup_learning_rate\n    to learning_rate_base for warmup_steps, then transitions to a cosine decay\n    schedule.\n    :param global_step {int}: global step.\n    :param learning_rate_base {float}: base learning rate.\n    :param total_steps {int}: total number of training steps.\n    :param warmup_learning_rate {float}: initial learning rate for warm up. (default: {0.0}).\n    :param warmup_steps {int}: number of warmup steps. (default: {0}).\n    :param hold_base_rate_steps {int}: Optional number of steps to hold base learning rate before decaying. (default: {0}).\n    :param global_step {int}: global step.\n    :Returns : a float representing learning rate.\n    :Raises ValueError: if warmup_learning_rate is larger than learning_rate_base, or if warmup_steps is larger than total_steps.\n    \"\"\"\n\n    if total_steps < warmup_steps:\n        raise ValueError('total_steps must be larger or equal to warmup_steps.')\n    learning_rate = 0.5 * learning_rate_base * (1 + np.cos(\n        np.pi *\n        (global_step - warmup_steps - hold_base_rate_steps\n         ) / float(total_steps - warmup_steps - hold_base_rate_steps)))\n    if hold_base_rate_steps > 0:\n        learning_rate = np.where(global_step > warmup_steps + hold_base_rate_steps,\n                                 learning_rate, learning_rate_base)\n    if warmup_steps > 0:\n        if learning_rate_base < warmup_learning_rate:\n            raise ValueError('learning_rate_base must be larger or equal to warmup_learning_rate.')\n        slope = (learning_rate_base - warmup_learning_rate) / warmup_steps\n        warmup_rate = slope * global_step + warmup_learning_rate\n        learning_rate = np.where(global_step < warmup_steps, warmup_rate,\n                                 learning_rate)\n    return np.where(global_step > total_steps, 0.0, learning_rate)\n\n\nclass WarmUpCosineDecayScheduler(Callback):\n    \"\"\"Cosine decay with warmup learning rate scheduler\"\"\"\n\n    def __init__(self,\n                 learning_rate_base,\n                 total_steps,\n                 global_step_init=0,\n                 warmup_learning_rate=0.0,\n                 warmup_steps=0,\n                 hold_base_rate_steps=0,\n                 verbose=0):\n        \"\"\"\n        Constructor for cosine decay with warmup learning rate scheduler.\n        :param learning_rate_base {float}: base learning rate.\n        :param total_steps {int}: total number of training steps.\n        :param global_step_init {int}: initial global step, e.g. from previous checkpoint.\n        :param warmup_learning_rate {float}: initial learning rate for warm up. (default: {0.0}).\n        :param warmup_steps {int}: number of warmup steps. (default: {0}).\n        :param hold_base_rate_steps {int}: Optional number of steps to hold base learning rate before decaying. (default: {0}).\n        :param verbose {int}: quiet, 1: update messages. (default: {0}).\n        \"\"\"\n\n        super(WarmUpCosineDecayScheduler, self).__init__()\n        self.learning_rate_base = learning_rate_base\n        self.total_steps = total_steps\n        self.global_step = global_step_init\n        self.warmup_learning_rate = warmup_learning_rate\n        self.warmup_steps = warmup_steps\n        self.hold_base_rate_steps = hold_base_rate_steps\n        self.verbose = verbose\n        self.learning_rates = []\n\n    def on_batch_end(self, batch, logs=None):\n        self.global_step = self.global_step + 1\n        lr = K.get_value(self.model.optimizer.lr)\n        self.learning_rates.append(lr)\n\n    def on_batch_begin(self, batch, logs=None):\n        lr = cosine_decay_with_warmup(global_step=self.global_step,\n                                      learning_rate_base=self.learning_rate_base,\n                                      total_steps=self.total_steps,\n                                      warmup_learning_rate=self.warmup_learning_rate,\n                                      warmup_steps=self.warmup_steps,\n                                      hold_base_rate_steps=self.hold_base_rate_steps)\n        K.set_value(self.model.optimizer.lr, lr)\n        if self.verbose > 0:\n            print('\\nBatch %02d: setting learning rate to %s.' % (self.global_step + 1, lr))","metadata":{"execution":{"iopub.status.busy":"2024-04-30T12:44:50.458284Z","iopub.execute_input":"2024-04-30T12:44:50.458590Z","iopub.status.idle":"2024-04-30T12:44:50.477654Z","shell.execute_reply.started":"2024-04-30T12:44:50.458546Z","shell.execute_reply":"2024-04-30T12:44:50.476805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model(input_shape):\n    input_tensor = Input(shape=input_shape)\n    base_model = EfficientNetB5(weights=None, \n                                include_top=False,\n                                input_tensor=input_tensor)\n    base_model.load_weights('../input/efficientnet-keras-weights-b0b5/efficientnet-b5_imagenet_1000_notop.h5')\n\n    x = GlobalAveragePooling2D()(base_model.output)\n    final_output = Dense(1, activation='linear', name='final_output')(x)\n    model = Model(input_tensor, final_output)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2024-04-30T12:44:54.250178Z","iopub.execute_input":"2024-04-30T12:44:54.250623Z","iopub.status.idle":"2024-04-30T12:44:54.256642Z","shell.execute_reply.started":"2024-04-30T12:44:54.250428Z","shell.execute_reply":"2024-04-30T12:44:54.255879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = create_model(input_shape=(HEIGHT, WIDTH, CHANNELS))\n\nfor layer in model.layers:\n    layer.trainable = False\n\nfor i in range(-2, 0):\n    model.layers[i].trainable = True\n\ncosine_lr_1st = WarmUpCosineDecayScheduler(learning_rate_base=WARMUP_LEARNING_RATE,\n                                           total_steps=TOTAL_STEPS_1st,\n                                           warmup_learning_rate=0.0,\n                                           warmup_steps=WARMUP_STEPS_1st,\n                                           hold_base_rate_steps=(2 * STEP_SIZE))\n\nmetric_list = [\"accuracy\"]\ncallback_list = [cosine_lr_1st]\noptimizer = optimizers.Adam(lr=WARMUP_LEARNING_RATE)\nmodel.compile(optimizer=optimizer, loss='mean_squared_error', metrics=metric_list)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T12:44:56.572445Z","iopub.execute_input":"2024-04-30T12:44:56.572753Z","iopub.status.idle":"2024-04-30T12:45:24.061157Z","shell.execute_reply.started":"2024-04-30T12:44:56.572709Z","shell.execute_reply":"2024-04-30T12:45:24.060168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEP_SIZE_TRAIN = train_generator.n//train_generator.batch_size\nSTEP_SIZE_VALID = valid_generator.n//valid_generator.batch_size\n\nhistory_warmup = model.fit_generator(generator=train_generator,\n                                     steps_per_epoch=STEP_SIZE_TRAIN,\n                                     validation_data=valid_generator,\n                                     validation_steps=STEP_SIZE_VALID,\n                                     epochs=WARMUP_EPOCHS,\n                                     callbacks=callback_list,\n                                     verbose=2).history","metadata":{"execution":{"iopub.status.busy":"2024-04-30T12:45:43.457161Z","iopub.execute_input":"2024-04-30T12:45:43.457619Z","iopub.status.idle":"2024-04-30T12:48:59.196186Z","shell.execute_reply.started":"2024-04-30T12:45:43.457419Z","shell.execute_reply":"2024-04-30T12:48:59.195447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in model.layers:\n    layer.trainable = True\n\nes = EarlyStopping(monitor='val_loss', mode='min', patience=ES_PATIENCE, restore_best_weights=True, verbose=1)\ncosine_lr_2nd = WarmUpCosineDecayScheduler(learning_rate_base=LEARNING_RATE,\n                                           total_steps=TOTAL_STEPS_2nd,\n                                           warmup_learning_rate=0.0,\n                                           warmup_steps=WARMUP_STEPS_2nd,\n                                           hold_base_rate_steps=(3 * STEP_SIZE))\n\ncallback_list = [es, cosine_lr_2nd]\noptimizer = optimizers.Adam(lr=LEARNING_RATE)\nmodel.compile(optimizer=optimizer, loss='mean_squared_error', metrics=metric_list)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T12:49:06.897432Z","iopub.execute_input":"2024-04-30T12:49:06.897771Z","iopub.status.idle":"2024-04-30T12:49:07.124852Z","shell.execute_reply.started":"2024-04-30T12:49:06.897725Z","shell.execute_reply":"2024-04-30T12:49:07.124105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit_generator(generator=train_generator,\n                              steps_per_epoch=STEP_SIZE_TRAIN,\n                              validation_data=valid_generator,\n                              validation_steps=STEP_SIZE_VALID,\n                              epochs=EPOCHS,\n                              callbacks=callback_list,\n                              verbose=2).history","metadata":{"execution":{"iopub.status.busy":"2024-04-30T12:49:24.704450Z","iopub.execute_input":"2024-04-30T12:49:24.704754Z","iopub.status.idle":"2024-04-30T13:16:17.744694Z","shell.execute_reply.started":"2024-04-30T12:49:24.704712Z","shell.execute_reply":"2024-04-30T13:16:17.743766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(2, 1, sharex='col', figsize=(20, 6))\n\nax1.plot(cosine_lr_1st.learning_rates)\nax1.set_title('Warm up learning rates')\n\nax2.plot(cosine_lr_2nd.learning_rates)\nax2.set_title('Fine-tune learning rates')\n\nplt.xlabel('Steps')\nplt.ylabel('Learning rate')\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:18:59.640834Z","iopub.execute_input":"2024-04-30T13:18:59.641240Z","iopub.status.idle":"2024-04-30T13:19:00.135577Z","shell.execute_reply.started":"2024-04-30T13:18:59.641179Z","shell.execute_reply":"2024-04-30T13:19:00.134689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(2, 1, sharex='col', figsize=(20, 14))\n\nax1.plot(history['loss'], label='Train loss')\nax1.plot(history['val_loss'], label='Validation loss')\nax1.legend(loc='best')\nax1.set_title('Loss')\n\nax2.plot(history['acc'], label='Train accuracy')\nax2.plot(history['val_acc'], label='Validation accuracy')\nax2.legend(loc='best')\nax2.set_title('Accuracy')\n\nplt.xlabel('Epochs')\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:19:03.642502Z","iopub.execute_input":"2024-04-30T13:19:03.642804Z","iopub.status.idle":"2024-04-30T13:19:04.116806Z","shell.execute_reply.started":"2024-04-30T13:19:03.642759Z","shell.execute_reply":"2024-04-30T13:19:04.115951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create empty arays to keep the predictions and labels\ndf_preds = pd.DataFrame(columns=['label', 'pred', 'set'])\ntrain_generator.reset()\nvalid_generator.reset()\n\n# Add train predictions and labels\nfor i in range(STEP_SIZE_TRAIN + 1):\n    im, lbl = next(train_generator)\n    preds = model.predict(im, batch_size=train_generator.batch_size)\n    for index in range(len(preds)):\n        df_preds.loc[len(df_preds)] = [lbl[index], preds[index][0], 'train']\n\n# Add validation predictions and labels\nfor i in range(STEP_SIZE_VALID + 1):\n    im, lbl = next(valid_generator)\n    preds = model.predict(im, batch_size=valid_generator.batch_size)\n    for index in range(len(preds)):\n        df_preds.loc[len(df_preds)] = [lbl[index], preds[index][0], 'validation']\n\ndf_preds['label'] = df_preds['label'].astype('int')","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:19:08.311788Z","iopub.execute_input":"2024-04-30T13:19:08.312133Z","iopub.status.idle":"2024-04-30T13:20:23.966646Z","shell.execute_reply.started":"2024-04-30T13:19:08.312069Z","shell.execute_reply":"2024-04-30T13:20:23.965913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def classify(x):\n    if x < 0.5:\n        return 0\n    elif x < 1.5:\n        return 1\n    elif x < 2.5:\n        return 2\n    elif x < 3.5:\n        return 3\n    return 4\n\n# Classify predictions\ndf_preds['predictions'] = df_preds['pred'].apply(lambda x: classify(x))\n\ntrain_preds = df_preds[df_preds['set'] == 'train']\nvalidation_preds = df_preds[df_preds['set'] == 'validation']","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:20:31.221402Z","iopub.execute_input":"2024-04-30T13:20:31.221699Z","iopub.status.idle":"2024-04-30T13:20:31.240308Z","shell.execute_reply.started":"2024-04-30T13:20:31.221657Z","shell.execute_reply":"2024-04-30T13:20:31.239594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = ['0 - No DR', '1 - Mild', '2 - Moderate', '3 - Severe', '4 - Proliferative DR']\ndef plot_confusion_matrix(train, validation, labels=labels):\n    train_labels, train_preds = train\n    validation_labels, validation_preds = validation\n    fig, (ax1, ax2) = plt.subplots(1, 2, sharex='col', figsize=(24, 7))\n    train_cnf_matrix = confusion_matrix(train_labels, train_preds)\n    validation_cnf_matrix = confusion_matrix(validation_labels, validation_preds)\n\n    train_cnf_matrix_norm = train_cnf_matrix.astype('float') / train_cnf_matrix.sum(axis=1)[:, np.newaxis]\n    validation_cnf_matrix_norm = validation_cnf_matrix.astype('float') / validation_cnf_matrix.sum(axis=1)[:, np.newaxis]\n\n    train_df_cm = pd.DataFrame(train_cnf_matrix_norm, index=labels, columns=labels)\n    validation_df_cm = pd.DataFrame(validation_cnf_matrix_norm, index=labels, columns=labels)\n\n    sns.heatmap(train_df_cm, annot=True, fmt='.2f', cmap=\"Blues\",ax=ax1).set_title('Train')\n    sns.heatmap(validation_df_cm, annot=True, fmt='.2f', cmap=sns.cubehelix_palette(8),ax=ax2).set_title('Validation')\n    plt.show()\n\nplot_confusion_matrix((train_preds['label'], train_preds['predictions']), (validation_preds['label'], validation_preds['predictions']))","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:20:35.897763Z","iopub.execute_input":"2024-04-30T13:20:35.898125Z","iopub.status.idle":"2024-04-30T13:20:36.911356Z","shell.execute_reply.started":"2024-04-30T13:20:35.898067Z","shell.execute_reply":"2024-04-30T13:20:36.910195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def evaluate_model(train, validation):\n    train_labels, train_preds = train\n    validation_labels, validation_preds = validation\n    print(\"Train        Cohen Kappa score: %.3f\" % cohen_kappa_score(train_preds, train_labels, weights='quadratic'))\n    print(\"Validation   Cohen Kappa score: %.3f\" % cohen_kappa_score(validation_preds, validation_labels, weights='quadratic'))\n    print(\"Complete set Cohen Kappa score: %.3f\" % cohen_kappa_score(np.append(train_preds, validation_preds), np.append(train_labels, validation_labels), weights='quadratic'))\n    \nevaluate_model((train_preds['label'], train_preds['predictions']), (validation_preds['label'], validation_preds['predictions']))","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:20:42.028593Z","iopub.execute_input":"2024-04-30T13:20:42.028930Z","iopub.status.idle":"2024-04-30T13:20:42.052556Z","shell.execute_reply.started":"2024-04-30T13:20:42.028872Z","shell.execute_reply":"2024-04-30T13:20:42.051809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def apply_tta(model, generator, steps=10):\n    step_size = generator.n//generator.batch_size\n    preds_tta = []\n    for i in range(steps):\n        generator.reset()\n        preds = model.predict_generator(generator, steps=step_size)\n        preds_tta.append(preds)\n\n    return np.mean(preds_tta, axis=0)\n\npreds = apply_tta(model, test_generator)\npredictions = [classify(x) for x in preds]\n\nresults = pd.DataFrame({'id_code':test['id_code'], 'diagnosis':predictions})\nresults['id_code'] = results['id_code'].map(lambda x: str(x)[:-4])","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:20:45.165185Z","iopub.execute_input":"2024-04-30T13:20:45.165479Z","iopub.status.idle":"2024-04-30T13:31:57.204429Z","shell.execute_reply.started":"2024-04-30T13:20:45.165437Z","shell.execute_reply":"2024-04-30T13:31:57.203671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cleaning created directories\nif os.path.exists(train_dest_path):\n    shutil.rmtree(train_dest_path)\nif os.path.exists(validation_dest_path):\n    shutil.rmtree(validation_dest_path)\nif os.path.exists(test_dest_path):\n    shutil.rmtree(test_dest_path)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:32:12.174835Z","iopub.execute_input":"2024-04-30T13:32:12.175157Z","iopub.status.idle":"2024-04-30T13:32:12.448606Z","shell.execute_reply.started":"2024-04-30T13:32:12.175110Z","shell.execute_reply":"2024-04-30T13:32:12.447914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.subplots(sharex='col', figsize=(24, 8.7))\nsns.countplot(x=\"diagnosis\", data=results).set_title('Test')\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:32:19.789541Z","iopub.execute_input":"2024-04-30T13:32:19.789878Z","iopub.status.idle":"2024-04-30T13:32:20.161203Z","shell.execute_reply.started":"2024-04-30T13:32:19.789821Z","shell.execute_reply":"2024-04-30T13:32:20.160411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results.to_csv('submission.csv', index=False)\ndisplay(results.head())","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:32:23.178087Z","iopub.execute_input":"2024-04-30T13:32:23.178400Z","iopub.status.idle":"2024-04-30T13:32:23.340569Z","shell.execute_reply.started":"2024-04-30T13:32:23.178357Z","shell.execute_reply":"2024-04-30T13:32:23.339682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save_weights('../working/effNetB5_bs32_img224_fold1.h5')","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:33:01.642788Z","iopub.execute_input":"2024-04-30T13:33:01.643139Z","iopub.status.idle":"2024-04-30T13:35:48.707327Z","shell.execute_reply.started":"2024-04-30T13:33:01.643079Z","shell.execute_reply":"2024-04-30T13:35:48.706627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"FOLD_2","metadata":{}},{"cell_type":"code","source":"import os\nimport sys\nimport cv2\nimport shutil\nimport random\nimport warnings\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom tensorflow import set_random_seed\nfrom sklearn.utils import class_weight\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, cohen_kappa_score\nfrom keras import backend as K\nfrom keras.models import Model\nfrom keras.utils import to_categorical\nfrom keras import optimizers, applications\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.layers import Dense, Dropout, GlobalAveragePooling2D, Input\nfrom keras.callbacks import EarlyStopping, ReduceLROnPlateau, Callback, LearningRateScheduler\n\ndef seed_everything(seed=0):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    set_random_seed(0)\n\nseed = 0\nseed_everything(seed)\n%matplotlib inline\nsns.set(style=\"whitegrid\")\nwarnings.filterwarnings(\"ignore\")\nsys.path.append(os.path.abspath('../input/efficientnet/efficientnet-master/efficientnet-master/'))\nfrom efficientnet import *","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:38:59.037259Z","iopub.execute_input":"2024-04-30T13:38:59.037637Z","iopub.status.idle":"2024-04-30T13:38:59.060047Z","shell.execute_reply.started":"2024-04-30T13:38:59.037548Z","shell.execute_reply":"2024-04-30T13:38:59.059168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fold_set = pd.read_csv('/kaggle/input/5-fold/5-fold.csv')\nX_train = fold_set[fold_set['fold_1'] == 'train']\nX_val = fold_set[fold_set['fold_1'] == 'validation']\ntest = pd.read_csv('../input/aptos2019-blindness-detection/test.csv')\nprint('Number of train samples: ', X_train.shape[0])\nprint('Number of validation samples: ', X_val.shape[0])\nprint('Number of test samples: ', test.shape[0])\n\n# Preprocecss data\nX_train[\"id_code\"] = X_train[\"id_code\"].apply(lambda x: x + \".png\")\nX_val[\"id_code\"] = X_val[\"id_code\"].apply(lambda x: x + \".png\")\ntest[\"id_code\"] = test[\"id_code\"].apply(lambda x: x + \".png\")\ndisplay(X_train.head())","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:39:23.235880Z","iopub.execute_input":"2024-04-30T13:39:23.236252Z","iopub.status.idle":"2024-04-30T13:39:24.281913Z","shell.execute_reply.started":"2024-04-30T13:39:23.236193Z","shell.execute_reply":"2024-04-30T13:39:24.281101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model parameters\nFACTOR = 4\nBATCH_SIZE = 8 * FACTOR\nEPOCHS = 20\nWARMUP_EPOCHS = 5\nLEARNING_RATE = 1e-4 * FACTOR\nWARMUP_LEARNING_RATE = 1e-3 * FACTOR\nHEIGHT = 224\nWIDTH = 224\nCHANNELS = 3\nES_PATIENCE = 5\nRLROP_PATIENCE = 3\nDECAY_DROP = 0.5\nLR_WARMUP_EPOCHS_1st = 2\nLR_WARMUP_EPOCHS_2nd = 5\nSTEP_SIZE = len(X_train) // BATCH_SIZE\nTOTAL_STEPS_1st = WARMUP_EPOCHS * STEP_SIZE\nTOTAL_STEPS_2nd = EPOCHS * STEP_SIZE\nWARMUP_STEPS_1st = LR_WARMUP_EPOCHS_1st * STEP_SIZE\nWARMUP_STEPS_2nd = LR_WARMUP_EPOCHS_2nd * STEP_SIZE","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:39:35.080363Z","iopub.execute_input":"2024-04-30T13:39:35.080680Z","iopub.status.idle":"2024-04-30T13:39:35.088586Z","shell.execute_reply.started":"2024-04-30T13:39:35.080635Z","shell.execute_reply":"2024-04-30T13:39:35.087505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_base_path = '/kaggle/input/aptos2019-blindness-detection/train_images/'\ntest_base_path = '/kaggle/input/aptos2019-blindness-detection/test_images/'\ntrain_dest_path = 'base_dir/train_images/'\nvalidation_dest_path = 'base_dir/validation_images/'\ntest_dest_path =  'base_dir/test_images/'\n\n# Making sure directories don't exist\nif os.path.exists(train_dest_path):\n    shutil.rmtree(train_dest_path)\nif os.path.exists(validation_dest_path):\n    shutil.rmtree(validation_dest_path)\nif os.path.exists(test_dest_path):\n    shutil.rmtree(test_dest_path)\n    \n# Creating train, validation and test directories\nos.makedirs(train_dest_path)\nos.makedirs(validation_dest_path)\nos.makedirs(test_dest_path)\n\ndef crop_image(img, tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n            img = np.stack([img1,img2,img3],axis=-1)\n            \n        return img\n\ndef circle_crop(img):\n    img = crop_image(img)\n\n    height, width, depth = img.shape\n    largest_side = np.max((height, width))\n    img = cv2.resize(img, (largest_side, largest_side))\n\n    height, width, depth = img.shape\n\n    x = width//2\n    y = height//2\n    r = np.amin((x, y))\n\n    circle_img = np.zeros((height, width), np.uint8)\n    cv2.circle(circle_img, (x, y), int(r), 1, thickness=-1)\n    img = cv2.bitwise_and(img, img, mask=circle_img)\n    img = crop_image(img)\n\n    return img\n    \ndef preprocess_image(base_path, save_path, image_id, HEIGHT, WIDTH, sigmaX=10):\n    image = cv2.imread(base_path + image_id)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    image = circle_crop(image)\n    image = cv2.resize(image, (HEIGHT, WIDTH))\n    image = cv2.addWeighted(image, 4, cv2.GaussianBlur(image, (0,0), sigmaX), -4 , 128)\n    cv2.imwrite(save_path + image_id, image)\n    \n# Pre-procecss train set\nfor i, image_id in enumerate(X_train['id_code']):\n    preprocess_image(train_base_path, train_dest_path, image_id, HEIGHT, WIDTH)\n    \n# Pre-procecss validation set\nfor i, image_id in enumerate(X_val['id_code']):\n    preprocess_image(train_base_path, validation_dest_path, image_id, HEIGHT, WIDTH)\n    \n# Pre-procecss test set\nfor i, image_id in enumerate(test['id_code']):\n    preprocess_image(test_base_path, test_dest_path, image_id, HEIGHT, WIDTH)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:40:05.811903Z","iopub.execute_input":"2024-04-30T13:40:05.812239Z","iopub.status.idle":"2024-04-30T13:59:05.009787Z","shell.execute_reply.started":"2024-04-30T13:40:05.812190Z","shell.execute_reply":"2024-04-30T13:59:05.009016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datagen=ImageDataGenerator(rescale=1./255, \n                           rotation_range=360,\n                           horizontal_flip=True,\n                           vertical_flip=True)\n\ntrain_generator=datagen.flow_from_dataframe(\n                        dataframe=X_train,\n                        directory=train_dest_path,\n                        x_col=\"id_code\",\n                        y_col=\"diagnosis\",\n                        class_mode=\"raw\",\n                        batch_size=BATCH_SIZE,\n                        target_size=(HEIGHT, WIDTH),\n                        seed=seed)\n\nvalid_generator=datagen.flow_from_dataframe(\n                        dataframe=X_val,\n                        directory=validation_dest_path,\n                        x_col=\"id_code\",\n                        y_col=\"diagnosis\",\n                        class_mode=\"raw\",\n                        batch_size=BATCH_SIZE,\n                        target_size=(HEIGHT, WIDTH),\n                        seed=seed)\n\ntest_generator=datagen.flow_from_dataframe(  \n                       dataframe=test,\n                       directory=test_dest_path,\n                       x_col=\"id_code\",\n                       batch_size=1,\n                       class_mode=None,\n                       shuffle=False,\n                       target_size=(HEIGHT, WIDTH),\n                       seed=seed)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:59:54.382702Z","iopub.execute_input":"2024-04-30T13:59:54.383175Z","iopub.status.idle":"2024-04-30T13:59:54.464687Z","shell.execute_reply.started":"2024-04-30T13:59:54.382963Z","shell.execute_reply":"2024-04-30T13:59:54.463941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cosine_decay_with_warmup(global_step,\n                             learning_rate_base,\n                             total_steps,\n                             warmup_learning_rate=0.0,\n                             warmup_steps=0,\n                             hold_base_rate_steps=0):\n    \"\"\"\n    Cosine decay schedule with warm up period.\n    In this schedule, the learning rate grows linearly from warmup_learning_rate\n    to learning_rate_base for warmup_steps, then transitions to a cosine decay\n    schedule.\n    :param global_step {int}: global step.\n    :param learning_rate_base {float}: base learning rate.\n    :param total_steps {int}: total number of training steps.\n    :param warmup_learning_rate {float}: initial learning rate for warm up. (default: {0.0}).\n    :param warmup_steps {int}: number of warmup steps. (default: {0}).\n    :param hold_base_rate_steps {int}: Optional number of steps to hold base learning rate before decaying. (default: {0}).\n    :param global_step {int}: global step.\n    :Returns : a float representing learning rate.\n    :Raises ValueError: if warmup_learning_rate is larger than learning_rate_base, or if warmup_steps is larger than total_steps.\n    \"\"\"\n\n    if total_steps < warmup_steps:\n        raise ValueError('total_steps must be larger or equal to warmup_steps.')\n    learning_rate = 0.5 * learning_rate_base * (1 + np.cos(\n        np.pi *\n        (global_step - warmup_steps - hold_base_rate_steps\n         ) / float(total_steps - warmup_steps - hold_base_rate_steps)))\n    if hold_base_rate_steps > 0:\n        learning_rate = np.where(global_step > warmup_steps + hold_base_rate_steps,\n                                 learning_rate, learning_rate_base)\n    if warmup_steps > 0:\n        if learning_rate_base < warmup_learning_rate:\n            raise ValueError('learning_rate_base must be larger or equal to warmup_learning_rate.')\n        slope = (learning_rate_base - warmup_learning_rate) / warmup_steps\n        warmup_rate = slope * global_step + warmup_learning_rate\n        learning_rate = np.where(global_step < warmup_steps, warmup_rate,\n                                 learning_rate)\n    return np.where(global_step > total_steps, 0.0, learning_rate)\n\n\nclass WarmUpCosineDecayScheduler(Callback):\n    \"\"\"Cosine decay with warmup learning rate scheduler\"\"\"\n\n    def __init__(self,\n                 learning_rate_base,\n                 total_steps,\n                 global_step_init=0,\n                 warmup_learning_rate=0.0,\n                 warmup_steps=0,\n                 hold_base_rate_steps=0,\n                 verbose=0):\n        \"\"\"\n        Constructor for cosine decay with warmup learning rate scheduler.\n        :param learning_rate_base {float}: base learning rate.\n        :param total_steps {int}: total number of training steps.\n        :param global_step_init {int}: initial global step, e.g. from previous checkpoint.\n        :param warmup_learning_rate {float}: initial learning rate for warm up. (default: {0.0}).\n        :param warmup_steps {int}: number of warmup steps. (default: {0}).\n        :param hold_base_rate_steps {int}: Optional number of steps to hold base learning rate before decaying. (default: {0}).\n        :param verbose {int}: quiet, 1: update messages. (default: {0}).\n        \"\"\"\n\n        super(WarmUpCosineDecayScheduler, self).__init__()\n        self.learning_rate_base = learning_rate_base\n        self.total_steps = total_steps\n        self.global_step = global_step_init\n        self.warmup_learning_rate = warmup_learning_rate\n        self.warmup_steps = warmup_steps\n        self.hold_base_rate_steps = hold_base_rate_steps\n        self.verbose = verbose\n        self.learning_rates = []\n\n    def on_batch_end(self, batch, logs=None):\n        self.global_step = self.global_step + 1\n        lr = K.get_value(self.model.optimizer.lr)\n        self.learning_rates.append(lr)\n\n    def on_batch_begin(self, batch, logs=None):\n        lr = cosine_decay_with_warmup(global_step=self.global_step,\n                                      learning_rate_base=self.learning_rate_base,\n                                      total_steps=self.total_steps,\n                                      warmup_learning_rate=self.warmup_learning_rate,\n                                      warmup_steps=self.warmup_steps,\n                                      hold_base_rate_steps=self.hold_base_rate_steps)\n        K.set_value(self.model.optimizer.lr, lr)\n        if self.verbose > 0:\n            print('\\nBatch %02d: setting learning rate to %s.' % (self.global_step + 1, lr))","metadata":{"execution":{"iopub.status.busy":"2024-04-30T13:59:58.356106Z","iopub.execute_input":"2024-04-30T13:59:58.356396Z","iopub.status.idle":"2024-04-30T13:59:58.376216Z","shell.execute_reply.started":"2024-04-30T13:59:58.356355Z","shell.execute_reply":"2024-04-30T13:59:58.375189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model(input_shape):\n    input_tensor = Input(shape=input_shape)\n    base_model = EfficientNetB5(weights=None, \n                                include_top=False,\n                                input_tensor=input_tensor)\n    base_model.load_weights('../input/efficientnet-keras-weights-b0b5/efficientnet-b5_imagenet_1000_notop.h5')\n\n    x = GlobalAveragePooling2D()(base_model.output)\n    final_output = Dense(1, activation='linear', name='final_output')(x)\n    model = Model(input_tensor, final_output)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2024-04-30T14:00:01.682982Z","iopub.execute_input":"2024-04-30T14:00:01.683302Z","iopub.status.idle":"2024-04-30T14:00:01.689628Z","shell.execute_reply.started":"2024-04-30T14:00:01.683257Z","shell.execute_reply":"2024-04-30T14:00:01.688774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = create_model(input_shape=(HEIGHT, WIDTH, CHANNELS))\n\nfor layer in model.layers:\n    layer.trainable = False\n\nfor i in range(-2, 0):\n    model.layers[i].trainable = True\n\ncosine_lr_1st = WarmUpCosineDecayScheduler(learning_rate_base=WARMUP_LEARNING_RATE,\n                                           total_steps=TOTAL_STEPS_1st,\n                                           warmup_learning_rate=0.0,\n                                           warmup_steps=WARMUP_STEPS_1st,\n                                           hold_base_rate_steps=(2 * STEP_SIZE))\n\nmetric_list = [\"accuracy\"]\ncallback_list = [cosine_lr_1st]\noptimizer = optimizers.Adam(lr=WARMUP_LEARNING_RATE)\nmodel.compile(optimizer=optimizer, loss='mean_squared_error', metrics=metric_list)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T14:00:06.109383Z","iopub.execute_input":"2024-04-30T14:00:06.109701Z","iopub.status.idle":"2024-04-30T14:00:39.663598Z","shell.execute_reply.started":"2024-04-30T14:00:06.109644Z","shell.execute_reply":"2024-04-30T14:00:39.662864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEP_SIZE_TRAIN = train_generator.n//train_generator.batch_size\nSTEP_SIZE_VALID = valid_generator.n//valid_generator.batch_size\n\nhistory_warmup = model.fit_generator(generator=train_generator,\n                                     steps_per_epoch=STEP_SIZE_TRAIN,\n                                     validation_data=valid_generator,\n                                     validation_steps=STEP_SIZE_VALID,\n                                     epochs=WARMUP_EPOCHS,\n                                     callbacks=callback_list,\n                                     verbose=2).history","metadata":{"execution":{"iopub.status.busy":"2024-04-30T14:00:56.488466Z","iopub.execute_input":"2024-04-30T14:00:56.488759Z","iopub.status.idle":"2024-04-30T14:04:20.498426Z","shell.execute_reply.started":"2024-04-30T14:00:56.488716Z","shell.execute_reply":"2024-04-30T14:04:20.497452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in model.layers:\n    layer.trainable = True\n\nes = EarlyStopping(monitor='val_loss', mode='min', patience=ES_PATIENCE, restore_best_weights=True, verbose=1)\ncosine_lr_2nd = WarmUpCosineDecayScheduler(learning_rate_base=LEARNING_RATE,\n                                           total_steps=TOTAL_STEPS_2nd,\n                                           warmup_learning_rate=0.0,\n                                           warmup_steps=WARMUP_STEPS_2nd,\n                                           hold_base_rate_steps=(3 * STEP_SIZE))\n\ncallback_list = [es, cosine_lr_2nd]\noptimizer = optimizers.Adam(lr=LEARNING_RATE)\nmodel.compile(optimizer=optimizer, loss='mean_squared_error', metrics=metric_list)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T14:05:10.861551Z","iopub.execute_input":"2024-04-30T14:05:10.861845Z","iopub.status.idle":"2024-04-30T14:05:11.127606Z","shell.execute_reply.started":"2024-04-30T14:05:10.861804Z","shell.execute_reply":"2024-04-30T14:05:11.126064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit_generator(generator=train_generator,\n                              steps_per_epoch=STEP_SIZE_TRAIN,\n                              validation_data=valid_generator,\n                              validation_steps=STEP_SIZE_VALID,\n                              epochs=EPOCHS,\n                              callbacks=callback_list,\n                              verbose=2).history","metadata":{"execution":{"iopub.status.busy":"2024-04-30T14:05:24.027364Z","iopub.execute_input":"2024-04-30T14:05:24.027794Z","iopub.status.idle":"2024-04-30T14:24:47.789909Z","shell.execute_reply.started":"2024-04-30T14:05:24.027718Z","shell.execute_reply":"2024-04-30T14:24:47.789045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(2, 1, sharex='col', figsize=(20, 6))\n\nax1.plot(cosine_lr_1st.learning_rates)\nax1.set_title('Warm up learning rates')\n\nax2.plot(cosine_lr_2nd.learning_rates)\nax2.set_title('Fine-tune learning rates')\n\nplt.xlabel('Steps')\nplt.ylabel('Learning rate')\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T14:26:47.518234Z","iopub.execute_input":"2024-04-30T14:26:47.518551Z","iopub.status.idle":"2024-04-30T14:26:47.830913Z","shell.execute_reply.started":"2024-04-30T14:26:47.518499Z","shell.execute_reply":"2024-04-30T14:26:47.830130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(2, 1, sharex='col', figsize=(20, 14))\n\nax1.plot(history['loss'], label='Train loss')\nax1.plot(history['val_loss'], label='Validation loss')\nax1.legend(loc='best')\nax1.set_title('Loss')\n\nax2.plot(history['acc'], label='Train accuracy')\nax2.plot(history['val_acc'], label='Validation accuracy')\nax2.legend(loc='best')\nax2.set_title('Accuracy')\n\nplt.xlabel('Epochs')\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T14:26:50.471509Z","iopub.execute_input":"2024-04-30T14:26:50.471833Z","iopub.status.idle":"2024-04-30T14:26:50.988174Z","shell.execute_reply.started":"2024-04-30T14:26:50.471787Z","shell.execute_reply":"2024-04-30T14:26:50.987202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create empty arays to keep the predictions and labels\ndf_preds = pd.DataFrame(columns=['label', 'pred', 'set'])\ntrain_generator.reset()\nvalid_generator.reset()\n\n# Add train predictions and labels\nfor i in range(STEP_SIZE_TRAIN + 1):\n    im, lbl = next(train_generator)\n    preds = model.predict(im, batch_size=train_generator.batch_size)\n    for index in range(len(preds)):\n        df_preds.loc[len(df_preds)] = [lbl[index], preds[index][0], 'train']\n\n# Add validation predictions and labels\nfor i in range(STEP_SIZE_VALID + 1):\n    im, lbl = next(valid_generator)\n    preds = model.predict(im, batch_size=valid_generator.batch_size)\n    for index in range(len(preds)):\n        df_preds.loc[len(df_preds)] = [lbl[index], preds[index][0], 'validation']\n\ndf_preds['label'] = df_preds['label'].astype('int')","metadata":{"execution":{"iopub.status.busy":"2024-04-30T14:26:53.882980Z","iopub.execute_input":"2024-04-30T14:26:53.883289Z","iopub.status.idle":"2024-04-30T14:28:13.486589Z","shell.execute_reply.started":"2024-04-30T14:26:53.883245Z","shell.execute_reply":"2024-04-30T14:28:13.485862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def classify(x):\n    if x < 0.5:\n        return 0\n    elif x < 1.5:\n        return 1\n    elif x < 2.5:\n        return 2\n    elif x < 3.5:\n        return 3\n    return 4\n\n# Classify predictions\ndf_preds['predictions'] = df_preds['pred'].apply(lambda x: classify(x))\n\ntrain_preds = df_preds[df_preds['set'] == 'train']\nvalidation_preds = df_preds[df_preds['set'] == 'validation']","metadata":{"execution":{"iopub.status.busy":"2024-04-30T14:29:40.341390Z","iopub.execute_input":"2024-04-30T14:29:40.341697Z","iopub.status.idle":"2024-04-30T14:29:40.357849Z","shell.execute_reply.started":"2024-04-30T14:29:40.341652Z","shell.execute_reply":"2024-04-30T14:29:40.357105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = ['0 - No DR', '1 - Mild', '2 - Moderate', '3 - Severe', '4 - Proliferative DR']\ndef plot_confusion_matrix(train, validation, labels=labels):\n    train_labels, train_preds = train\n    validation_labels, validation_preds = validation\n    fig, (ax1, ax2) = plt.subplots(1, 2, sharex='col', figsize=(24, 7))\n    train_cnf_matrix = confusion_matrix(train_labels, train_preds)\n    validation_cnf_matrix = confusion_matrix(validation_labels, validation_preds)\n\n    train_cnf_matrix_norm = train_cnf_matrix.astype('float') / train_cnf_matrix.sum(axis=1)[:, np.newaxis]\n    validation_cnf_matrix_norm = validation_cnf_matrix.astype('float') / validation_cnf_matrix.sum(axis=1)[:, np.newaxis]\n\n    train_df_cm = pd.DataFrame(train_cnf_matrix_norm, index=labels, columns=labels)\n    validation_df_cm = pd.DataFrame(validation_cnf_matrix_norm, index=labels, columns=labels)\n\n    sns.heatmap(train_df_cm, annot=True, fmt='.2f', cmap=\"Blues\",ax=ax1).set_title('Train')\n    sns.heatmap(validation_df_cm, annot=True, fmt='.2f', cmap=sns.cubehelix_palette(8),ax=ax2).set_title('Validation')\n    plt.show()\n\nplot_confusion_matrix((train_preds['label'], train_preds['predictions']), (validation_preds['label'], validation_preds['predictions']))","metadata":{"execution":{"iopub.status.busy":"2024-04-30T14:29:43.518481Z","iopub.execute_input":"2024-04-30T14:29:43.518793Z","iopub.status.idle":"2024-04-30T14:29:44.994367Z","shell.execute_reply.started":"2024-04-30T14:29:43.518747Z","shell.execute_reply":"2024-04-30T14:29:44.993502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def evaluate_model(train, validation):\n    train_labels, train_preds = train\n    validation_labels, validation_preds = validation\n    print(\"Train        Cohen Kappa score: %.3f\" % cohen_kappa_score(train_preds, train_labels, weights='quadratic'))\n    print(\"Validation   Cohen Kappa score: %.3f\" % cohen_kappa_score(validation_preds, validation_labels, weights='quadratic'))\n    print(\"Complete set Cohen Kappa score: %.3f\" % cohen_kappa_score(np.append(train_preds, validation_preds), np.append(train_labels, validation_labels), weights='quadratic'))\n    \nevaluate_model((train_preds['label'], train_preds['predictions']), (validation_preds['label'], validation_preds['predictions']))","metadata":{"execution":{"iopub.status.busy":"2024-04-30T14:29:48.982397Z","iopub.execute_input":"2024-04-30T14:29:48.982698Z","iopub.status.idle":"2024-04-30T14:29:49.008198Z","shell.execute_reply.started":"2024-04-30T14:29:48.982656Z","shell.execute_reply":"2024-04-30T14:29:49.007452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def apply_tta(model, generator, steps=10):\n    step_size = generator.n//generator.batch_size\n    preds_tta = []\n    for i in range(steps):\n        generator.reset()\n        preds = model.predict_generator(generator, steps=step_size)\n        preds_tta.append(preds)\n\n    return np.mean(preds_tta, axis=0)\n\npreds = apply_tta(model, test_generator)\npredictions = [classify(x) for x in preds]\n\nresults = pd.DataFrame({'id_code':test['id_code'], 'diagnosis':predictions})\nresults['id_code'] = results['id_code'].map(lambda x: str(x)[:-4])","metadata":{"execution":{"iopub.status.busy":"2024-04-30T14:29:51.807144Z","iopub.execute_input":"2024-04-30T14:29:51.807460Z","iopub.status.idle":"2024-04-30T14:42:55.227123Z","shell.execute_reply.started":"2024-04-30T14:29:51.807396Z","shell.execute_reply":"2024-04-30T14:42:55.226376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cleaning created directories\nif os.path.exists(train_dest_path):\n    shutil.rmtree(train_dest_path)\nif os.path.exists(validation_dest_path):\n    shutil.rmtree(validation_dest_path)\nif os.path.exists(test_dest_path):\n    shutil.rmtree(test_dest_path)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T14:44:44.866885Z","iopub.execute_input":"2024-04-30T14:44:44.867364Z","iopub.status.idle":"2024-04-30T14:44:45.151694Z","shell.execute_reply.started":"2024-04-30T14:44:44.867175Z","shell.execute_reply":"2024-04-30T14:44:45.151028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.subplots(sharex='col', figsize=(24, 8.7))\nsns.countplot(x=\"diagnosis\", data=results, palette=\"GnBu_d\").set_title('Test')\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T14:44:47.378647Z","iopub.execute_input":"2024-04-30T14:44:47.378945Z","iopub.status.idle":"2024-04-30T14:44:47.762556Z","shell.execute_reply.started":"2024-04-30T14:44:47.378902Z","shell.execute_reply":"2024-04-30T14:44:47.761756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results.to_csv('submission.csv', index=False)\ndisplay(results.head())","metadata":{"execution":{"iopub.status.busy":"2024-04-30T14:44:50.966334Z","iopub.execute_input":"2024-04-30T14:44:50.966692Z","iopub.status.idle":"2024-04-30T14:44:50.987436Z","shell.execute_reply.started":"2024-04-30T14:44:50.966643Z","shell.execute_reply":"2024-04-30T14:44:50.986422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save_weights('../working/effNetB5_bs32_img224_fold2.h5')","metadata":{"execution":{"iopub.status.busy":"2024-04-30T14:44:53.303462Z","iopub.execute_input":"2024-04-30T14:44:53.303755Z","iopub.status.idle":"2024-04-30T14:50:58.035454Z","shell.execute_reply.started":"2024-04-30T14:44:53.303713Z","shell.execute_reply":"2024-04-30T14:50:58.034630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"FOLD 3\n","metadata":{}},{"cell_type":"code","source":"import os\nimport sys\nimport cv2\nimport shutil\nimport random\nimport warnings\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom tensorflow import set_random_seed\nfrom sklearn.utils import class_weight\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, cohen_kappa_score\nfrom keras import backend as K\nfrom keras.models import Model\nfrom keras.utils import to_categorical\nfrom keras import optimizers, applications\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.layers import Dense, Dropout, GlobalAveragePooling2D, Input\nfrom keras.callbacks import EarlyStopping, ReduceLROnPlateau, Callback, LearningRateScheduler\n\ndef seed_everything(seed=0):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    set_random_seed(0)\n\nseed = 0\nseed_everything(seed)\n%matplotlib inline\nsns.set(style=\"whitegrid\")\nwarnings.filterwarnings(\"ignore\")\nsys.path.append(os.path.abspath('../input/efficientnet/efficientnet-master/efficientnet-master/'))\nfrom efficientnet import *","metadata":{"execution":{"iopub.status.busy":"2024-04-30T14:53:39.015177Z","iopub.execute_input":"2024-04-30T14:53:39.015509Z","iopub.status.idle":"2024-04-30T14:53:39.035282Z","shell.execute_reply.started":"2024-04-30T14:53:39.015456Z","shell.execute_reply":"2024-04-30T14:53:39.034554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fold_set = pd.read_csv('/kaggle/input/5-fold/5-fold.csv')\nX_train = fold_set[fold_set['fold_2'] == 'train']\nX_val = fold_set[fold_set['fold_2'] == 'validation']\ntest = pd.read_csv('../input/aptos2019-blindness-detection/test.csv')\nprint('Number of train samples: ', X_train.shape[0])\nprint('Number of validation samples: ', X_val.shape[0])\nprint('Number of test samples: ', test.shape[0])\n\n# Preprocecss data\nX_train[\"id_code\"] = X_train[\"id_code\"].apply(lambda x: x + \".png\")\nX_val[\"id_code\"] = X_val[\"id_code\"].apply(lambda x: x + \".png\")\ntest[\"id_code\"] = test[\"id_code\"].apply(lambda x: x + \".png\")\ndisplay(X_train.head())","metadata":{"execution":{"iopub.status.busy":"2024-04-30T14:54:16.918455Z","iopub.execute_input":"2024-04-30T14:54:16.918775Z","iopub.status.idle":"2024-04-30T14:54:18.649433Z","shell.execute_reply.started":"2024-04-30T14:54:16.918732Z","shell.execute_reply":"2024-04-30T14:54:18.648577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model parameters\nFACTOR = 4\nBATCH_SIZE = 8 * FACTOR\nEPOCHS = 20\nWARMUP_EPOCHS = 5\nLEARNING_RATE = 1e-4 * FACTOR\nWARMUP_LEARNING_RATE = 1e-3 * FACTOR\nHEIGHT = 224\nWIDTH = 224\nCHANNELS = 3\nES_PATIENCE = 5\nRLROP_PATIENCE = 3\nDECAY_DROP = 0.5\nLR_WARMUP_EPOCHS_1st = 2\nLR_WARMUP_EPOCHS_2nd = 5\nSTEP_SIZE = len(X_train) // BATCH_SIZE\nTOTAL_STEPS_1st = WARMUP_EPOCHS * STEP_SIZE\nTOTAL_STEPS_2nd = EPOCHS * STEP_SIZE\nWARMUP_STEPS_1st = LR_WARMUP_EPOCHS_1st * STEP_SIZE\nWARMUP_STEPS_2nd = LR_WARMUP_EPOCHS_2nd * STEP_SIZE","metadata":{"execution":{"iopub.status.busy":"2024-04-30T14:54:38.078624Z","iopub.execute_input":"2024-04-30T14:54:38.078920Z","iopub.status.idle":"2024-04-30T14:54:38.086432Z","shell.execute_reply.started":"2024-04-30T14:54:38.078877Z","shell.execute_reply":"2024-04-30T14:54:38.085463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_base_path = '../input/aptos2019-blindness-detection/train_images/'\ntest_base_path = '../input/aptos2019-blindness-detection/test_images/'\ntrain_dest_path = 'base_dir/train_images/'\nvalidation_dest_path = 'base_dir/validation_images/'\ntest_dest_path =  'base_dir/test_images/'\n\n# Making sure directories don't exist\nif os.path.exists(train_dest_path):\n    shutil.rmtree(train_dest_path)\nif os.path.exists(validation_dest_path):\n    shutil.rmtree(validation_dest_path)\nif os.path.exists(test_dest_path):\n    shutil.rmtree(test_dest_path)\n    \n# Creating train, validation and test directories\nos.makedirs(train_dest_path)\nos.makedirs(validation_dest_path)\nos.makedirs(test_dest_path)\n\ndef crop_image(img, tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n            img = np.stack([img1,img2,img3],axis=-1)\n            \n        return img\n\ndef circle_crop(img):\n    img = crop_image(img)\n\n    height, width, depth = img.shape\n    largest_side = np.max((height, width))\n    img = cv2.resize(img, (largest_side, largest_side))\n\n    height, width, depth = img.shape\n\n    x = width//2\n    y = height//2\n    r = np.amin((x, y))\n\n    circle_img = np.zeros((height, width), np.uint8)\n    cv2.circle(circle_img, (x, y), int(r), 1, thickness=-1)\n    img = cv2.bitwise_and(img, img, mask=circle_img)\n    img = crop_image(img)\n\n    return img\n    \ndef preprocess_image(base_path, save_path, image_id, HEIGHT, WIDTH, sigmaX=10):\n    image = cv2.imread(base_path + image_id)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    image = circle_crop(image)\n    image = cv2.resize(image, (HEIGHT, WIDTH))\n    image = cv2.addWeighted(image, 4, cv2.GaussianBlur(image, (0,0), sigmaX), -4 , 128)\n    cv2.imwrite(save_path + image_id, image)\n    \n# Pre-procecss train set\nfor i, image_id in enumerate(X_train['id_code']):\n    preprocess_image(train_base_path, train_dest_path, image_id, HEIGHT, WIDTH)\n    \n# Pre-procecss validation set\nfor i, image_id in enumerate(X_val['id_code']):\n    preprocess_image(train_base_path, validation_dest_path, image_id, HEIGHT, WIDTH)\n    \n# Pre-procecss test set\nfor i, image_id in enumerate(test['id_code']):\n    preprocess_image(test_base_path, test_dest_path, image_id, HEIGHT, WIDTH)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T14:54:52.278706Z","iopub.execute_input":"2024-04-30T14:54:52.279020Z","iopub.status.idle":"2024-04-30T15:13:27.909173Z","shell.execute_reply.started":"2024-04-30T14:54:52.278962Z","shell.execute_reply":"2024-04-30T15:13:27.908417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datagen=ImageDataGenerator(rescale=1./255, \n                           rotation_range=360,\n                           horizontal_flip=True,\n                           vertical_flip=True)\n\ntrain_generator=datagen.flow_from_dataframe(\n                        dataframe=X_train,\n                        directory=train_dest_path,\n                        x_col=\"id_code\",\n                        y_col=\"diagnosis\",\n                        class_mode=\"raw\",\n                        batch_size=BATCH_SIZE,\n                        target_size=(HEIGHT, WIDTH),\n                        seed=seed)\n\nvalid_generator=datagen.flow_from_dataframe(\n                        dataframe=X_val,\n                        directory=validation_dest_path,\n                        x_col=\"id_code\",\n                        y_col=\"diagnosis\",\n                        class_mode=\"raw\",\n                        batch_size=BATCH_SIZE,\n                        target_size=(HEIGHT, WIDTH),\n                        seed=seed)\n\ntest_generator=datagen.flow_from_dataframe(  \n                       dataframe=test,\n                       directory=test_dest_path,\n                       x_col=\"id_code\",\n                       batch_size=1,\n                       class_mode=None,\n                       shuffle=False,\n                       target_size=(HEIGHT, WIDTH),\n                       seed=seed)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T15:13:33.935897Z","iopub.execute_input":"2024-04-30T15:13:33.936252Z","iopub.status.idle":"2024-04-30T15:13:34.014863Z","shell.execute_reply.started":"2024-04-30T15:13:33.936195Z","shell.execute_reply":"2024-04-30T15:13:34.014119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cosine_decay_with_warmup(global_step,\n                             learning_rate_base,\n                             total_steps,\n                             warmup_learning_rate=0.0,\n                             warmup_steps=0,\n                             hold_base_rate_steps=0):\n    \"\"\"\n    Cosine decay schedule with warm up period.\n    In this schedule, the learning rate grows linearly from warmup_learning_rate\n    to learning_rate_base for warmup_steps, then transitions to a cosine decay\n    schedule.\n    :param global_step {int}: global step.\n    :param learning_rate_base {float}: base learning rate.\n    :param total_steps {int}: total number of training steps.\n    :param warmup_learning_rate {float}: initial learning rate for warm up. (default: {0.0}).\n    :param warmup_steps {int}: number of warmup steps. (default: {0}).\n    :param hold_base_rate_steps {int}: Optional number of steps to hold base learning rate before decaying. (default: {0}).\n    :param global_step {int}: global step.\n    :Returns : a float representing learning rate.\n    :Raises ValueError: if warmup_learning_rate is larger than learning_rate_base, or if warmup_steps is larger than total_steps.\n    \"\"\"\n\n    if total_steps < warmup_steps:\n        raise ValueError('total_steps must be larger or equal to warmup_steps.')\n    learning_rate = 0.5 * learning_rate_base * (1 + np.cos(\n        np.pi *\n        (global_step - warmup_steps - hold_base_rate_steps\n         ) / float(total_steps - warmup_steps - hold_base_rate_steps)))\n    if hold_base_rate_steps > 0:\n        learning_rate = np.where(global_step > warmup_steps + hold_base_rate_steps,\n                                 learning_rate, learning_rate_base)\n    if warmup_steps > 0:\n        if learning_rate_base < warmup_learning_rate:\n            raise ValueError('learning_rate_base must be larger or equal to warmup_learning_rate.')\n        slope = (learning_rate_base - warmup_learning_rate) / warmup_steps\n        warmup_rate = slope * global_step + warmup_learning_rate\n        learning_rate = np.where(global_step < warmup_steps, warmup_rate,\n                                 learning_rate)\n    return np.where(global_step > total_steps, 0.0, learning_rate)\n\n\nclass WarmUpCosineDecayScheduler(Callback):\n    \"\"\"Cosine decay with warmup learning rate scheduler\"\"\"\n\n    def __init__(self,\n                 learning_rate_base,\n                 total_steps,\n                 global_step_init=0,\n                 warmup_learning_rate=0.0,\n                 warmup_steps=0,\n                 hold_base_rate_steps=0,\n                 verbose=0):\n        \"\"\"\n        Constructor for cosine decay with warmup learning rate scheduler.\n        :param learning_rate_base {float}: base learning rate.\n        :param total_steps {int}: total number of training steps.\n        :param global_step_init {int}: initial global step, e.g. from previous checkpoint.\n        :param warmup_learning_rate {float}: initial learning rate for warm up. (default: {0.0}).\n        :param warmup_steps {int}: number of warmup steps. (default: {0}).\n        :param hold_base_rate_steps {int}: Optional number of steps to hold base learning rate before decaying. (default: {0}).\n        :param verbose {int}: quiet, 1: update messages. (default: {0}).\n        \"\"\"\n\n        super(WarmUpCosineDecayScheduler, self).__init__()\n        self.learning_rate_base = learning_rate_base\n        self.total_steps = total_steps\n        self.global_step = global_step_init\n        self.warmup_learning_rate = warmup_learning_rate\n        self.warmup_steps = warmup_steps\n        self.hold_base_rate_steps = hold_base_rate_steps\n        self.verbose = verbose\n        self.learning_rates = []\n\n    def on_batch_end(self, batch, logs=None):\n        self.global_step = self.global_step + 1\n        lr = K.get_value(self.model.optimizer.lr)\n        self.learning_rates.append(lr)\n\n    def on_batch_begin(self, batch, logs=None):\n        lr = cosine_decay_with_warmup(global_step=self.global_step,\n                                      learning_rate_base=self.learning_rate_base,\n                                      total_steps=self.total_steps,\n                                      warmup_learning_rate=self.warmup_learning_rate,\n                                      warmup_steps=self.warmup_steps,\n                                      hold_base_rate_steps=self.hold_base_rate_steps)\n        K.set_value(self.model.optimizer.lr, lr)\n        if self.verbose > 0:\n            print('\\nBatch %02d: setting learning rate to %s.' % (self.global_step + 1, lr))","metadata":{"execution":{"iopub.status.busy":"2024-04-30T15:13:36.830092Z","iopub.execute_input":"2024-04-30T15:13:36.830393Z","iopub.status.idle":"2024-04-30T15:13:36.849549Z","shell.execute_reply.started":"2024-04-30T15:13:36.830349Z","shell.execute_reply":"2024-04-30T15:13:36.848443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model(input_shape):\n    input_tensor = Input(shape=input_shape)\n    base_model = EfficientNetB5(weights=None, \n                                include_top=False,\n                                input_tensor=input_tensor)\n    base_model.load_weights('../input/efficientnet-keras-weights-b0b5/efficientnet-b5_imagenet_1000_notop.h5')\n\n    x = GlobalAveragePooling2D()(base_model.output)\n    final_output = Dense(1, activation='linear', name='final_output')(x)\n    model = Model(input_tensor, final_output)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2024-04-30T15:13:40.507934Z","iopub.execute_input":"2024-04-30T15:13:40.508286Z","iopub.status.idle":"2024-04-30T15:13:40.514694Z","shell.execute_reply.started":"2024-04-30T15:13:40.508227Z","shell.execute_reply":"2024-04-30T15:13:40.513665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = create_model(input_shape=(HEIGHT, WIDTH, CHANNELS))\n\nfor layer in model.layers:\n    layer.trainable = False\n\nfor i in range(-2, 0):\n    model.layers[i].trainable = True\n\ncosine_lr_1st = WarmUpCosineDecayScheduler(learning_rate_base=WARMUP_LEARNING_RATE,\n                                           total_steps=TOTAL_STEPS_1st,\n                                           warmup_learning_rate=0.0,\n                                           warmup_steps=WARMUP_STEPS_1st,\n                                           hold_base_rate_steps=(2 * STEP_SIZE))\n\nmetric_list = [\"accuracy\"]\ncallback_list = [cosine_lr_1st]\noptimizer = optimizers.Adam(lr=WARMUP_LEARNING_RATE)\nmodel.compile(optimizer=optimizer, loss='mean_squared_error', metrics=metric_list)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T15:13:42.864906Z","iopub.execute_input":"2024-04-30T15:13:42.865497Z","iopub.status.idle":"2024-04-30T15:14:21.407773Z","shell.execute_reply.started":"2024-04-30T15:13:42.865294Z","shell.execute_reply":"2024-04-30T15:14:21.406682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEP_SIZE_TRAIN = train_generator.n//train_generator.batch_size\nSTEP_SIZE_VALID = valid_generator.n//valid_generator.batch_size\n\nhistory_warmup = model.fit_generator(generator=train_generator,\n                                     steps_per_epoch=STEP_SIZE_TRAIN,\n                                     validation_data=valid_generator,\n                                     validation_steps=STEP_SIZE_VALID,\n                                     epochs=WARMUP_EPOCHS,\n                                     callbacks=callback_list,\n                                     verbose=2).history","metadata":{"execution":{"iopub.status.busy":"2024-04-30T15:23:52.529952Z","iopub.execute_input":"2024-04-30T15:23:52.530354Z","iopub.status.idle":"2024-04-30T15:27:22.965332Z","shell.execute_reply.started":"2024-04-30T15:23:52.530300Z","shell.execute_reply":"2024-04-30T15:27:22.964173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in model.layers:\n    layer.trainable = True\n\nes = EarlyStopping(monitor='val_loss', mode='min', patience=ES_PATIENCE, restore_best_weights=True, verbose=1)\ncosine_lr_2nd = WarmUpCosineDecayScheduler(learning_rate_base=LEARNING_RATE,\n                                           total_steps=TOTAL_STEPS_2nd,\n                                           warmup_learning_rate=0.0,\n                                           warmup_steps=WARMUP_STEPS_2nd,\n                                           hold_base_rate_steps=(3 * STEP_SIZE))\n\ncallback_list = [es, cosine_lr_2nd]\noptimizer = optimizers.Adam(lr=LEARNING_RATE)\nmodel.compile(optimizer=optimizer, loss='mean_squared_error', metrics=metric_list)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T15:31:10.299329Z","iopub.execute_input":"2024-04-30T15:31:10.299751Z","iopub.status.idle":"2024-04-30T15:31:10.544256Z","shell.execute_reply.started":"2024-04-30T15:31:10.299669Z","shell.execute_reply":"2024-04-30T15:31:10.543230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit_generator(generator=train_generator,\n                              steps_per_epoch=STEP_SIZE_TRAIN,\n                              validation_data=valid_generator,\n                              validation_steps=STEP_SIZE_VALID,\n                              epochs=EPOCHS,\n                              callbacks=callback_list,\n                              verbose=2).history","metadata":{"execution":{"iopub.status.busy":"2024-04-30T15:31:29.403417Z","iopub.execute_input":"2024-04-30T15:31:29.403722Z","iopub.status.idle":"2024-04-30T15:47:14.402427Z","shell.execute_reply.started":"2024-04-30T15:31:29.403678Z","shell.execute_reply":"2024-04-30T15:47:14.401532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(2, 1, sharex='col', figsize=(20, 6))\n\nax1.plot(cosine_lr_1st.learning_rates)\nax1.set_title('Warm up learning rates')\n\nax2.plot(cosine_lr_2nd.learning_rates)\nax2.set_title('Fine-tune learning rates')\n\nplt.xlabel('Steps')\nplt.ylabel('Learning rate')\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T15:51:13.648849Z","iopub.execute_input":"2024-04-30T15:51:13.649234Z","iopub.status.idle":"2024-04-30T15:51:14.112199Z","shell.execute_reply.started":"2024-04-30T15:51:13.649182Z","shell.execute_reply":"2024-04-30T15:51:14.111118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(2, 1, sharex='col', figsize=(20, 14))\n\nax1.plot(history['loss'], label='Train loss')\nax1.plot(history['val_loss'], label='Validation loss')\nax1.legend(loc='best')\nax1.set_title('Loss')\n\nax2.plot(history['acc'], label='Train accuracy')\nax2.plot(history['val_acc'], label='Validation accuracy')\nax2.legend(loc='best')\nax2.set_title('Accuracy')\n\nplt.xlabel('Epochs')\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T15:51:17.881100Z","iopub.execute_input":"2024-04-30T15:51:17.881497Z","iopub.status.idle":"2024-04-30T15:51:18.509429Z","shell.execute_reply.started":"2024-04-30T15:51:17.881438Z","shell.execute_reply":"2024-04-30T15:51:18.508629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create empty arays to keep the predictions and labels\ndf_preds = pd.DataFrame(columns=['label', 'pred', 'set'])\ntrain_generator.reset()\nvalid_generator.reset()\n\n# Add train predictions and labels\nfor i in range(STEP_SIZE_TRAIN + 1):\n    im, lbl = next(train_generator)\n    preds = model.predict(im, batch_size=train_generator.batch_size)\n    for index in range(len(preds)):\n        df_preds.loc[len(df_preds)] = [lbl[index], preds[index][0], 'train']\n\n# Add validation predictions and labels\nfor i in range(STEP_SIZE_VALID + 1):\n    im, lbl = next(valid_generator)\n    preds = model.predict(im, batch_size=valid_generator.batch_size)\n    for index in range(len(preds)):\n        df_preds.loc[len(df_preds)] = [lbl[index], preds[index][0], 'validation']\n\ndf_preds['label'] = df_preds['label'].astype('int')","metadata":{"execution":{"iopub.status.busy":"2024-04-30T15:51:22.095495Z","iopub.execute_input":"2024-04-30T15:51:22.095794Z","iopub.status.idle":"2024-04-30T15:52:42.002829Z","shell.execute_reply.started":"2024-04-30T15:51:22.095752Z","shell.execute_reply":"2024-04-30T15:52:42.001875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def classify(x):\n    if x < 0.5:\n        return 0\n    elif x < 1.5:\n        return 1\n    elif x < 2.5:\n        return 2\n    elif x < 3.5:\n        return 3\n    return 4\n\n# Classify predictions\ndf_preds['predictions'] = df_preds['pred'].apply(lambda x: classify(x))\n\ntrain_preds = df_preds[df_preds['set'] == 'train']\nvalidation_preds = df_preds[df_preds['set'] == 'validation']","metadata":{"execution":{"iopub.status.busy":"2024-04-30T15:54:57.165948Z","iopub.execute_input":"2024-04-30T15:54:57.166297Z","iopub.status.idle":"2024-04-30T15:54:57.183138Z","shell.execute_reply.started":"2024-04-30T15:54:57.166253Z","shell.execute_reply":"2024-04-30T15:54:57.182375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = ['0 - No DR', '1 - Mild', '2 - Moderate', '3 - Severe', '4 - Proliferative DR']\ndef plot_confusion_matrix(train, validation, labels=labels):\n    train_labels, train_preds = train\n    validation_labels, validation_preds = validation\n    fig, (ax1, ax2) = plt.subplots(1, 2, sharex='col', figsize=(24, 7))\n    train_cnf_matrix = confusion_matrix(train_labels, train_preds)\n    validation_cnf_matrix = confusion_matrix(validation_labels, validation_preds)\n\n    train_cnf_matrix_norm = train_cnf_matrix.astype('float') / train_cnf_matrix.sum(axis=1)[:, np.newaxis]\n    validation_cnf_matrix_norm = validation_cnf_matrix.astype('float') / validation_cnf_matrix.sum(axis=1)[:, np.newaxis]\n\n    train_df_cm = pd.DataFrame(train_cnf_matrix_norm, index=labels, columns=labels)\n    validation_df_cm = pd.DataFrame(validation_cnf_matrix_norm, index=labels, columns=labels)\n\n    sns.heatmap(train_df_cm, annot=True, fmt='.2f', cmap=\"Blues\",ax=ax1).set_title('Train')\n    sns.heatmap(validation_df_cm, annot=True, fmt='.2f', cmap=sns.cubehelix_palette(8),ax=ax2).set_title('Validation')\n    plt.show()\n\nplot_confusion_matrix((train_preds['label'], train_preds['predictions']), (validation_preds['label'], validation_preds['predictions']))","metadata":{"execution":{"iopub.status.busy":"2024-04-30T15:55:01.126851Z","iopub.execute_input":"2024-04-30T15:55:01.127195Z","iopub.status.idle":"2024-04-30T15:55:01.729608Z","shell.execute_reply.started":"2024-04-30T15:55:01.127135Z","shell.execute_reply":"2024-04-30T15:55:01.728407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def evaluate_model(train, validation):\n    train_labels, train_preds = train\n    validation_labels, validation_preds = validation\n    print(\"Train        Cohen Kappa score: %.3f\" % cohen_kappa_score(train_preds, train_labels, weights='quadratic'))\n    print(\"Validation   Cohen Kappa score: %.3f\" % cohen_kappa_score(validation_preds, validation_labels, weights='quadratic'))\n    print(\"Complete set Cohen Kappa score: %.3f\" % cohen_kappa_score(np.append(train_preds, validation_preds), np.append(train_labels, validation_labels), weights='quadratic'))\n    \nevaluate_model((train_preds['label'], train_preds['predictions']), (validation_preds['label'], validation_preds['predictions']))","metadata":{"execution":{"iopub.status.busy":"2024-04-30T15:55:07.680208Z","iopub.execute_input":"2024-04-30T15:55:07.680551Z","iopub.status.idle":"2024-04-30T15:55:07.705104Z","shell.execute_reply.started":"2024-04-30T15:55:07.680475Z","shell.execute_reply":"2024-04-30T15:55:07.704402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def apply_tta(model, generator, steps=10):\n    step_size = generator.n//generator.batch_size\n    preds_tta = []\n    for i in range(steps):\n        generator.reset()\n        preds = model.predict_generator(generator, steps=step_size)\n        preds_tta.append(preds)\n\n    return np.mean(preds_tta, axis=0)\n\npreds = apply_tta(model, test_generator)\npredictions = [classify(x) for x in preds]\n\nresults = pd.DataFrame({'id_code':test['id_code'], 'diagnosis':predictions})\nresults['id_code'] = results['id_code'].map(lambda x: str(x)[:-4])","metadata":{"execution":{"iopub.status.busy":"2024-04-30T15:55:10.571966Z","iopub.execute_input":"2024-04-30T15:55:10.572319Z","iopub.status.idle":"2024-04-30T16:07:18.813020Z","shell.execute_reply.started":"2024-04-30T15:55:10.572258Z","shell.execute_reply":"2024-04-30T16:07:18.812356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cleaning created directories\nif os.path.exists(train_dest_path):\n    shutil.rmtree(train_dest_path)\nif os.path.exists(validation_dest_path):\n    shutil.rmtree(validation_dest_path)\nif os.path.exists(test_dest_path):\n    shutil.rmtree(test_dest_path)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T16:12:47.523613Z","iopub.execute_input":"2024-04-30T16:12:47.523926Z","iopub.status.idle":"2024-04-30T16:12:47.787494Z","shell.execute_reply.started":"2024-04-30T16:12:47.523880Z","shell.execute_reply":"2024-04-30T16:12:47.786868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.subplots(sharex='col', figsize=(24, 8.7))\nsns.countplot(x=\"diagnosis\", data=results).set_title('Test')\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T16:12:49.692719Z","iopub.execute_input":"2024-04-30T16:12:49.692998Z","iopub.status.idle":"2024-04-30T16:12:49.918614Z","shell.execute_reply.started":"2024-04-30T16:12:49.692961Z","shell.execute_reply":"2024-04-30T16:12:49.917696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results.to_csv('submission.csv', index=False)\ndisplay(results.head())","metadata":{"execution":{"iopub.status.busy":"2024-04-30T16:12:52.494662Z","iopub.execute_input":"2024-04-30T16:12:52.494986Z","iopub.status.idle":"2024-04-30T16:12:52.517281Z","shell.execute_reply.started":"2024-04-30T16:12:52.494939Z","shell.execute_reply":"2024-04-30T16:12:52.516289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save_weights('../working/effNetB5_bs32_img224_fold3.h5')","metadata":{"execution":{"iopub.status.busy":"2024-04-30T16:12:56.472810Z","iopub.execute_input":"2024-04-30T16:12:56.473137Z","iopub.status.idle":"2024-04-30T16:22:02.797622Z","shell.execute_reply.started":"2024-04-30T16:12:56.473086Z","shell.execute_reply":"2024-04-30T16:22:02.796674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"FOLD_4","metadata":{}},{"cell_type":"code","source":"import os\nimport sys\nimport cv2\nimport shutil\nimport random\nimport warnings\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom tensorflow import set_random_seed\nfrom sklearn.utils import class_weight\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, cohen_kappa_score\nfrom keras import backend as K\nfrom keras.models import Model\nfrom keras.utils import to_categorical\nfrom keras import optimizers, applications\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.layers import Dense, Dropout, GlobalAveragePooling2D, Input\nfrom keras.callbacks import EarlyStopping, ReduceLROnPlateau, Callback, LearningRateScheduler\n\ndef seed_everything(seed=0):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    set_random_seed(0)\n\nseed = 0\nseed_everything(seed)\n%matplotlib inline\nsns.set(style=\"whitegrid\")\nwarnings.filterwarnings(\"ignore\")\nsys.path.append(os.path.abspath('../input/efficientnet/efficientnet-master/efficientnet-master/'))\nfrom efficientnet import *","metadata":{"execution":{"iopub.status.busy":"2024-04-30T23:58:42.487094Z","iopub.execute_input":"2024-04-30T23:58:42.487371Z","iopub.status.idle":"2024-04-30T23:58:47.477419Z","shell.execute_reply.started":"2024-04-30T23:58:42.487329Z","shell.execute_reply":"2024-04-30T23:58:47.476732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fold_set = pd.read_csv('/kaggle/input/5-fold/5-fold.csv')\nX_train = fold_set[fold_set['fold_3'] == 'train']\nX_val = fold_set[fold_set['fold_3'] == 'validation']\ntest = pd.read_csv('../input/aptos2019-blindness-detection/test.csv')\nprint('Number of train samples: ', X_train.shape[0])\nprint('Number of validation samples: ', X_val.shape[0])\nprint('Number of test samples: ', test.shape[0])\n\n# Preprocecss data\nX_train[\"id_code\"] = X_train[\"id_code\"].apply(lambda x: x + \".png\")\nX_val[\"id_code\"] = X_val[\"id_code\"].apply(lambda x: x + \".png\")\ntest[\"id_code\"] = test[\"id_code\"].apply(lambda x: x + \".png\")\ndisplay(X_train.head())","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:02:05.375367Z","iopub.execute_input":"2024-05-01T00:02:05.375729Z","iopub.status.idle":"2024-05-01T00:02:05.616832Z","shell.execute_reply.started":"2024-05-01T00:02:05.375672Z","shell.execute_reply":"2024-05-01T00:02:05.616115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model parameters\nFACTOR = 4\nBATCH_SIZE = 8 * FACTOR\nEPOCHS = 20\nWARMUP_EPOCHS = 5\nLEARNING_RATE = 1e-4 * FACTOR\nWARMUP_LEARNING_RATE = 1e-3 * FACTOR\nHEIGHT = 224\nWIDTH = 224\nCHANNELS = 3\nES_PATIENCE = 5\nRLROP_PATIENCE = 3\nDECAY_DROP = 0.5\nLR_WARMUP_EPOCHS_1st = 2\nLR_WARMUP_EPOCHS_2nd = 5\nSTEP_SIZE = len(X_train) // BATCH_SIZE\nTOTAL_STEPS_1st = WARMUP_EPOCHS * STEP_SIZE\nTOTAL_STEPS_2nd = EPOCHS * STEP_SIZE\nWARMUP_STEPS_1st = LR_WARMUP_EPOCHS_1st * STEP_SIZE\nWARMUP_STEPS_2nd = LR_WARMUP_EPOCHS_2nd * STEP_SIZE","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:02:07.581429Z","iopub.execute_input":"2024-05-01T00:02:07.581721Z","iopub.status.idle":"2024-05-01T00:02:07.588803Z","shell.execute_reply.started":"2024-05-01T00:02:07.581678Z","shell.execute_reply":"2024-05-01T00:02:07.587910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_base_path = '../input/aptos2019-blindness-detection/train_images/'\ntest_base_path = '../input/aptos2019-blindness-detection/test_images/'\ntrain_dest_path = 'base_dir/train_images/'\nvalidation_dest_path = 'base_dir/validation_images/'\ntest_dest_path =  'base_dir/test_images/'\n\n# Making sure directories don't exist\nif os.path.exists(train_dest_path):\n    shutil.rmtree(train_dest_path)\nif os.path.exists(validation_dest_path):\n    shutil.rmtree(validation_dest_path)\nif os.path.exists(test_dest_path):\n    shutil.rmtree(test_dest_path)\n    \n# Creating train, validation and test directories\nos.makedirs(train_dest_path)\nos.makedirs(validation_dest_path)\nos.makedirs(test_dest_path)\n\ndef crop_image(img, tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n            img = np.stack([img1,img2,img3],axis=-1)\n            \n        return img\n\ndef circle_crop(img):\n    img = crop_image(img)\n\n    height, width, depth = img.shape\n    largest_side = np.max((height, width))\n    img = cv2.resize(img, (largest_side, largest_side))\n\n    height, width, depth = img.shape\n\n    x = width//2\n    y = height//2\n    r = np.amin((x, y))\n\n    circle_img = np.zeros((height, width), np.uint8)\n    cv2.circle(circle_img, (x, y), int(r), 1, thickness=-1)\n    img = cv2.bitwise_and(img, img, mask=circle_img)\n    img = crop_image(img)\n\n    return img\n    \ndef preprocess_image(base_path, save_path, image_id, HEIGHT, WIDTH, sigmaX=10):\n    image = cv2.imread(base_path + image_id)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    image = circle_crop(image)\n    image = cv2.resize(image, (HEIGHT, WIDTH))\n    image = cv2.addWeighted(image, 4, cv2.GaussianBlur(image, (0,0), sigmaX), -4 , 128)\n    cv2.imwrite(save_path + image_id, image)\n    \n# Pre-procecss train set\nfor i, image_id in enumerate(X_train['id_code']):\n    preprocess_image(train_base_path, train_dest_path, image_id, HEIGHT, WIDTH)\n    \n# Pre-procecss validation set\nfor i, image_id in enumerate(X_val['id_code']):\n    preprocess_image(train_base_path, validation_dest_path, image_id, HEIGHT, WIDTH)\n    \n# Pre-procecss test set\nfor i, image_id in enumerate(test['id_code']):\n    preprocess_image(test_base_path, test_dest_path, image_id, HEIGHT, WIDTH)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:02:09.958098Z","iopub.execute_input":"2024-05-01T00:02:09.958586Z","iopub.status.idle":"2024-05-01T00:24:54.968200Z","shell.execute_reply.started":"2024-05-01T00:02:09.958360Z","shell.execute_reply":"2024-05-01T00:24:54.967342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datagen=ImageDataGenerator(rescale=1./255, \n                           rotation_range=360,\n                           horizontal_flip=True,\n                           vertical_flip=True)\n\ntrain_generator=datagen.flow_from_dataframe(\n                        dataframe=X_train,\n                        directory=train_dest_path,\n                        x_col=\"id_code\",\n                        y_col=\"diagnosis\",\n                        class_mode=\"raw\",\n                        batch_size=BATCH_SIZE,\n                        target_size=(HEIGHT, WIDTH),\n                        seed=seed)\n\nvalid_generator=datagen.flow_from_dataframe(\n                        dataframe=X_val,\n                        directory=validation_dest_path,\n                        x_col=\"id_code\",\n                        y_col=\"diagnosis\",\n                        class_mode=\"raw\",\n                        batch_size=BATCH_SIZE,\n                        target_size=(HEIGHT, WIDTH),\n                        seed=seed)\n\ntest_generator=datagen.flow_from_dataframe(  \n                       dataframe=test,\n                       directory=test_dest_path,\n                       x_col=\"id_code\",\n                       batch_size=1,\n                       class_mode=None,\n                       shuffle=False,\n                       target_size=(HEIGHT, WIDTH),\n                       seed=seed)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:36:46.017266Z","iopub.execute_input":"2024-05-01T00:36:46.017587Z","iopub.status.idle":"2024-05-01T00:36:46.099816Z","shell.execute_reply.started":"2024-05-01T00:36:46.017545Z","shell.execute_reply":"2024-05-01T00:36:46.098879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cosine_decay_with_warmup(global_step,\n                             learning_rate_base,\n                             total_steps,\n                             warmup_learning_rate=0.0,\n                             warmup_steps=0,\n                             hold_base_rate_steps=0):\n    \"\"\"\n    Cosine decay schedule with warm up period.\n    In this schedule, the learning rate grows linearly from warmup_learning_rate\n    to learning_rate_base for warmup_steps, then transitions to a cosine decay\n    schedule.\n    :param global_step {int}: global step.\n    :param learning_rate_base {float}: base learning rate.\n    :param total_steps {int}: total number of training steps.\n    :param warmup_learning_rate {float}: initial learning rate for warm up. (default: {0.0}).\n    :param warmup_steps {int}: number of warmup steps. (default: {0}).\n    :param hold_base_rate_steps {int}: Optional number of steps to hold base learning rate before decaying. (default: {0}).\n    :param global_step {int}: global step.\n    :Returns : a float representing learning rate.\n    :Raises ValueError: if warmup_learning_rate is larger than learning_rate_base, or if warmup_steps is larger than total_steps.\n    \"\"\"\n\n    if total_steps < warmup_steps:\n        raise ValueError('total_steps must be larger or equal to warmup_steps.')\n    learning_rate = 0.5 * learning_rate_base * (1 + np.cos(\n        np.pi *\n        (global_step - warmup_steps - hold_base_rate_steps\n         ) / float(total_steps - warmup_steps - hold_base_rate_steps)))\n    if hold_base_rate_steps > 0:\n        learning_rate = np.where(global_step > warmup_steps + hold_base_rate_steps,\n                                 learning_rate, learning_rate_base)\n    if warmup_steps > 0:\n        if learning_rate_base < warmup_learning_rate:\n            raise ValueError('learning_rate_base must be larger or equal to warmup_learning_rate.')\n        slope = (learning_rate_base - warmup_learning_rate) / warmup_steps\n        warmup_rate = slope * global_step + warmup_learning_rate\n        learning_rate = np.where(global_step < warmup_steps, warmup_rate,\n                                 learning_rate)\n    return np.where(global_step > total_steps, 0.0, learning_rate)\n\n\nclass WarmUpCosineDecayScheduler(Callback):\n    \"\"\"Cosine decay with warmup learning rate scheduler\"\"\"\n\n    def __init__(self,\n                 learning_rate_base,\n                 total_steps,\n                 global_step_init=0,\n                 warmup_learning_rate=0.0,\n                 warmup_steps=0,\n                 hold_base_rate_steps=0,\n                 verbose=0):\n        \"\"\"\n        Constructor for cosine decay with warmup learning rate scheduler.\n        :param learning_rate_base {float}: base learning rate.\n        :param total_steps {int}: total number of training steps.\n        :param global_step_init {int}: initial global step, e.g. from previous checkpoint.\n        :param warmup_learning_rate {float}: initial learning rate for warm up. (default: {0.0}).\n        :param warmup_steps {int}: number of warmup steps. (default: {0}).\n        :param hold_base_rate_steps {int}: Optional number of steps to hold base learning rate before decaying. (default: {0}).\n        :param verbose {int}: quiet, 1: update messages. (default: {0}).\n        \"\"\"\n\n        super(WarmUpCosineDecayScheduler, self).__init__()\n        self.learning_rate_base = learning_rate_base\n        self.total_steps = total_steps\n        self.global_step = global_step_init\n        self.warmup_learning_rate = warmup_learning_rate\n        self.warmup_steps = warmup_steps\n        self.hold_base_rate_steps = hold_base_rate_steps\n        self.verbose = verbose\n        self.learning_rates = []\n\n    def on_batch_end(self, batch, logs=None):\n        self.global_step = self.global_step + 1\n        lr = K.get_value(self.model.optimizer.lr)\n        self.learning_rates.append(lr)\n\n    def on_batch_begin(self, batch, logs=None):\n        lr = cosine_decay_with_warmup(global_step=self.global_step,\n                                      learning_rate_base=self.learning_rate_base,\n                                      total_steps=self.total_steps,\n                                      warmup_learning_rate=self.warmup_learning_rate,\n                                      warmup_steps=self.warmup_steps,\n                                      hold_base_rate_steps=self.hold_base_rate_steps)\n        K.set_value(self.model.optimizer.lr, lr)\n        if self.verbose > 0:\n            print('\\nBatch %02d: setting learning rate to %s.' % (self.global_step + 1, lr))","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:36:49.142869Z","iopub.execute_input":"2024-05-01T00:36:49.143148Z","iopub.status.idle":"2024-05-01T00:36:49.162765Z","shell.execute_reply.started":"2024-05-01T00:36:49.143107Z","shell.execute_reply":"2024-05-01T00:36:49.161813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model(input_shape):\n    input_tensor = Input(shape=input_shape)\n    base_model = EfficientNetB5(weights=None, \n                                include_top=False,\n                                input_tensor=input_tensor)\n    base_model.load_weights('../input/efficientnet-keras-weights-b0b5/efficientnet-b5_imagenet_1000_notop.h5')\n\n    x = GlobalAveragePooling2D()(base_model.output)\n    final_output = Dense(1, activation='linear', name='final_output')(x)\n    model = Model(input_tensor, final_output)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:36:52.839316Z","iopub.execute_input":"2024-05-01T00:36:52.839621Z","iopub.status.idle":"2024-05-01T00:36:52.845451Z","shell.execute_reply.started":"2024-05-01T00:36:52.839578Z","shell.execute_reply":"2024-05-01T00:36:52.844711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = create_model(input_shape=(HEIGHT, WIDTH, CHANNELS))\n\nfor layer in model.layers:\n    layer.trainable = False\n\nfor i in range(-2, 0):\n    model.layers[i].trainable = True\n\ncosine_lr_1st = WarmUpCosineDecayScheduler(learning_rate_base=WARMUP_LEARNING_RATE,\n                                           total_steps=TOTAL_STEPS_1st,\n                                           warmup_learning_rate=0.0,\n                                           warmup_steps=WARMUP_STEPS_1st,\n                                           hold_base_rate_steps=(2 * STEP_SIZE))\n\nmetric_list = [\"accuracy\"]\ncallback_list = [cosine_lr_1st]\noptimizer = optimizers.Adam(lr=WARMUP_LEARNING_RATE)\nmodel.compile(optimizer=optimizer, loss='mean_squared_error', metrics=metric_list)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:36:55.306584Z","iopub.execute_input":"2024-05-01T00:36:55.306959Z","iopub.status.idle":"2024-05-01T00:37:23.067592Z","shell.execute_reply.started":"2024-05-01T00:36:55.306898Z","shell.execute_reply":"2024-05-01T00:37:23.066567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEP_SIZE_TRAIN = train_generator.n//train_generator.batch_size\nSTEP_SIZE_VALID = valid_generator.n//valid_generator.batch_size\n\nhistory_warmup = model.fit_generator(generator=train_generator,\n                                     steps_per_epoch=STEP_SIZE_TRAIN,\n                                     validation_data=valid_generator,\n                                     validation_steps=STEP_SIZE_VALID,\n                                     epochs=WARMUP_EPOCHS,\n                                     callbacks=callback_list,\n                                     verbose=2).history","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:37:37.426345Z","iopub.execute_input":"2024-05-01T00:37:37.426698Z","iopub.status.idle":"2024-05-01T00:40:53.862884Z","shell.execute_reply.started":"2024-05-01T00:37:37.426647Z","shell.execute_reply":"2024-05-01T00:40:53.862070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in model.layers:\n    layer.trainable = True\n\nes = EarlyStopping(monitor='val_loss', mode='min', patience=ES_PATIENCE, restore_best_weights=True, verbose=1)\ncosine_lr_2nd = WarmUpCosineDecayScheduler(learning_rate_base=LEARNING_RATE,\n                                           total_steps=TOTAL_STEPS_2nd,\n                                           warmup_learning_rate=0.0,\n                                           warmup_steps=WARMUP_STEPS_2nd,\n                                           hold_base_rate_steps=(3 * STEP_SIZE))\n\ncallback_list = [es, cosine_lr_2nd]\noptimizer = optimizers.Adam(lr=LEARNING_RATE)\nmodel.compile(optimizer=optimizer, loss='mean_squared_error', metrics=metric_list)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:41:12.833636Z","iopub.execute_input":"2024-05-01T00:41:12.833937Z","iopub.status.idle":"2024-05-01T00:41:13.064294Z","shell.execute_reply.started":"2024-05-01T00:41:12.833893Z","shell.execute_reply":"2024-05-01T00:41:13.063527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit_generator(generator=train_generator,\n                              steps_per_epoch=STEP_SIZE_TRAIN,\n                              validation_data=valid_generator,\n                              validation_steps=STEP_SIZE_VALID,\n                              epochs=EPOCHS,\n                              callbacks=callback_list,\n                              verbose=2).history","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:41:27.129633Z","iopub.execute_input":"2024-05-01T00:41:27.130183Z","iopub.status.idle":"2024-05-01T01:07:06.228585Z","shell.execute_reply.started":"2024-05-01T00:41:27.129988Z","shell.execute_reply":"2024-05-01T01:07:06.227218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(2, 1, sharex='col', figsize=(20, 6))\n\nax1.plot(cosine_lr_1st.learning_rates)\nax1.set_title('Warm up learning rates')\n\nax2.plot(cosine_lr_2nd.learning_rates)\nax2.set_title('Fine-tune learning rates')\n\nplt.xlabel('Steps')\nplt.ylabel('Learning rate')\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-01T01:12:51.476160Z","iopub.execute_input":"2024-05-01T01:12:51.476543Z","iopub.status.idle":"2024-05-01T01:12:51.876821Z","shell.execute_reply.started":"2024-05-01T01:12:51.476476Z","shell.execute_reply":"2024-05-01T01:12:51.876078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(2, 1, sharex='col', figsize=(20, 14))\n\nax1.plot(history['loss'], label='Train loss')\nax1.plot(history['val_loss'], label='Validation loss')\nax1.legend(loc='best')\nax1.set_title('Loss')\n\nax2.plot(history['acc'], label='Train accuracy')\nax2.plot(history['val_acc'], label='Validation accuracy')\nax2.legend(loc='best')\nax2.set_title('Accuracy')\n\nplt.xlabel('Epochs')\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-01T01:12:54.298926Z","iopub.execute_input":"2024-05-01T01:12:54.299391Z","iopub.status.idle":"2024-05-01T01:12:54.933117Z","shell.execute_reply.started":"2024-05-01T01:12:54.299303Z","shell.execute_reply":"2024-05-01T01:12:54.932201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create empty arays to keep the predictions and labels\ndf_preds = pd.DataFrame(columns=['label', 'pred', 'set'])\ntrain_generator.reset()\nvalid_generator.reset()\n\n# Add train predictions and labels\nfor i in range(STEP_SIZE_TRAIN + 1):\n    im, lbl = next(train_generator)\n    preds = model.predict(im, batch_size=train_generator.batch_size)\n    for index in range(len(preds)):\n        df_preds.loc[len(df_preds)] = [lbl[index], preds[index][0], 'train']\n\n# Add validation predictions and labels\nfor i in range(STEP_SIZE_VALID + 1):\n    im, lbl = next(valid_generator)\n    preds = model.predict(im, batch_size=valid_generator.batch_size)\n    for index in range(len(preds)):\n        df_preds.loc[len(df_preds)] = [lbl[index], preds[index][0], 'validation']\n\ndf_preds['label'] = df_preds['label'].astype('int')","metadata":{"execution":{"iopub.status.busy":"2024-05-01T01:12:57.442258Z","iopub.execute_input":"2024-05-01T01:12:57.442591Z","iopub.status.idle":"2024-05-01T01:14:11.099648Z","shell.execute_reply.started":"2024-05-01T01:12:57.442541Z","shell.execute_reply":"2024-05-01T01:14:11.098990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def classify(x):\n    if x < 0.5:\n        return 0\n    elif x < 1.5:\n        return 1\n    elif x < 2.5:\n        return 2\n    elif x < 3.5:\n        return 3\n    return 4\n\n# Classify predictions\ndf_preds['predictions'] = df_preds['pred'].apply(lambda x: classify(x))\n\ntrain_preds = df_preds[df_preds['set'] == 'train']\nvalidation_preds = df_preds[df_preds['set'] == 'validation']","metadata":{"execution":{"iopub.status.busy":"2024-05-01T01:19:43.040021Z","iopub.execute_input":"2024-05-01T01:19:43.040348Z","iopub.status.idle":"2024-05-01T01:19:43.058095Z","shell.execute_reply.started":"2024-05-01T01:19:43.040304Z","shell.execute_reply":"2024-05-01T01:19:43.057415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = ['0 - No DR', '1 - Mild', '2 - Moderate', '3 - Severe', '4 - Proliferative DR']\ndef plot_confusion_matrix(train, validation, labels=labels):\n    train_labels, train_preds = train\n    validation_labels, validation_preds = validation\n    fig, (ax1, ax2) = plt.subplots(1, 2, sharex='col', figsize=(24, 7))\n    train_cnf_matrix = confusion_matrix(train_labels, train_preds)\n    validation_cnf_matrix = confusion_matrix(validation_labels, validation_preds)\n\n    train_cnf_matrix_norm = train_cnf_matrix.astype('float') / train_cnf_matrix.sum(axis=1)[:, np.newaxis]\n    validation_cnf_matrix_norm = validation_cnf_matrix.astype('float') / validation_cnf_matrix.sum(axis=1)[:, np.newaxis]\n\n    train_df_cm = pd.DataFrame(train_cnf_matrix_norm, index=labels, columns=labels)\n    validation_df_cm = pd.DataFrame(validation_cnf_matrix_norm, index=labels, columns=labels)\n\n    sns.heatmap(train_df_cm, annot=True, fmt='.2f', cmap=\"Blues\",ax=ax1).set_title('Train')\n    sns.heatmap(validation_df_cm, annot=True, fmt='.2f', cmap=sns.cubehelix_palette(8),ax=ax2).set_title('Validation')\n    plt.show()\n\nplot_confusion_matrix((train_preds['label'], train_preds['predictions']), (validation_preds['label'], validation_preds['predictions']))","metadata":{"execution":{"iopub.status.busy":"2024-05-01T01:19:46.319247Z","iopub.execute_input":"2024-05-01T01:19:46.319616Z","iopub.status.idle":"2024-05-01T01:19:46.952720Z","shell.execute_reply.started":"2024-05-01T01:19:46.319562Z","shell.execute_reply":"2024-05-01T01:19:46.951768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def evaluate_model(train, validation):\n    train_labels, train_preds = train\n    validation_labels, validation_preds = validation\n    print(\"Train        Cohen Kappa score: %.3f\" % cohen_kappa_score(train_preds, train_labels, weights='quadratic'))\n    print(\"Validation   Cohen Kappa score: %.3f\" % cohen_kappa_score(validation_preds, validation_labels, weights='quadratic'))\n    print(\"Complete set Cohen Kappa score: %.3f\" % cohen_kappa_score(np.append(train_preds, validation_preds), np.append(train_labels, validation_labels), weights='quadratic'))\n    \nevaluate_model((train_preds['label'], train_preds['predictions']), (validation_preds['label'], validation_preds['predictions']))","metadata":{"execution":{"iopub.status.busy":"2024-05-01T01:19:50.704566Z","iopub.execute_input":"2024-05-01T01:19:50.704919Z","iopub.status.idle":"2024-05-01T01:19:50.728722Z","shell.execute_reply.started":"2024-05-01T01:19:50.704871Z","shell.execute_reply":"2024-05-01T01:19:50.727995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def apply_tta(model, generator, steps=10):\n    step_size = generator.n//generator.batch_size\n    preds_tta = []\n    for i in range(steps):\n        generator.reset()\n        preds = model.predict_generator(generator, steps=step_size)\n        preds_tta.append(preds)\n\n    return np.mean(preds_tta, axis=0)\n\npreds = apply_tta(model, test_generator)\npredictions = [classify(x) for x in preds]\n\nresults = pd.DataFrame({'id_code':test['id_code'], 'diagnosis':predictions})\nresults['id_code'] = results['id_code'].map(lambda x: str(x)[:-4])","metadata":{"execution":{"iopub.status.busy":"2024-05-01T01:19:52.556213Z","iopub.execute_input":"2024-05-01T01:19:52.556535Z","iopub.status.idle":"2024-05-01T01:30:31.895283Z","shell.execute_reply.started":"2024-05-01T01:19:52.556492Z","shell.execute_reply":"2024-05-01T01:30:31.894574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cleaning created directories\nif os.path.exists(train_dest_path):\n    shutil.rmtree(train_dest_path)\nif os.path.exists(validation_dest_path):\n    shutil.rmtree(validation_dest_path)\nif os.path.exists(test_dest_path):\n    shutil.rmtree(test_dest_path)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T01:37:32.354468Z","iopub.execute_input":"2024-05-01T01:37:32.354830Z","iopub.status.idle":"2024-05-01T01:37:32.619021Z","shell.execute_reply.started":"2024-05-01T01:37:32.354769Z","shell.execute_reply":"2024-05-01T01:37:32.618388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.subplots(sharex='col', figsize=(24, 8.7))\nsns.countplot(x=\"diagnosis\", data=results).set_title('Test')\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-01T01:37:34.342638Z","iopub.execute_input":"2024-05-01T01:37:34.343063Z","iopub.status.idle":"2024-05-01T01:37:34.708422Z","shell.execute_reply.started":"2024-05-01T01:37:34.342993Z","shell.execute_reply":"2024-05-01T01:37:34.707544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results.to_csv('submission.csv', index=False)\ndisplay(results.head())","metadata":{"execution":{"iopub.status.busy":"2024-05-01T01:37:36.908815Z","iopub.execute_input":"2024-05-01T01:37:36.909111Z","iopub.status.idle":"2024-05-01T01:37:37.068116Z","shell.execute_reply.started":"2024-05-01T01:37:36.909068Z","shell.execute_reply":"2024-05-01T01:37:37.067218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save_weights('../working/effNetB5_bs32_img224_fold4.h5')","metadata":{"execution":{"iopub.status.busy":"2024-05-01T01:37:39.069282Z","iopub.execute_input":"2024-05-01T01:37:39.069645Z","iopub.status.idle":"2024-05-01T01:40:18.587190Z","shell.execute_reply.started":"2024-05-01T01:37:39.069582Z","shell.execute_reply":"2024-05-01T01:40:18.586201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"FOLD_5","metadata":{}},{"cell_type":"code","source":"import os\nimport sys\nimport cv2\nimport shutil\nimport random\nimport warnings\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom tensorflow import set_random_seed\nfrom sklearn.utils import class_weight\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, cohen_kappa_score\nfrom keras import backend as K\nfrom keras.models import Model\nfrom keras.utils import to_categorical\nfrom keras import optimizers, applications\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.layers import Dense, Dropout, GlobalAveragePooling2D, Input\nfrom keras.callbacks import EarlyStopping, ReduceLROnPlateau, Callback, LearningRateScheduler\n\ndef seed_everything(seed=0):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    set_random_seed(0)\n\nseed = 0\nseed_everything(seed)\n%matplotlib inline\nsns.set(style=\"whitegrid\")\nwarnings.filterwarnings(\"ignore\")\nsys.path.append(os.path.abspath('../input/efficientnet/efficientnet-master/efficientnet-master/'))\nfrom efficientnet import *","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fold_set = pd.read_csv('/kaggle/input/5-fold/5-fold.csv')\nX_train = fold_set[fold_set['fold_4'] == 'train']\nX_val = fold_set[fold_set['fold_4'] == 'validation']\ntest = pd.read_csv('../input/aptos2019-blindness-detection/test.csv')\nprint('Number of train samples: ', X_train.shape[0])\nprint('Number of validation samples: ', X_val.shape[0])\nprint('Number of test samples: ', test.shape[0])\n\n# Preprocecss data\nX_train[\"id_code\"] = X_train[\"id_code\"].apply(lambda x: x + \".png\")\nX_val[\"id_code\"] = X_val[\"id_code\"].apply(lambda x: x + \".png\")\ntest[\"id_code\"] = test[\"id_code\"].apply(lambda x: x + \".png\")\ndisplay(X_train.head())","metadata":{"execution":{"iopub.status.busy":"2024-05-01T01:43:58.230989Z","iopub.execute_input":"2024-05-01T01:43:58.231322Z","iopub.status.idle":"2024-05-01T01:43:59.210807Z","shell.execute_reply.started":"2024-05-01T01:43:58.231275Z","shell.execute_reply":"2024-05-01T01:43:59.210053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model parameters\nFACTOR = 4\nBATCH_SIZE = 8 * FACTOR\nEPOCHS = 20\nWARMUP_EPOCHS = 5\nLEARNING_RATE = 1e-4 * FACTOR\nWARMUP_LEARNING_RATE = 1e-3 * FACTOR\nHEIGHT = 224\nWIDTH = 224\nCHANNELS = 3\nES_PATIENCE = 5\nRLROP_PATIENCE = 3\nDECAY_DROP = 0.5\nLR_WARMUP_EPOCHS_1st = 2\nLR_WARMUP_EPOCHS_2nd = 5\nSTEP_SIZE = len(X_train) // BATCH_SIZE\nTOTAL_STEPS_1st = WARMUP_EPOCHS * STEP_SIZE\nTOTAL_STEPS_2nd = EPOCHS * STEP_SIZE\nWARMUP_STEPS_1st = LR_WARMUP_EPOCHS_1st * STEP_SIZE\nWARMUP_STEPS_2nd = LR_WARMUP_EPOCHS_2nd * STEP_SIZE","metadata":{"execution":{"iopub.status.busy":"2024-05-01T01:44:00.919821Z","iopub.execute_input":"2024-05-01T01:44:00.920119Z","iopub.status.idle":"2024-05-01T01:44:00.927784Z","shell.execute_reply.started":"2024-05-01T01:44:00.920077Z","shell.execute_reply":"2024-05-01T01:44:00.926765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_base_path = '../input/aptos2019-blindness-detection/train_images/'\ntest_base_path = '../input/aptos2019-blindness-detection/test_images/'\ntrain_dest_path = 'base_dir/train_images/'\nvalidation_dest_path = 'base_dir/validation_images/'\ntest_dest_path =  'base_dir/test_images/'\n\n# Making sure directories don't exist\nif os.path.exists(train_dest_path):\n    shutil.rmtree(train_dest_path)\nif os.path.exists(validation_dest_path):\n    shutil.rmtree(validation_dest_path)\nif os.path.exists(test_dest_path):\n    shutil.rmtree(test_dest_path)\n    \n# Creating train, validation and test directories\nos.makedirs(train_dest_path)\nos.makedirs(validation_dest_path)\nos.makedirs(test_dest_path)\n\ndef crop_image(img, tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n            img = np.stack([img1,img2,img3],axis=-1)\n            \n        return img\n\ndef circle_crop(img):\n    img = crop_image(img)\n\n    height, width, depth = img.shape\n    largest_side = np.max((height, width))\n    img = cv2.resize(img, (largest_side, largest_side))\n\n    height, width, depth = img.shape\n\n    x = width//2\n    y = height//2\n    r = np.amin((x, y))\n\n    circle_img = np.zeros((height, width), np.uint8)\n    cv2.circle(circle_img, (x, y), int(r), 1, thickness=-1)\n    img = cv2.bitwise_and(img, img, mask=circle_img)\n    img = crop_image(img)\n\n    return img\n    \ndef preprocess_image(base_path, save_path, image_id, HEIGHT, WIDTH, sigmaX=10):\n    image = cv2.imread(base_path + image_id)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    image = circle_crop(image)\n    image = cv2.resize(image, (HEIGHT, WIDTH))\n    image = cv2.addWeighted(image, 4, cv2.GaussianBlur(image, (0,0), sigmaX), -4 , 128)\n    cv2.imwrite(save_path + image_id, image)\n    \n# Pre-procecss train set\nfor i, image_id in enumerate(X_train['id_code']):\n    preprocess_image(train_base_path, train_dest_path, image_id, HEIGHT, WIDTH)\n    \n# Pre-procecss validation set\nfor i, image_id in enumerate(X_val['id_code']):\n    preprocess_image(train_base_path, validation_dest_path, image_id, HEIGHT, WIDTH)\n    \n# Pre-procecss test set\nfor i, image_id in enumerate(test['id_code']):\n    preprocess_image(test_base_path, test_dest_path, image_id, HEIGHT, WIDTH)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T01:44:02.664542Z","iopub.execute_input":"2024-05-01T01:44:02.664847Z","iopub.status.idle":"2024-05-01T02:02:22.194964Z","shell.execute_reply.started":"2024-05-01T01:44:02.664804Z","shell.execute_reply":"2024-05-01T02:02:22.194233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datagen=ImageDataGenerator(rescale=1./255, \n                           rotation_range=360,\n                           horizontal_flip=True,\n                           vertical_flip=True)\n\ntrain_generator=datagen.flow_from_dataframe(\n                        dataframe=X_train,\n                        directory=train_dest_path,\n                        x_col=\"id_code\",\n                        y_col=\"diagnosis\",\n                        class_mode=\"raw\",\n                        batch_size=BATCH_SIZE,\n                        target_size=(HEIGHT, WIDTH),\n                        seed=seed)\n\nvalid_generator=datagen.flow_from_dataframe(\n                        dataframe=X_val,\n                        directory=validation_dest_path,\n                        x_col=\"id_code\",\n                        y_col=\"diagnosis\",\n                        class_mode=\"raw\",\n                        batch_size=BATCH_SIZE,\n                        target_size=(HEIGHT, WIDTH),\n                        seed=seed)\n\ntest_generator=datagen.flow_from_dataframe(  \n                       dataframe=test,\n                       directory=test_dest_path,\n                       x_col=\"id_code\",\n                       batch_size=1,\n                       class_mode=None,\n                       shuffle=False,\n                       target_size=(HEIGHT, WIDTH),\n                       seed=seed)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T02:08:15.985433Z","iopub.execute_input":"2024-05-01T02:08:15.985754Z","iopub.status.idle":"2024-05-01T02:08:16.064210Z","shell.execute_reply.started":"2024-05-01T02:08:15.985711Z","shell.execute_reply":"2024-05-01T02:08:16.063462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cosine_decay_with_warmup(global_step,\n                             learning_rate_base,\n                             total_steps,\n                             warmup_learning_rate=0.0,\n                             warmup_steps=0,\n                             hold_base_rate_steps=0):\n    \"\"\"\n    Cosine decay schedule with warm up period.\n    In this schedule, the learning rate grows linearly from warmup_learning_rate\n    to learning_rate_base for warmup_steps, then transitions to a cosine decay\n    schedule.\n    :param global_step {int}: global step.\n    :param learning_rate_base {float}: base learning rate.\n    :param total_steps {int}: total number of training steps.\n    :param warmup_learning_rate {float}: initial learning rate for warm up. (default: {0.0}).\n    :param warmup_steps {int}: number of warmup steps. (default: {0}).\n    :param hold_base_rate_steps {int}: Optional number of steps to hold base learning rate before decaying. (default: {0}).\n    :param global_step {int}: global step.\n    :Returns : a float representing learning rate.\n    :Raises ValueError: if warmup_learning_rate is larger than learning_rate_base, or if warmup_steps is larger than total_steps.\n    \"\"\"\n\n    if total_steps < warmup_steps:\n        raise ValueError('total_steps must be larger or equal to warmup_steps.')\n    learning_rate = 0.5 * learning_rate_base * (1 + np.cos(\n        np.pi *\n        (global_step - warmup_steps - hold_base_rate_steps\n         ) / float(total_steps - warmup_steps - hold_base_rate_steps)))\n    if hold_base_rate_steps > 0:\n        learning_rate = np.where(global_step > warmup_steps + hold_base_rate_steps,\n                                 learning_rate, learning_rate_base)\n    if warmup_steps > 0:\n        if learning_rate_base < warmup_learning_rate:\n            raise ValueError('learning_rate_base must be larger or equal to warmup_learning_rate.')\n        slope = (learning_rate_base - warmup_learning_rate) / warmup_steps\n        warmup_rate = slope * global_step + warmup_learning_rate\n        learning_rate = np.where(global_step < warmup_steps, warmup_rate,\n                                 learning_rate)\n    return np.where(global_step > total_steps, 0.0, learning_rate)\n\n\nclass WarmUpCosineDecayScheduler(Callback):\n    \"\"\"Cosine decay with warmup learning rate scheduler\"\"\"\n\n    def __init__(self,\n                 learning_rate_base,\n                 total_steps,\n                 global_step_init=0,\n                 warmup_learning_rate=0.0,\n                 warmup_steps=0,\n                 hold_base_rate_steps=0,\n                 verbose=0):\n        \"\"\"\n        Constructor for cosine decay with warmup learning rate scheduler.\n        :param learning_rate_base {float}: base learning rate.\n        :param total_steps {int}: total number of training steps.\n        :param global_step_init {int}: initial global step, e.g. from previous checkpoint.\n        :param warmup_learning_rate {float}: initial learning rate for warm up. (default: {0.0}).\n        :param warmup_steps {int}: number of warmup steps. (default: {0}).\n        :param hold_base_rate_steps {int}: Optional number of steps to hold base learning rate before decaying. (default: {0}).\n        :param verbose {int}: quiet, 1: update messages. (default: {0}).\n        \"\"\"\n\n        super(WarmUpCosineDecayScheduler, self).__init__()\n        self.learning_rate_base = learning_rate_base\n        self.total_steps = total_steps\n        self.global_step = global_step_init\n        self.warmup_learning_rate = warmup_learning_rate\n        self.warmup_steps = warmup_steps\n        self.hold_base_rate_steps = hold_base_rate_steps\n        self.verbose = verbose\n        self.learning_rates = []\n\n    def on_batch_end(self, batch, logs=None):\n        self.global_step = self.global_step + 1\n        lr = K.get_value(self.model.optimizer.lr)\n        self.learning_rates.append(lr)\n\n    def on_batch_begin(self, batch, logs=None):\n        lr = cosine_decay_with_warmup(global_step=self.global_step,\n                                      learning_rate_base=self.learning_rate_base,\n                                      total_steps=self.total_steps,\n                                      warmup_learning_rate=self.warmup_learning_rate,\n                                      warmup_steps=self.warmup_steps,\n                                      hold_base_rate_steps=self.hold_base_rate_steps)\n        K.set_value(self.model.optimizer.lr, lr)\n        if self.verbose > 0:\n            print('\\nBatch %02d: setting learning rate to %s.' % (self.global_step + 1, lr))","metadata":{"execution":{"iopub.status.busy":"2024-05-01T02:08:24.042482Z","iopub.execute_input":"2024-05-01T02:08:24.042777Z","iopub.status.idle":"2024-05-01T02:08:24.061343Z","shell.execute_reply.started":"2024-05-01T02:08:24.042734Z","shell.execute_reply":"2024-05-01T02:08:24.060571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model(input_shape):\n    input_tensor = Input(shape=input_shape)\n    base_model = EfficientNetB5(weights=None, \n                                include_top=False,\n                                input_tensor=input_tensor)\n    base_model.load_weights('../input/efficientnet-keras-weights-b0b5/efficientnet-b5_imagenet_1000_notop.h5')\n\n    x = GlobalAveragePooling2D()(base_model.output)\n    final_output = Dense(1, activation='linear', name='final_output')(x)\n    model = Model(input_tensor, final_output)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2024-05-01T02:08:40.756946Z","iopub.execute_input":"2024-05-01T02:08:40.757239Z","iopub.status.idle":"2024-05-01T02:08:40.763533Z","shell.execute_reply.started":"2024-05-01T02:08:40.757197Z","shell.execute_reply":"2024-05-01T02:08:40.762674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = create_model(input_shape=(HEIGHT, WIDTH, CHANNELS))\n\nfor layer in model.layers:\n    layer.trainable = False\n\nfor i in range(-2, 0):\n    model.layers[i].trainable = True\n\ncosine_lr_1st = WarmUpCosineDecayScheduler(learning_rate_base=WARMUP_LEARNING_RATE,\n                                           total_steps=TOTAL_STEPS_1st,\n                                           warmup_learning_rate=0.0,\n                                           warmup_steps=WARMUP_STEPS_1st,\n                                           hold_base_rate_steps=(2 * STEP_SIZE))\n\nmetric_list = [\"accuracy\"]\ncallback_list = [cosine_lr_1st]\noptimizer = optimizers.Adam(lr=WARMUP_LEARNING_RATE)\nmodel.compile(optimizer=optimizer, loss='mean_squared_error', metrics=metric_list)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-05-01T02:08:48.179542Z","iopub.execute_input":"2024-05-01T02:08:48.179831Z","iopub.status.idle":"2024-05-01T02:09:19.128154Z","shell.execute_reply.started":"2024-05-01T02:08:48.179791Z","shell.execute_reply":"2024-05-01T02:09:19.127402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEP_SIZE_TRAIN = train_generator.n//train_generator.batch_size\nSTEP_SIZE_VALID = valid_generator.n//valid_generator.batch_size\n\nhistory_warmup = model.fit_generator(generator=train_generator,\n                                     steps_per_epoch=STEP_SIZE_TRAIN,\n                                     validation_data=valid_generator,\n                                     validation_steps=STEP_SIZE_VALID,\n                                     epochs=WARMUP_EPOCHS,\n                                     callbacks=callback_list,\n                                     verbose=2).history","metadata":{"execution":{"iopub.status.busy":"2024-05-01T02:09:36.429335Z","iopub.execute_input":"2024-05-01T02:09:36.429655Z","iopub.status.idle":"2024-05-01T02:12:56.166422Z","shell.execute_reply.started":"2024-05-01T02:09:36.429611Z","shell.execute_reply":"2024-05-01T02:12:56.165135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in model.layers:\n    layer.trainable = True\n\nes = EarlyStopping(monitor='val_loss', mode='min', patience=ES_PATIENCE, restore_best_weights=True, verbose=1)\ncosine_lr_2nd = WarmUpCosineDecayScheduler(learning_rate_base=LEARNING_RATE,\n                                           total_steps=TOTAL_STEPS_2nd,\n                                           warmup_learning_rate=0.0,\n                                           warmup_steps=WARMUP_STEPS_2nd,\n                                           hold_base_rate_steps=(3 * STEP_SIZE))\n\ncallback_list = [es, cosine_lr_2nd]\noptimizer = optimizers.Adam(lr=LEARNING_RATE)\nmodel.compile(optimizer=optimizer, loss='mean_squared_error', metrics=metric_list)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-05-01T02:21:50.916234Z","iopub.execute_input":"2024-05-01T02:21:50.916589Z","iopub.status.idle":"2024-05-01T02:21:51.157954Z","shell.execute_reply.started":"2024-05-01T02:21:50.916533Z","shell.execute_reply":"2024-05-01T02:21:51.157269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit_generator(generator=train_generator,\n                              steps_per_epoch=STEP_SIZE_TRAIN,\n                              validation_data=valid_generator,\n                              validation_steps=STEP_SIZE_VALID,\n                              epochs=EPOCHS,\n                              callbacks=callback_list,\n                              verbose=2).history","metadata":{"execution":{"iopub.status.busy":"2024-05-01T02:22:12.200535Z","iopub.execute_input":"2024-05-01T02:22:12.200841Z","iopub.status.idle":"2024-05-01T02:43:46.530152Z","shell.execute_reply.started":"2024-05-01T02:22:12.200799Z","shell.execute_reply":"2024-05-01T02:43:46.529188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(2, 1, sharex='col', figsize=(20, 6))\n\nax1.plot(cosine_lr_1st.learning_rates)\nax1.set_title('Warm up learning rates')\n\nax2.plot(cosine_lr_2nd.learning_rates)\nax2.set_title('Fine-tune learning rates')\n\nplt.xlabel('Steps')\nplt.ylabel('Learning rate')\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-01T02:46:06.723577Z","iopub.execute_input":"2024-05-01T02:46:06.723901Z","iopub.status.idle":"2024-05-01T02:46:07.190409Z","shell.execute_reply.started":"2024-05-01T02:46:06.723864Z","shell.execute_reply":"2024-05-01T02:46:07.189542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(2, 1, sharex='col', figsize=(20, 14))\n\nax1.plot(history['loss'], label='Train loss')\nax1.plot(history['val_loss'], label='Validation loss')\nax1.legend(loc='best')\nax1.set_title('Loss')\n\nax2.plot(history['acc'], label='Train accuracy')\nax2.plot(history['val_acc'], label='Validation accuracy')\nax2.legend(loc='best')\nax2.set_title('Accuracy')\n\nplt.xlabel('Epochs')\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-01T02:46:09.656280Z","iopub.execute_input":"2024-05-01T02:46:09.656606Z","iopub.status.idle":"2024-05-01T02:46:10.309454Z","shell.execute_reply.started":"2024-05-01T02:46:09.656564Z","shell.execute_reply":"2024-05-01T02:46:10.308712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create empty arays to keep the predictions and labels\ndf_preds = pd.DataFrame(columns=['label', 'pred', 'set'])\ntrain_generator.reset()\nvalid_generator.reset()\n\n# Add train predictions and labels\nfor i in range(STEP_SIZE_TRAIN + 1):\n    im, lbl = next(train_generator)\n    preds = model.predict(im, batch_size=train_generator.batch_size)\n    for index in range(len(preds)):\n        df_preds.loc[len(df_preds)] = [lbl[index], preds[index][0], 'train']\n\n# Add validation predictions and labels\nfor i in range(STEP_SIZE_VALID + 1):\n    im, lbl = next(valid_generator)\n    preds = model.predict(im, batch_size=valid_generator.batch_size)\n    for index in range(len(preds)):\n        df_preds.loc[len(df_preds)] = [lbl[index], preds[index][0], 'validation']\n\ndf_preds['label'] = df_preds['label'].astype('int')","metadata":{"execution":{"iopub.status.busy":"2024-05-01T02:46:43.742447Z","iopub.execute_input":"2024-05-01T02:46:43.742800Z","iopub.status.idle":"2024-05-01T02:48:02.109891Z","shell.execute_reply.started":"2024-05-01T02:46:43.742739Z","shell.execute_reply":"2024-05-01T02:48:02.108774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def classify(x):\n    if x < 0.5:\n        return 0\n    elif x < 1.5:\n        return 1\n    elif x < 2.5:\n        return 2\n    elif x < 3.5:\n        return 3\n    return 4\n\n# Classify predictions\ndf_preds['predictions'] = df_preds['pred'].apply(lambda x: classify(x))\n\ntrain_preds = df_preds[df_preds['set'] == 'train']\nvalidation_preds = df_preds[df_preds['set'] == 'validation']","metadata":{"execution":{"iopub.status.busy":"2024-05-01T02:49:01.268061Z","iopub.execute_input":"2024-05-01T02:49:01.268391Z","iopub.status.idle":"2024-05-01T02:49:01.285778Z","shell.execute_reply.started":"2024-05-01T02:49:01.268333Z","shell.execute_reply":"2024-05-01T02:49:01.284904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = ['0 - No DR', '1 - Mild', '2 - Moderate', '3 - Severe', '4 - Proliferative DR']\ndef plot_confusion_matrix(train, validation, labels=labels):\n    train_labels, train_preds = train\n    validation_labels, validation_preds = validation\n    fig, (ax1, ax2) = plt.subplots(1, 2, sharex='col', figsize=(24, 7))\n    train_cnf_matrix = confusion_matrix(train_labels, train_preds)\n    validation_cnf_matrix = confusion_matrix(validation_labels, validation_preds)\n\n    train_cnf_matrix_norm = train_cnf_matrix.astype('float') / train_cnf_matrix.sum(axis=1)[:, np.newaxis]\n    validation_cnf_matrix_norm = validation_cnf_matrix.astype('float') / validation_cnf_matrix.sum(axis=1)[:, np.newaxis]\n\n    train_df_cm = pd.DataFrame(train_cnf_matrix_norm, index=labels, columns=labels)\n    validation_df_cm = pd.DataFrame(validation_cnf_matrix_norm, index=labels, columns=labels)\n\n    sns.heatmap(train_df_cm, annot=True, fmt='.2f', cmap=\"Blues\",ax=ax1).set_title('Train')\n    sns.heatmap(validation_df_cm, annot=True, fmt='.2f', cmap=sns.cubehelix_palette(8),ax=ax2).set_title('Validation')\n    plt.show()\n\nplot_confusion_matrix((train_preds['label'], train_preds['predictions']), (validation_preds['label'], validation_preds['predictions']))","metadata":{"execution":{"iopub.status.busy":"2024-05-01T02:49:04.380504Z","iopub.execute_input":"2024-05-01T02:49:04.380846Z","iopub.status.idle":"2024-05-01T02:49:05.367817Z","shell.execute_reply.started":"2024-05-01T02:49:04.380790Z","shell.execute_reply":"2024-05-01T02:49:05.366724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def evaluate_model(train, validation):\n    train_labels, train_preds = train\n    validation_labels, validation_preds = validation\n    print(\"Train        Cohen Kappa score: %.3f\" % cohen_kappa_score(train_preds, train_labels, weights='quadratic'))\n    print(\"Validation   Cohen Kappa score: %.3f\" % cohen_kappa_score(validation_preds, validation_labels, weights='quadratic'))\n    print(\"Complete set Cohen Kappa score: %.3f\" % cohen_kappa_score(np.append(train_preds, validation_preds), np.append(train_labels, validation_labels), weights='quadratic'))\n    \nevaluate_model((train_preds['label'], train_preds['predictions']), (validation_preds['label'], validation_preds['predictions']))","metadata":{"execution":{"iopub.status.busy":"2024-05-01T02:49:07.683363Z","iopub.execute_input":"2024-05-01T02:49:07.683687Z","iopub.status.idle":"2024-05-01T02:49:07.708190Z","shell.execute_reply.started":"2024-05-01T02:49:07.683641Z","shell.execute_reply":"2024-05-01T02:49:07.707492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def apply_tta(model, generator, steps=10):\n    step_size = generator.n//generator.batch_size\n    preds_tta = []\n    for i in range(steps):\n        generator.reset()\n        preds = model.predict_generator(generator, steps=step_size)\n        preds_tta.append(preds)\n\n    return np.mean(preds_tta, axis=0)\n\npreds = apply_tta(model, test_generator)\npredictions = [classify(x) for x in preds]\n\nresults = pd.DataFrame({'id_code':test['id_code'], 'diagnosis':predictions})\nresults['id_code'] = results['id_code'].map(lambda x: str(x)[:-4])","metadata":{"execution":{"iopub.status.busy":"2024-05-01T02:49:09.397481Z","iopub.execute_input":"2024-05-01T02:49:09.397807Z","iopub.status.idle":"2024-05-01T03:01:09.856120Z","shell.execute_reply.started":"2024-05-01T02:49:09.397763Z","shell.execute_reply":"2024-05-01T03:01:09.855451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cleaning created directories\nif os.path.exists(train_dest_path):\n    shutil.rmtree(train_dest_path)\nif os.path.exists(validation_dest_path):\n    shutil.rmtree(validation_dest_path)\nif os.path.exists(test_dest_path):\n    shutil.rmtree(test_dest_path)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T03:06:29.306829Z","iopub.execute_input":"2024-05-01T03:06:29.307181Z","iopub.status.idle":"2024-05-01T03:06:29.588973Z","shell.execute_reply.started":"2024-05-01T03:06:29.307123Z","shell.execute_reply":"2024-05-01T03:06:29.588042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.subplots(sharex='col', figsize=(24, 8.7))\nsns.countplot(x=\"diagnosis\", data=results).set_title('Test')\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-01T03:06:31.521047Z","iopub.execute_input":"2024-05-01T03:06:31.521414Z","iopub.status.idle":"2024-05-01T03:06:31.887991Z","shell.execute_reply.started":"2024-05-01T03:06:31.521337Z","shell.execute_reply":"2024-05-01T03:06:31.887176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results.to_csv('submission.csv', index=False)\ndisplay(results.head())","metadata":{"execution":{"iopub.status.busy":"2024-05-01T03:06:34.176364Z","iopub.execute_input":"2024-05-01T03:06:34.176677Z","iopub.status.idle":"2024-05-01T03:06:34.195659Z","shell.execute_reply.started":"2024-05-01T03:06:34.176633Z","shell.execute_reply":"2024-05-01T03:06:34.194619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save_weights('../working/effNetB5_bs32_img224_fold5.h5')","metadata":{"execution":{"iopub.status.busy":"2024-05-01T03:06:37.544908Z","iopub.execute_input":"2024-05-01T03:06:37.545224Z","iopub.status.idle":"2024-05-01T03:12:34.883774Z","shell.execute_reply.started":"2024-05-01T03:06:37.545183Z","shell.execute_reply":"2024-05-01T03:12:34.882982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"5_FOLD","metadata":{}},{"cell_type":"code","source":"import os\nimport sys\nimport cv2\nimport shutil\nimport random\nimport warnings\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom tensorflow import set_random_seed\nfrom sklearn.metrics import confusion_matrix, cohen_kappa_score\nfrom keras.models import Model\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.layers import Dense, GlobalAveragePooling2D, Input\n\ndef seed_everything(seed=0):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    set_random_seed(0)\n\nseed = 0\nseed_everything(seed)\n%matplotlib inline\nsns.set(style=\"whitegrid\")\nwarnings.filterwarnings(\"ignore\")\nsys.path.append(os.path.abspath('../input/efficientnet/efficientnet-master/efficientnet-master/'))\nfrom efficientnet import *","metadata":{"execution":{"iopub.status.busy":"2024-05-01T13:20:15.448882Z","iopub.execute_input":"2024-05-01T13:20:15.449215Z","iopub.status.idle":"2024-05-01T13:20:20.786335Z","shell.execute_reply.started":"2024-05-01T13:20:15.449165Z","shell.execute_reply":"2024-05-01T13:20:20.785453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hold_out_set = pd.read_csv('/kaggle/input/5-fold/hold-out.csv')\nX_train = hold_out_set[hold_out_set['set'] == 'train']\nX_val = hold_out_set[hold_out_set['set'] == 'validation']\ntest = pd.read_csv('../input/aptos2019-blindness-detection/test.csv')\nprint('Number of train samples: ', X_train.shape[0])\nprint('Number of validation samples: ', X_val.shape[0])\nprint('Number of test samples: ', test.shape[0])\n\n# Preprocecss data\nX_train[\"id_code\"] = X_train[\"id_code\"].apply(lambda x: x + \".png\")\nX_val[\"id_code\"] = X_val[\"id_code\"].apply(lambda x: x + \".png\")\ntest[\"id_code\"] = test[\"id_code\"].apply(lambda x: x + \".png\")\ndisplay(X_train.head())","metadata":{"execution":{"iopub.status.busy":"2024-05-01T13:20:20.788524Z","iopub.execute_input":"2024-05-01T13:20:20.788785Z","iopub.status.idle":"2024-05-01T13:20:21.063654Z","shell.execute_reply.started":"2024-05-01T13:20:20.788741Z","shell.execute_reply":"2024-05-01T13:20:21.062849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model parameters\nHEIGHT = 224\nWIDTH = 224\nCHANNELS = 3\n\nweights_paths = ['/kaggle/input/effnet/effNetB5_bs32_img224_fold1.h5', '/kaggle/input/effnet/effNetB5_bs32_img224_fold4.h5', \n                 '/kaggle/input/effnet/effNetB5_bs32_img224_fold2.h5', '/kaggle/input/effnet/effNetB5_bs32_img224_fold5.h5', \n                 '/kaggle/input/effnet/effNetB5_bs32_img224_fold3.h5']\nn_folds = len(weights_paths)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T13:20:21.065057Z","iopub.execute_input":"2024-05-01T13:20:21.065323Z","iopub.status.idle":"2024-05-01T13:20:21.069986Z","shell.execute_reply.started":"2024-05-01T13:20:21.065281Z","shell.execute_reply":"2024-05-01T13:20:21.069232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = ['0 - No DR', '1 - Mild', '2 - Moderate', '3 - Severe', '4 - Proliferative DR']\ndef plot_confusion_matrix(train, validation, labels=labels):\n    train_labels, train_preds = train\n    validation_labels, validation_preds = validation\n    fig, (ax1, ax2) = plt.subplots(1, 2, sharex='col', figsize=(24, 7))\n    train_cnf_matrix = confusion_matrix(train_labels, train_preds)\n    validation_cnf_matrix = confusion_matrix(validation_labels, validation_preds)\n\n    train_cnf_matrix_norm = train_cnf_matrix.astype('float') / train_cnf_matrix.sum(axis=1)[:, np.newaxis]\n    validation_cnf_matrix_norm = validation_cnf_matrix.astype('float') / validation_cnf_matrix.sum(axis=1)[:, np.newaxis]\n\n    train_df_cm = pd.DataFrame(train_cnf_matrix_norm, index=labels, columns=labels)\n    validation_df_cm = pd.DataFrame(validation_cnf_matrix_norm, index=labels, columns=labels)\n\n    sns.heatmap(train_df_cm, annot=True, fmt='.2f', cmap=\"Blues\",ax=ax1).set_title('Train')\n    sns.heatmap(validation_df_cm, annot=True, fmt='.2f', cmap=sns.cubehelix_palette(8),ax=ax2).set_title('Validation')\n    plt.show()\n    \ndef evaluate_model(train, validation):\n    train_labels, train_preds = train\n    validation_labels, validation_preds = validation\n    print(\"Train        Cohen Kappa score: %.3f\" % cohen_kappa_score(train_preds, train_labels, weights='quadratic'))\n    print(\"Validation   Cohen Kappa score: %.3f\" % cohen_kappa_score(validation_preds, validation_labels, weights='quadratic'))\n    print(\"Complete set Cohen Kappa score: %.3f\" % cohen_kappa_score(np.append(train_preds, validation_preds), np.append(train_labels, validation_labels), weights='quadratic'))\n\ndef classify(x):\n    if x < 0.5:\n        return 0\n    elif x < 1.5:\n        return 1\n    elif x < 2.5:\n        return 2\n    elif x < 3.5:\n        return 3\n    return 4\n\ndef ensemble_preds(model_list, generator):\n    preds_ensemble = []\n    for model in model_list:\n        generator.reset()\n        preds = model.predict_generator(generator, steps=generator.n)\n        preds_ensemble.append(preds)\n\n    return np.mean(preds_ensemble, axis=0)\n\ndef apply_tta(model, generator, steps=10):\n    step_size = generator.n//generator.batch_size\n    preds_tta = []\n    for i in range(steps):\n        generator.reset()\n        preds = model.predict_generator(generator, steps=step_size)\n        preds_tta.append(preds)\n\n    return np.mean(preds_tta, axis=0)\n\ndef test_ensemble_preds(model_list, generator):\n    preds_ensemble = []\n    for model in model_list:\n        preds = apply_tta(model, generator)\n        preds_ensemble.append(preds)\n\n    return np.mean(preds_ensemble, axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T13:20:21.071848Z","iopub.execute_input":"2024-05-01T13:20:21.072175Z","iopub.status.idle":"2024-05-01T13:20:21.094775Z","shell.execute_reply.started":"2024-05-01T13:20:21.072116Z","shell.execute_reply":"2024-05-01T13:20:21.093925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_base_path = '../input/aptos2019-blindness-detection/train_images/'\ntest_base_path = '../input/aptos2019-blindness-detection/test_images/'\ntrain_dest_path = 'base_dir/train_images/'\nvalidation_dest_path = 'base_dir/validation_images/'\ntest_dest_path =  'base_dir/test_images/'\n\n# Making sure directories don't exist\nif os.path.exists(train_dest_path):\n    shutil.rmtree(train_dest_path)\nif os.path.exists(validation_dest_path):\n    shutil.rmtree(validation_dest_path)\nif os.path.exists(test_dest_path):\n    shutil.rmtree(test_dest_path)\n    \n# Creating train, validation and test directories\nos.makedirs(train_dest_path)\nos.makedirs(validation_dest_path)\nos.makedirs(test_dest_path)\n\ndef crop_image(img, tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n            img = np.stack([img1,img2,img3],axis=-1)\n            \n        return img\n\ndef circle_crop(img):\n    img = crop_image(img)\n\n    height, width, depth = img.shape\n    largest_side = np.max((height, width))\n    img = cv2.resize(img, (largest_side, largest_side))\n\n    height, width, depth = img.shape\n\n    x = width//2\n    y = height//2\n    r = np.amin((x, y))\n\n    circle_img = np.zeros((height, width), np.uint8)\n    cv2.circle(circle_img, (x, y), int(r), 1, thickness=-1)\n    img = cv2.bitwise_and(img, img, mask=circle_img)\n    img = crop_image(img)\n\n    return img\n    \ndef preprocess_image(base_path, save_path, image_id, HEIGHT, WIDTH, sigmaX=10):\n    image = cv2.imread(base_path + image_id)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    image = circle_crop(image)\n    image = cv2.resize(image, (HEIGHT, WIDTH))\n    image = cv2.addWeighted(image, 4, cv2.GaussianBlur(image, (0,0), sigmaX), -4 , 128)\n    cv2.imwrite(save_path + image_id, image)\n    \n# Pre-procecss train set\nfor i, image_id in enumerate(X_train['id_code']):\n    preprocess_image(train_base_path, train_dest_path, image_id, HEIGHT, WIDTH)\n    \n# Pre-procecss validation set\nfor i, image_id in enumerate(X_val['id_code']):\n    preprocess_image(train_base_path, validation_dest_path, image_id, HEIGHT, WIDTH)\n    \n# Pre-procecss test set\nfor i, image_id in enumerate(test['id_code']):\n    preprocess_image(test_base_path, test_dest_path, image_id, HEIGHT, WIDTH)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T13:20:21.097781Z","iopub.execute_input":"2024-05-01T13:20:21.098032Z","iopub.status.idle":"2024-05-01T13:44:43.324957Z","shell.execute_reply.started":"2024-05-01T13:20:21.097991Z","shell.execute_reply":"2024-05-01T13:44:43.324237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datagen=ImageDataGenerator(rescale=1./255, \n                           rotation_range=360,\n                           horizontal_flip=True,\n                           vertical_flip=True)\n\ntrain_generator=datagen.flow_from_dataframe(\n                        dataframe=X_train,\n                        directory=train_dest_path,\n                        x_col=\"id_code\",\n                        y_col=\"diagnosis\",\n                        class_mode=\"raw\",\n                        batch_size=1,\n                        shuffle=False,\n                        target_size=(HEIGHT, WIDTH),\n                        seed=seed)\n\nvalid_generator=datagen.flow_from_dataframe(\n                        dataframe=X_val,\n                        directory=validation_dest_path,\n                        x_col=\"id_code\",\n                        y_col=\"diagnosis\",\n                        class_mode=\"raw\",\n                        batch_size=1,\n                        shuffle=False,\n                        target_size=(HEIGHT, WIDTH),\n                        seed=seed)\n\ntest_generator=datagen.flow_from_dataframe(  \n                       dataframe=test,\n                       directory=test_dest_path,\n                       x_col=\"id_code\",\n                       batch_size=1,\n                       class_mode=None,\n                       shuffle=False,\n                       target_size=(HEIGHT, WIDTH),\n                       seed=seed)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T13:44:43.329183Z","iopub.execute_input":"2024-05-01T13:44:43.329441Z","iopub.status.idle":"2024-05-01T13:44:43.408534Z","shell.execute_reply.started":"2024-05-01T13:44:43.329400Z","shell.execute_reply":"2024-05-01T13:44:43.407590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model(input_shape, weights_path):\n    input_tensor = Input(shape=input_shape)\n    base_model = EfficientNetB5(weights=None, \n                                include_top=False,\n                                input_tensor=input_tensor)\n\n    x = GlobalAveragePooling2D()(base_model.output)\n    final_output = Dense(1, activation='linear', name='final_output')(x)\n    model = Model(input_tensor, final_output)\n    model.load_weights(weights_path)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2024-05-01T13:44:43.409992Z","iopub.execute_input":"2024-05-01T13:44:43.410296Z","iopub.status.idle":"2024-05-01T13:44:43.416219Z","shell.execute_reply.started":"2024-05-01T13:44:43.410249Z","shell.execute_reply":"2024-05-01T13:44:43.415411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_list = []\n\nfor weights_path in weights_paths:\n    model_list.append(create_model(input_shape=(HEIGHT, WIDTH, CHANNELS), weights_path=weights_path))","metadata":{"execution":{"iopub.status.busy":"2024-05-01T13:44:43.417433Z","iopub.execute_input":"2024-05-01T13:44:43.417748Z","iopub.status.idle":"2024-05-01T13:47:25.941085Z","shell.execute_reply.started":"2024-05-01T13:44:43.417647Z","shell.execute_reply":"2024-05-01T13:47:25.940375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train predictions\npreds_ensemble = ensemble_preds(model_list, train_generator)\npreds_ensemble = [classify(x) for x in preds_ensemble]\ntrain_preds = pd.DataFrame({'label':train_generator.labels, 'pred':preds_ensemble})\n\n# Validation predictions\npreds_ensemble = ensemble_preds(model_list, valid_generator)\npreds_ensemble = [classify(x) for x in preds_ensemble]\nvalidation_preds = pd.DataFrame({'label':valid_generator.labels, 'pred':preds_ensemble})","metadata":{"execution":{"iopub.status.busy":"2024-05-01T13:47:25.942370Z","iopub.execute_input":"2024-05-01T13:47:25.942642Z","iopub.status.idle":"2024-05-01T14:00:17.378982Z","shell.execute_reply.started":"2024-05-01T13:47:25.942592Z","shell.execute_reply":"2024-05-01T14:00:17.378192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_confusion_matrix((train_preds['label'], train_preds['pred']), (validation_preds['label'], validation_preds['pred']))","metadata":{"execution":{"iopub.status.busy":"2024-05-01T14:00:17.380517Z","iopub.execute_input":"2024-05-01T14:00:17.380857Z","iopub.status.idle":"2024-05-01T14:00:18.445797Z","shell.execute_reply.started":"2024-05-01T14:00:17.380789Z","shell.execute_reply":"2024-05-01T14:00:18.444511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluate_model((train_preds['label'], train_preds['pred']), (validation_preds['label'], validation_preds['pred']))","metadata":{"execution":{"iopub.status.busy":"2024-05-01T14:00:18.447871Z","iopub.execute_input":"2024-05-01T14:00:18.448589Z","iopub.status.idle":"2024-05-01T14:00:18.477410Z","shell.execute_reply.started":"2024-05-01T14:00:18.448510Z","shell.execute_reply":"2024-05-01T14:00:18.476656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = test_ensemble_preds(model_list, test_generator)\npredictions = [classify(x) for x in preds]\n\nresults = pd.DataFrame({'id_code':test['id_code'], 'diagnosis':predictions})\nresults['id_code'] = results['id_code'].map(lambda x: str(x)[:-4])","metadata":{"execution":{"iopub.status.busy":"2024-05-01T14:00:18.478797Z","iopub.execute_input":"2024-05-01T14:00:18.479090Z","iopub.status.idle":"2024-05-01T15:03:49.067262Z","shell.execute_reply.started":"2024-05-01T14:00:18.479023Z","shell.execute_reply":"2024-05-01T15:03:49.066412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cleaning created directories\nif os.path.exists(train_dest_path):\n    shutil.rmtree(train_dest_path)\nif os.path.exists(validation_dest_path):\n    shutil.rmtree(validation_dest_path)\nif os.path.exists(test_dest_path):\n    shutil.rmtree(test_dest_path)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T15:03:49.068717Z","iopub.execute_input":"2024-05-01T15:03:49.068962Z","iopub.status.idle":"2024-05-01T15:03:49.341266Z","shell.execute_reply.started":"2024-05-01T15:03:49.068922Z","shell.execute_reply":"2024-05-01T15:03:49.340545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.subplots(sharex='col', figsize=(24, 8.7))\nsns.countplot(x=\"diagnosis\", data=results).set_title('Test')\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-01T15:03:49.342868Z","iopub.execute_input":"2024-05-01T15:03:49.343234Z","iopub.status.idle":"2024-05-01T15:03:49.715376Z","shell.execute_reply.started":"2024-05-01T15:03:49.343173Z","shell.execute_reply":"2024-05-01T15:03:49.714446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results.to_csv('submission.csv', index=False)\ndisplay(results.head())","metadata":{"execution":{"iopub.status.busy":"2024-05-01T15:03:49.716865Z","iopub.execute_input":"2024-05-01T15:03:49.717220Z","iopub.status.idle":"2024-05-01T15:03:49.874700Z","shell.execute_reply.started":"2024-05-01T15:03:49.717152Z","shell.execute_reply":"2024-05-01T15:03:49.873923Z"},"trusted":true},"execution_count":null,"outputs":[]}]}