{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# EfficientNetB3Trained with Old and New Data\n\n\n---","metadata":{}},{"cell_type":"markdown","source":"The Inference kernel of this notebook can be found here: https://www.kaggle.com/fanconic/efficientnetb3-inference-keras?scriptVersionId=18596729\n\n - LB Score: 0.786\n - Private Score: 0.910","metadata":{}},{"cell_type":"code","source":"# To have reproducible results and compare them\nnr_seed = 11\nimport numpy as np \nnp.random.seed(nr_seed)\nimport tensorflow as tf\ntf.set_random_seed(nr_seed)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:16:36.477095Z","iopub.execute_input":"2023-06-28T23:16:36.477436Z","iopub.status.idle":"2023-06-28T23:16:38.701574Z","shell.execute_reply.started":"2023-06-28T23:16:36.477377Z","shell.execute_reply":"2023-06-28T23:16:38.699468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import libraries\n!pip install -U '../input/install/efficientnet-0.0.3-py2.py3-none-any.whl'\nimport json\nimport math\nfrom tqdm import tqdm, tqdm_notebook\nimport gc\nimport warnings\nimport os\n\nimport cv2\nfrom PIL import Image\n\nimport pandas as pd\nimport scipy\nimport matplotlib.pyplot as plt\n\nfrom keras import backend as K\nfrom keras import layers\nfrom efficientnet import EfficientNetB3\nfrom keras.callbacks import Callback, ModelCheckpoint, ReduceLROnPlateau\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential\nfrom keras.optimizers import Adam\nfrom keras.losses import binary_crossentropy, categorical_crossentropy\nfrom skimage.color import rgb2hsv, lab2lch\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score\n\nwarnings.filterwarnings(\"ignore\")\n\n%matplotlib inline","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-28T23:16:38.703853Z","iopub.execute_input":"2023-06-28T23:16:38.704153Z","iopub.status.idle":"2023-06-28T23:17:07.077247Z","shell.execute_reply.started":"2023-06-28T23:16:38.704095Z","shell.execute_reply":"2023-06-28T23:17:07.076284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image size\nWIDTH= 320\nHEIGHT = 320\n# Batch size\nBATCH_SIZE = 32","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:17:07.078842Z","iopub.execute_input":"2023-06-28T23:17:07.079145Z","iopub.status.idle":"2023-06-28T23:17:07.09197Z","shell.execute_reply.started":"2023-06-28T23:17:07.07909Z","shell.execute_reply":"2023-06-28T23:17:07.091188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading & Merging","metadata":{}},{"cell_type":"code","source":"new_train = pd.read_csv('../input/aptos2019-blindness-detection/train.csv')\nold_train = pd.read_csv('../input/diabetic-retinopathy-resized/trainLabels_cropped.csv')\nduplicates = pd.read_csv('../input/aptos-trained-weights/inconsistent.csv')\nprint(new_train.shape)\nprint(old_train.shape)\nprint(duplicates.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:17:07.093597Z","iopub.execute_input":"2023-06-28T23:17:07.094193Z","iopub.status.idle":"2023-06-28T23:17:07.171986Z","shell.execute_reply.started":"2023-06-28T23:17:07.094135Z","shell.execute_reply":"2023-06-28T23:17:07.171163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for img_name in duplicates['id_code'].values:\n    new_train = new_train[new_train['id_code'] != img_name]\nprint(new_train.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:17:07.176523Z","iopub.execute_input":"2023-06-28T23:17:07.176799Z","iopub.status.idle":"2023-06-28T23:17:07.276634Z","shell.execute_reply.started":"2023-06-28T23:17:07.17675Z","shell.execute_reply":"2023-06-28T23:17:07.275688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"old_train = old_train[['image','level']]\nold_train.columns = new_train.columns\nold_train.diagnosis.value_counts()\n\n# path columns\nnew_train['id_code'] = '../input/aptos2019-blindness-detection/train_images/' + new_train['id_code'].astype(str) + '.png'\nold_train['id_code'] = '../input/diabetic-retinopathy-resized/resized_train/resized_train/' + old_train['id_code'].astype(str) + '.jpeg'\n\ntrain_df = old_train.copy()\nval_df = new_train.copy()\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:17:07.279872Z","iopub.execute_input":"2023-06-28T23:17:07.280327Z","iopub.status.idle":"2023-06-28T23:17:07.349352Z","shell.execute_reply.started":"2023-06-28T23:17:07.280136Z","shell.execute_reply":"2023-06-28T23:17:07.348506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.sample(frac=1).reset_index(drop=True)\ndef preprocess_image(image_path):\n    img = cv2.imread(image_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)    \n    return img\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:17:07.350745Z","iopub.execute_input":"2023-06-28T23:17:07.351033Z","iopub.status.idle":"2023-06-28T23:17:07.36167Z","shell.execute_reply.started":"2023-06-28T23:17:07.350985Z","shell.execute_reply":"2023-06-28T23:17:07.360888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_samples(df, columns=4, rows=3):\n    fig=plt.figure(figsize=(5*columns, 4*rows))\n\n    for i in range(columns * rows ):\n        image_path = df.loc[i,'id_code']\n        image_id = df.loc[i,'diagnosis']\n        img = preprocess_image(f'{image_path}')\n        fig.add_subplot(rows, columns, i+1)\n        plt.title(image_id)\n        plt.imshow(img)\n    \n    plt.tight_layout()\n\ndisplay_samples(train_df)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:17:07.364405Z","iopub.execute_input":"2023-06-28T23:17:07.3656Z","iopub.status.idle":"2023-06-28T23:17:11.658997Z","shell.execute_reply.started":"2023-06-28T23:17:07.364736Z","shell.execute_reply":"2023-06-28T23:17:11.654583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train - Valid split\nUse new Data for validation and Old data for training£","metadata":{}},{"cell_type":"code","source":"# Let's shuffle the datasets\ntrain_df = train_df.sample(frac=1).reset_index(drop=True)\nval_df = val_df.sample(frac=1).reset_index(drop=True)\nprint(train_df.shape)\nprint(val_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:17:11.660119Z","iopub.execute_input":"2023-06-28T23:17:11.6604Z","iopub.status.idle":"2023-06-28T23:17:11.677462Z","shell.execute_reply.started":"2023-06-28T23:17:11.660351Z","shell.execute_reply":"2023-06-28T23:17:11.676775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Process Images","metadata":{}},{"cell_type":"markdown","source":"Crop function: https://www.kaggle.com/ratthachat/aptos-updated-preprocessing-ben-s-cropping ","metadata":{}},{"cell_type":"code","source":"def crop_image1(img,tol=7):\n    # img is image data\n    # tol  is tolerance\n        \n    mask = img>tol\n    return img[np.ix_(mask.any(1),mask.any(0))]\n\ndef crop_image_from_gray(img,tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n            img = np.stack([img1,img2,img3],axis=-1)\n   \n        return img\n\n\n# Make all images circular (possible data loss)\ndef circle_crop(img):   \n    \"\"\"\n    Create circular crop around image centre    \n    \"\"\"    \n    \n    img = crop_image_from_gray(img)    \n    \n    height, width, depth = img.shape    \n    \n    x = int(width/2)\n    y = int(height/2)\n    r = np.amin((x,y))\n    \n    circle_img = np.zeros((height, width), np.uint8)\n    cv2.circle(circle_img, (x,y), int(r), 1, thickness=-1)\n    img = cv2.bitwise_and(img, img, mask=circle_img)\n    img = crop_image_from_gray(img)\n    \n    return img \n\n\ndef preprocess_image(image_path, width=320, height=320, new_data=False):\n    img = cv2.imread(image_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)    \n    if new_data:\n        img = crop_image_from_gray(img)\n    img = cv2.resize(img, (width,height))\n    #img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), 20) ,-4 ,128)\n\n    return img","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:17:11.678744Z","iopub.execute_input":"2023-06-28T23:17:11.679187Z","iopub.status.idle":"2023-06-28T23:17:11.704854Z","shell.execute_reply.started":"2023-06-28T23:17:11.679131Z","shell.execute_reply":"2023-06-28T23:17:11.703591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_samples(df, columns=4, rows=3):\n    fig=plt.figure(figsize=(5*columns, 4*rows))\n\n    for i in range(columns*rows):\n        image_path = df.loc[i,'id_code']\n        image_id = df.loc[i,'diagnosis']\n        img = preprocess_image(f'{image_path}', width=WIDTH, height=HEIGHT)\n        fig.add_subplot(rows, columns, i+1)\n        plt.title(image_id)\n        plt.imshow(img)\n    \n    plt.tight_layout()\n\ndisplay_samples(train_df)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-28T23:17:11.706392Z","iopub.execute_input":"2023-06-28T23:17:11.706884Z","iopub.status.idle":"2023-06-28T23:17:15.64493Z","shell.execute_reply.started":"2023-06-28T23:17:11.70683Z","shell.execute_reply":"2023-06-28T23:17:15.643942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Processing Images","metadata":{}},{"cell_type":"markdown","source":"__UPDATE:__ Here we are reading just the validation set. In order to use 320x320 images, we are going to load one bucket at a time only when needed. This will let our code run without memory-related errors.","metadata":{}},{"cell_type":"code","source":"# validation set\nN = val_df.shape[0]\nx_val = np.empty((N, HEIGHT, WIDTH, 3), dtype=np.uint8)\n\n# Create packet to save the modified Validation set, to reduced memory size \nfor i, image_id in enumerate(tqdm_notebook(val_df['id_code'])):\n    x_val[i, :, :, :] = preprocess_image(\n        f'{image_id}',\n        height=HEIGHT, width=WIDTH, new_data=True\n    )","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:17:15.646248Z","iopub.execute_input":"2023-06-28T23:17:15.646669Z","iopub.status.idle":"2023-06-28T23:28:25.160263Z","shell.execute_reply.started":"2023-06-28T23:17:15.646622Z","shell.execute_reply":"2023-06-28T23:28:25.159302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = pd.get_dummies(train_df['diagnosis']).values\ny_val = pd.get_dummies(val_df['diagnosis']).values\n\nprint(y_train.shape)\nprint(x_val.shape)\nprint(y_val.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:28:25.161951Z","iopub.execute_input":"2023-06-28T23:28:25.162497Z","iopub.status.idle":"2023-06-28T23:28:25.175705Z","shell.execute_reply.started":"2023-06-28T23:28:25.162443Z","shell.execute_reply":"2023-06-28T23:28:25.174824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:28:25.179199Z","iopub.execute_input":"2023-06-28T23:28:25.179696Z","iopub.status.idle":"2023-06-28T23:28:25.187348Z","shell.execute_reply.started":"2023-06-28T23:28:25.179472Z","shell.execute_reply":"2023-06-28T23:28:25.186471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating multilabels\n\nInstead of predicting a single label, we will change our target to be a multilabel problem; i.e., if the target is a certain class, then it encompasses all the classes before it. E.g. encoding a class 4 retinopathy would usually be `[0, 0, 0, 1]`, but in our case we will predict `[1, 1, 1, 1]`. For more details, please check out [Lex's kernel](https://www.kaggle.com/lextoumbourou/blindness-detection-resnet34-ordinal-targets).","metadata":{}},{"cell_type":"code","source":"y_train_multi = np.empty(y_train.shape, dtype=y_train.dtype)\ny_train_multi[:, 4] = y_train[:, 4]\n\nfor i in range(3, -1, -1):\n    y_train_multi[:, i] = np.logical_or(y_train[:, i], y_train_multi[:, i+1])\n\ny_val_multi = np.empty(y_val.shape, dtype=y_val.dtype)\ny_val_multi[:, 4] = y_val[:, 4]\n\nfor i in range(3, -1, -1):\n    y_val_multi[:, i] = np.logical_or(y_val[:, i], y_val_multi[:, i+1])\n\nprint(\"Y_train multi: {}\".format(y_train_multi.shape))\nprint(\"Y_val multi: {}\".format(y_val_multi.shape))","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:28:25.188868Z","iopub.execute_input":"2023-06-28T23:28:25.189425Z","iopub.status.idle":"2023-06-28T23:28:25.200937Z","shell.execute_reply.started":"2023-06-28T23:28:25.189371Z","shell.execute_reply":"2023-06-28T23:28:25.199878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = y_train_multi\ny_val = y_val_multi","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:28:25.202444Z","iopub.execute_input":"2023-06-28T23:28:25.203031Z","iopub.status.idle":"2023-06-28T23:28:25.212592Z","shell.execute_reply.started":"2023-06-28T23:28:25.20283Z","shell.execute_reply":"2023-06-28T23:28:25.211571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# delete the uneeded df\ndel new_train\ndel old_train\ndel val_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:28:25.21427Z","iopub.execute_input":"2023-06-28T23:28:25.214843Z","iopub.status.idle":"2023-06-28T23:28:25.347366Z","shell.execute_reply.started":"2023-06-28T23:28:25.214651Z","shell.execute_reply":"2023-06-28T23:28:25.346715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating keras callback for QWK\n\n---\n\nI had to change this function, in order to consider the best kappa score among all the buckets.","metadata":{}},{"cell_type":"code","source":"class Metrics(Callback):\n\n    def on_epoch_end(self, epoch, logs={}):\n        X_val, y_val = self.validation_data[:2]\n        y_val = y_val.sum(axis=1) - 1\n        \n        y_pred = self.model.predict(X_val) > 0.5\n        y_pred = y_pred.astype(int).sum(axis=1) - 1\n\n        _val_kappa = cohen_kappa_score(\n            y_val,\n            y_pred, \n            weights='quadratic'\n        )\n\n        self.val_kappas.append(_val_kappa)\n\n        print(f\"val_kappa: {_val_kappa:.4f}\")\n        \n        if _val_kappa == max(self.val_kappas):\n            print(\"Validation Kappa has improved. Saving model.\")\n            self.model.save('model.h5')\n\n        return","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:28:25.348947Z","iopub.execute_input":"2023-06-28T23:28:25.349608Z","iopub.status.idle":"2023-06-28T23:28:25.359138Z","shell.execute_reply.started":"2023-06-28T23:28:25.349353Z","shell.execute_reply":"2023-06-28T23:28:25.358168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Generator","metadata":{}},{"cell_type":"markdown","source":"Create new data from old data to make the training dataset more bigger\n\ntype of augmentation \n* change in flip |(horizontal, vertical)\n* change zoom\n* change brightness \n* etc","metadata":{}},{"cell_type":"code","source":"def create_datagen():\n    return ImageDataGenerator(\n        horizontal_flip=True,\n        vertical_flip=True,\n        zoom_range= 0.3,\n        brightness_range=(0.5, 2),\n        fill_mode='constant',\n        cval=0\n    )","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:28:25.360493Z","iopub.execute_input":"2023-06-28T23:28:25.360968Z","iopub.status.idle":"2023-06-28T23:28:25.375487Z","shell.execute_reply.started":"2023-06-28T23:28:25.360919Z","shell.execute_reply":"2023-06-28T23:28:25.374835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check the differenct kinds of augmentations on the pictures.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 10, figsize=(20, 10))\nax = ax.ravel()\n\nimg = x_val[0].reshape(1,x_val[0].shape[0],x_val[0].shape[1], x_val[0].shape[2])\n\nax[0].imshow(img[0].astype('uint8'))\nax[1].imshow(next(ImageDataGenerator().flow(img))[0].astype('uint8'))\nax[2].imshow(next(ImageDataGenerator(horizontal_flip=True, fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[3].imshow(next(ImageDataGenerator(vertical_flip=True,fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[4].imshow(next(ImageDataGenerator(rotation_range=360, fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[5].imshow(next(ImageDataGenerator(zoom_range= (0.65,1), fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[6].imshow(next(ImageDataGenerator(height_shift_range=0.15, fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[7].imshow(next(ImageDataGenerator(width_shift_range=0.15, fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[8].imshow(next(ImageDataGenerator(brightness_range=(0.5, 2), fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[9].imshow(next(ImageDataGenerator(horizontal_flip=True,\n                                     vertical_flip=True,\n                                     rotation_range=360,zoom_range= (0.65,1),\n                                     brightness_range=(0.5, 2),\n                                     fill_mode='constant',cval=0).flow(img))[0].astype('uint8'))\n","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:28:25.378159Z","iopub.execute_input":"2023-06-28T23:28:25.379103Z","iopub.status.idle":"2023-06-28T23:28:27.249541Z","shell.execute_reply.started":"2023-06-28T23:28:25.378993Z","shell.execute_reply":"2023-06-28T23:28:27.248671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model: EfficientNetB3","metadata":{}},{"cell_type":"markdown","source":" ### imports the EfficientNetB3 model from the keras library and loads the pre-trained weights of the model.","metadata":{}},{"cell_type":"code","source":"efficientnetb3 = EfficientNetB3(\n        weights=None,\n        input_shape=(HEIGHT,WIDTH,3),\n        include_top=False\n                   )\n\nefficientnetb3.load_weights(\"../input/efficientnet-keras-weights-b0b5/efficientnet-b3_imagenet_1000_notop.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:28:27.251065Z","iopub.execute_input":"2023-06-28T23:28:27.251524Z","iopub.status.idle":"2023-06-28T23:28:46.447272Z","shell.execute_reply.started":"2023-06-28T23:28:27.251472Z","shell.execute_reply":"2023-06-28T23:28:46.446288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model():\n    model = Sequential()\n    model.add(efficientnetb3)\n    model.add(layers.GlobalAveragePooling2D())\n    model.add(layers.Dropout(0.5))\n    model.add(layers.BatchNormalization())\n    model.add(layers.Dense(5, activation='sigmoid'))\n    \n    model.compile(\n        loss='binary_crossentropy',\n        #loss=kappa_loss,\n        optimizer=Adam(lr=1e-4,decay=1e-6),\n        metrics=['accuracy']\n    )\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:28:46.448916Z","iopub.execute_input":"2023-06-28T23:28:46.449363Z","iopub.status.idle":"2023-06-28T23:28:46.457638Z","shell.execute_reply.started":"2023-06-28T23:28:46.449312Z","shell.execute_reply":"2023-06-28T23:28:46.457032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = build_model()\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:28:46.458912Z","iopub.execute_input":"2023-06-28T23:28:46.459464Z","iopub.status.idle":"2023-06-28T23:28:53.642547Z","shell.execute_reply.started":"2023-06-28T23:28:46.459412Z","shell.execute_reply":"2023-06-28T23:28:53.641742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pretraining with old Data","metadata":{}},{"cell_type":"code","source":"bucket_num = 8\ndiv = round(train_df.shape[0]/bucket_num)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:28:53.64403Z","iopub.execute_input":"2023-06-28T23:28:53.644372Z","iopub.status.idle":"2023-06-28T23:28:53.649562Z","shell.execute_reply.started":"2023-06-28T23:28:53.644287Z","shell.execute_reply":"2023-06-28T23:28:53.648676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_init = {\n    'val_loss': [0.0],\n    'val_acc': [0.0],\n    'loss': [0.0], \n    'acc': [0.0],\n    'bucket': [0.0]\n}\nresults = pd.DataFrame(df_init)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:28:53.650718Z","iopub.execute_input":"2023-06-28T23:28:53.651004Z","iopub.status.idle":"2023-06-28T23:28:53.661868Z","shell.execute_reply.started":"2023-06-28T23:28:53.650957Z","shell.execute_reply":"2023-06-28T23:28:53.66121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I found that changing the nr. of epochs for each bucket helped in terms of performances\nepochs = [5,5,5,5,5,5,5,5]\nkappa_metrics = Metrics()\nkappa_metrics.val_kappas = []\n\nlearn_control = ReduceLROnPlateau(monitor='val_acc', patience=5,\n                                  verbose=1,factor=.2, min_lr=1e-7)\n\ncheckpoint = ModelCheckpoint('val_model.h5', monitor='val_loss', verbose=1, save_best_only=True, mode='min')","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:28:53.663662Z","iopub.execute_input":"2023-06-28T23:28:53.664261Z","iopub.status.idle":"2023-06-28T23:28:53.674661Z","shell.execute_reply.started":"2023-06-28T23:28:53.664039Z","shell.execute_reply":"2023-06-28T23:28:53.673954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code \n\n1. It creates a data generator that generates augmented image batches from the images in the current bucket.\n2. It fits the model to the current data using the fit_generator method.\n3. It saves the history of the current fit in a DataFrame df_model and appends it to the results DataFrame.\n4. It deletes the data generator and the training data to save memory.","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"for i in range(0,bucket_num):\n    if i != (bucket_num-1):\n        print(\"Bucket Nr: {}\".format(i))\n        \n        N = train_df.iloc[i*div:(1+i)*div].shape[0]\n        x_train = np.empty((N, HEIGHT, WIDTH, 3), dtype=np.uint8)\n        for j, image_id in enumerate(tqdm_notebook(train_df.iloc[i*div:(1+i)*div,0])):\n            x_train[j, :, :, :] = preprocess_image(f'{image_id}', height=HEIGHT, width=WIDTH)\n\n        data_generator = create_datagen().flow(x_train, y_train[i*div:(1+i)*div,:], batch_size=BATCH_SIZE)\n        history = model.fit_generator(\n                        data_generator,\n                        steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                        epochs=epochs[i],\n                        validation_data=(x_val, y_val),\n                        callbacks=[kappa_metrics, learn_control, checkpoint]\n                        )\n        \n        dic = history.history\n        df_model = pd.DataFrame(dic)\n        df_model['bucket'] = i\n    else:\n        print(\"Bucket Nr: {}\".format(i))\n        \n        N = train_df.iloc[i*div:].shape[0]\n        x_train = np.empty((N, HEIGHT, WIDTH, 3), dtype=np.uint8)\n        for j, image_id in enumerate(tqdm_notebook(train_df.iloc[i*div:,0])):\n            x_train[j, :, :, :] = preprocess_image(f'{image_id}', height=HEIGHT, width=WIDTH)\n        data_generator = create_datagen().flow(x_train, y_train[i*div:,:], batch_size=BATCH_SIZE)\n        \n        history = model.fit_generator(\n                        data_generator,\n                        steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                        epochs=epochs[i],\n                        validation_data=(x_val, y_val),\n                        callbacks=[kappa_metrics, learn_control, checkpoint]\n                        )\n        \n        dic = history.history\n        df_model = pd.DataFrame(dic)\n        df_model['bucket'] = i\n\n    results = results.append(df_model)\n    \n    del data_generator\n    del x_train\n    gc.collect()\n    \n    print('-'*40)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:28:53.67649Z","iopub.execute_input":"2023-06-28T23:28:53.676856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = results.iloc[1:]\nresults['kappa'] = kappa_metrics.val_kappas\nresults = results.reset_index()\nresults = results.rename(index=str, columns={\"index\": \"epoch\"})\nresults","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results[['loss', 'val_loss']].plot()\nresults[['acc', 'val_acc']].plot()\nresults[['kappa']].plot()\nresults.to_csv('model_results.csv',index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fine Tune with new Data\nCreate New Train and Validation Set to finetune our model","metadata":{}},{"cell_type":"code","source":"model.load_weights('val_model.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_val, y_train, y_val = train_test_split(\n    x_val, y_val, \n    test_size=0.2, \n    random_state=nr_seed\n)\n\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_generator = create_datagen().flow(x_train, y_train, batch_size=BATCH_SIZE)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit_generator(\n                data_generator,\n                steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                epochs=20,\n                validation_data=(x_val, y_val),\n                callbacks=[kappa_metrics,learn_control,checkpoint]\n                )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_weights('val_model.h5')\npred_val = model.predict(x_val)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Optimize Threshold","metadata":{}},{"cell_type":"code","source":"def compute_score_inv(threshold):\n    y1 = pred_val > threshold\n    y1 = y1.astype(int).sum(axis=1) - 1\n    y2 = y_val.sum(axis=1) - 1\n    score = cohen_kappa_score(y1, y2, weights='quadratic')\n    return 1 - score\nsimplex = scipy.optimize.minimize(compute_score_inv, 0.5, method='nelder-mead')\n\nbest_threshold = simplex['x'][0]\nprint(best_threshold)\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}