{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# EfficientNetB3Trained with Old and New Data\n\n\n---","metadata":{}},{"cell_type":"markdown","source":"The Inference kernel of this notebook can be found here: https://www.kaggle.com/fanconic/efficientnetb3-inference-keras?scriptVersionId=18596729\n\n - LB Score: 0.786\n - Private Score: 0.910","metadata":{}},{"cell_type":"code","source":"# To have reproducible results and compare them\nnr_seed = 11\nimport numpy as np \nnp.random.seed(nr_seed)\nimport tensorflow as tf\ntf.set_random_seed(nr_seed)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:03:23.794874Z","iopub.execute_input":"2023-03-21T03:03:23.795177Z","iopub.status.idle":"2023-03-21T03:03:25.659885Z","shell.execute_reply.started":"2023-03-21T03:03:23.795121Z","shell.execute_reply":"2023-03-21T03:03:25.658104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import libraries\n!pip install -U '../input/install/efficientnet-0.0.3-py2.py3-none-any.whl'\nimport json\nimport math\nfrom tqdm import tqdm, tqdm_notebook\nimport gc\nimport warnings\nimport os\n\nimport cv2\nfrom PIL import Image\n\nimport pandas as pd\nimport scipy\nimport matplotlib.pyplot as plt\n\nfrom keras import backend as K\nfrom keras import layers\nfrom efficientnet import EfficientNetB3\nfrom keras.callbacks import Callback, ModelCheckpoint, ReduceLROnPlateau\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential\nfrom keras.optimizers import Adam\nfrom keras.losses import binary_crossentropy, categorical_crossentropy\nfrom skimage.color import rgb2hsv, lab2lch\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score\n\nwarnings.filterwarnings(\"ignore\")\n\n%matplotlib inline","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-03-21T03:03:25.662866Z","iopub.execute_input":"2023-03-21T03:03:25.663131Z","iopub.status.idle":"2023-03-21T03:03:33.640066Z","shell.execute_reply.started":"2023-03-21T03:03:25.663077Z","shell.execute_reply":"2023-03-21T03:03:33.639110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image size\nWIDTH= 320\nHEIGHT = 320\n# Batch size\nBATCH_SIZE = 32","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:03:33.642265Z","iopub.execute_input":"2023-03-21T03:03:33.642599Z","iopub.status.idle":"2023-03-21T03:03:33.653815Z","shell.execute_reply.started":"2023-03-21T03:03:33.642519Z","shell.execute_reply":"2023-03-21T03:03:33.653027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading & Merging","metadata":{}},{"cell_type":"code","source":"new_train = pd.read_csv('../input/aptos2019-blindness-detection/train.csv')\nold_train = pd.read_csv('../input/diabetic-retinopathy-resized/trainLabels_cropped.csv')\nduplicates = pd.read_csv('../input/aptos-trained-weights/inconsistent.csv')\nprint(new_train.shape)\nprint(old_train.shape)\nprint(duplicates.shape)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:03:33.656006Z","iopub.execute_input":"2023-03-21T03:03:33.656531Z","iopub.status.idle":"2023-03-21T03:03:33.756180Z","shell.execute_reply.started":"2023-03-21T03:03:33.656468Z","shell.execute_reply":"2023-03-21T03:03:33.755329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for img_name in duplicates['id_code'].values:\n    new_train = new_train[new_train['id_code'] != img_name]\nprint(new_train.shape)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:03:33.757612Z","iopub.execute_input":"2023-03-21T03:03:33.757947Z","iopub.status.idle":"2023-03-21T03:03:33.829635Z","shell.execute_reply.started":"2023-03-21T03:03:33.757883Z","shell.execute_reply":"2023-03-21T03:03:33.828830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"old_train = old_train[['image','level']]\nold_train.columns = new_train.columns\nold_train.diagnosis.value_counts()\n\n# path columns\nnew_train['id_code'] = '../input/aptos2019-blindness-detection/train_images/' + new_train['id_code'].astype(str) + '.png'\nold_train['id_code'] = '../input/diabetic-retinopathy-resized/resized_train/resized_train/' + old_train['id_code'].astype(str) + '.jpeg'\n\ntrain_df = old_train.copy()\nval_df = new_train.copy()\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:03:33.831032Z","iopub.execute_input":"2023-03-21T03:03:33.831554Z","iopub.status.idle":"2023-03-21T03:03:33.887039Z","shell.execute_reply.started":"2023-03-21T03:03:33.831499Z","shell.execute_reply":"2023-03-21T03:03:33.886075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train - Valid split\nUse new Data for validation and Old data for training£","metadata":{}},{"cell_type":"code","source":"# Let's shuffle the datasets\ntrain_df = train_df.sample(frac=1).reset_index(drop=True)\nval_df = val_df.sample(frac=1).reset_index(drop=True)\nprint(train_df.shape)\nprint(val_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:03:33.890232Z","iopub.execute_input":"2023-03-21T03:03:33.890493Z","iopub.status.idle":"2023-03-21T03:03:33.903872Z","shell.execute_reply.started":"2023-03-21T03:03:33.890436Z","shell.execute_reply":"2023-03-21T03:03:33.902680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Process Images","metadata":{}},{"cell_type":"markdown","source":"Crop function: https://www.kaggle.com/ratthachat/aptos-updated-preprocessing-ben-s-cropping ","metadata":{}},{"cell_type":"code","source":"def crop_image1(img,tol=7):\n    # img is image data\n    # tol  is tolerance\n        \n    mask = img>tol\n    return img[np.ix_(mask.any(1),mask.any(0))]\n\ndef crop_image_from_gray(img,tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n            img = np.stack([img1,img2,img3],axis=-1)\n   \n        return img\n\n\n# Make all images circular (possible data loss)\ndef circle_crop(img):   \n    \"\"\"\n    Create circular crop around image centre    \n    \"\"\"    \n    \n    img = crop_image_from_gray(img)    \n    \n    height, width, depth = img.shape    \n    \n    x = int(width/2)\n    y = int(height/2)\n    r = np.amin((x,y))\n    \n    circle_img = np.zeros((height, width), np.uint8)\n    cv2.circle(circle_img, (x,y), int(r), 1, thickness=-1)\n    img = cv2.bitwise_and(img, img, mask=circle_img)\n    img = crop_image_from_gray(img)\n    \n    return img \n\ndef preprocess_image(image_path, width=320, height=320, new_data=False):\n    img = cv2.imread(image_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)    \n    if new_data:\n        img = crop_image_from_gray(img)\n    img = cv2.resize(img, (width,height))\n    #img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), 20) ,-4 ,128)\n\n    return img","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:03:33.905215Z","iopub.execute_input":"2023-03-21T03:03:33.905773Z","iopub.status.idle":"2023-03-21T03:03:33.920677Z","shell.execute_reply.started":"2023-03-21T03:03:33.905467Z","shell.execute_reply":"2023-03-21T03:03:33.919806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_samples(df, columns=4, rows=3):\n    fig=plt.figure(figsize=(5*columns, 4*rows))\n\n    for i in range(columns*rows):\n        image_path = df.loc[i,'id_code']\n        image_id = df.loc[i,'diagnosis']\n        img = preprocess_image(f'{image_path}', width=WIDTH, height=HEIGHT)\n        fig.add_subplot(rows, columns, i+1)\n        plt.title(image_id)\n        plt.imshow(img)\n    \n    plt.tight_layout()\n\ndisplay_samples(train_df)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-03-21T03:03:33.922286Z","iopub.execute_input":"2023-03-21T03:03:33.922707Z","iopub.status.idle":"2023-03-21T03:03:37.468846Z","shell.execute_reply.started":"2023-03-21T03:03:33.922639Z","shell.execute_reply":"2023-03-21T03:03:37.468058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Processing Images","metadata":{}},{"cell_type":"markdown","source":"__UPDATE:__ Here we are reading just the validation set. In order to use 320x320 images, we are going to load one bucket at a time only when needed. This will let our code run without memory-related errors.","metadata":{}},{"cell_type":"code","source":"# validation set\nN = val_df.shape[0]\nx_val = np.empty((N, HEIGHT, WIDTH, 3), dtype=np.uint8)\n\nfor i, image_id in enumerate(tqdm_notebook(val_df['id_code'])):\n    x_val[i, :, :, :] = preprocess_image(\n        f'{image_id}',\n        height=HEIGHT, width=WIDTH, new_data=True\n    )","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:03:37.469888Z","iopub.execute_input":"2023-03-21T03:03:37.470138Z","iopub.status.idle":"2023-03-21T03:15:32.299190Z","shell.execute_reply.started":"2023-03-21T03:03:37.470100Z","shell.execute_reply":"2023-03-21T03:15:32.298491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = pd.get_dummies(train_df['diagnosis']).values\ny_val = pd.get_dummies(val_df['diagnosis']).values\n\nprint(y_train.shape)\nprint(x_val.shape)\nprint(y_val.shape)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:15:32.303062Z","iopub.execute_input":"2023-03-21T03:15:32.305357Z","iopub.status.idle":"2023-03-21T03:15:32.322492Z","shell.execute_reply.started":"2023-03-21T03:15:32.305290Z","shell.execute_reply":"2023-03-21T03:15:32.321661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating multilabels\n\nInstead of predicting a single label, we will change our target to be a multilabel problem; i.e., if the target is a certain class, then it encompasses all the classes before it. E.g. encoding a class 4 retinopathy would usually be `[0, 0, 0, 1]`, but in our case we will predict `[1, 1, 1, 1]`. For more details, please check out [Lex's kernel](https://www.kaggle.com/lextoumbourou/blindness-detection-resnet34-ordinal-targets).","metadata":{}},{"cell_type":"code","source":"y_train_multi = np.empty(y_train.shape, dtype=y_train.dtype)\ny_train_multi[:, 4] = y_train[:, 4]\n\nfor i in range(3, -1, -1):\n    y_train_multi[:, i] = np.logical_or(y_train[:, i], y_train_multi[:, i+1])\n\ny_val_multi = np.empty(y_val.shape, dtype=y_val.dtype)\ny_val_multi[:, 4] = y_val[:, 4]\n\nfor i in range(3, -1, -1):\n    y_val_multi[:, i] = np.logical_or(y_val[:, i], y_val_multi[:, i+1])\n\nprint(\"Y_train multi: {}\".format(y_train_multi.shape))\nprint(\"Y_val multi: {}\".format(y_val_multi.shape))","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:15:32.329206Z","iopub.execute_input":"2023-03-21T03:15:32.331245Z","iopub.status.idle":"2023-03-21T03:15:32.346108Z","shell.execute_reply.started":"2023-03-21T03:15:32.331194Z","shell.execute_reply":"2023-03-21T03:15:32.345200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = y_train_multi\ny_val = y_val_multi","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:15:32.350460Z","iopub.execute_input":"2023-03-21T03:15:32.352505Z","iopub.status.idle":"2023-03-21T03:15:32.357942Z","shell.execute_reply.started":"2023-03-21T03:15:32.352454Z","shell.execute_reply":"2023-03-21T03:15:32.357059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# delete the uneeded df\ndel new_train\ndel old_train\ndel val_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:15:32.362255Z","iopub.execute_input":"2023-03-21T03:15:32.364398Z","iopub.status.idle":"2023-03-21T03:15:32.538448Z","shell.execute_reply.started":"2023-03-21T03:15:32.364346Z","shell.execute_reply":"2023-03-21T03:15:32.537671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating keras callback for QWK\n\n---\n\nI had to change this function, in order to consider the best kappa score among all the buckets.","metadata":{}},{"cell_type":"code","source":"class Metrics(Callback):\n\n    def on_epoch_end(self, epoch, logs={}):\n        X_val, y_val = self.validation_data[:2]\n        y_val = y_val.sum(axis=1) - 1\n        \n        y_pred = self.model.predict(X_val) > 0.5\n        y_pred = y_pred.astype(int).sum(axis=1) - 1\n\n        _val_kappa = cohen_kappa_score(\n            y_val,\n            y_pred, \n            weights='quadratic'\n        )\n\n        self.val_kappas.append(_val_kappa)\n\n        print(f\"val_kappa: {_val_kappa:.4f}\")\n        \n        if _val_kappa == max(self.val_kappas):\n            print(\"Validation Kappa has improved. Saving model.\")\n            self.model.save('model.h5')\n\n        return","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:15:32.542729Z","iopub.execute_input":"2023-03-21T03:15:32.544748Z","iopub.status.idle":"2023-03-21T03:15:32.559663Z","shell.execute_reply.started":"2023-03-21T03:15:32.544695Z","shell.execute_reply":"2023-03-21T03:15:32.558774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Generator","metadata":{}},{"cell_type":"code","source":"def create_datagen():\n    return ImageDataGenerator(\n        horizontal_flip=True,\n        vertical_flip=True,\n        zoom_range= 0.3,\n        brightness_range=(0.5, 2),\n        fill_mode='constant',\n        cval=0\n    )","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:15:32.563823Z","iopub.execute_input":"2023-03-21T03:15:32.566538Z","iopub.status.idle":"2023-03-21T03:15:32.573334Z","shell.execute_reply.started":"2023-03-21T03:15:32.566491Z","shell.execute_reply":"2023-03-21T03:15:32.572548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check the differenct kinds of augmentations on the pictures.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 10, figsize=(20, 10))\nax = ax.ravel()\n\nimg = x_val[0].reshape(1,x_val[0].shape[0],x_val[0].shape[1], x_val[0].shape[2])\n\nax[0].imshow(img[0].astype('uint8'))\nax[1].imshow(next(ImageDataGenerator().flow(img))[0].astype('uint8'))\nax[2].imshow(next(ImageDataGenerator(horizontal_flip=True, fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[3].imshow(next(ImageDataGenerator(vertical_flip=True,fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[4].imshow(next(ImageDataGenerator(rotation_range=360, fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[5].imshow(next(ImageDataGenerator(zoom_range= (0.65,1), fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[6].imshow(next(ImageDataGenerator(height_shift_range=0.15, fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[7].imshow(next(ImageDataGenerator(width_shift_range=0.15, fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[8].imshow(next(ImageDataGenerator(brightness_range=(0.5, 2), fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[9].imshow(next(ImageDataGenerator(horizontal_flip=True,\n                                     vertical_flip=True,\n                                     rotation_range=360,zoom_range= (0.65,1),\n                                     brightness_range=(0.5, 2),\n                                     fill_mode='constant',cval=0).flow(img))[0].astype('uint8'))\n","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:15:32.577693Z","iopub.execute_input":"2023-03-21T03:15:32.580372Z","iopub.status.idle":"2023-03-21T03:15:34.413076Z","shell.execute_reply.started":"2023-03-21T03:15:32.580324Z","shell.execute_reply":"2023-03-21T03:15:34.412043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model: EfficientNetB3","metadata":{}},{"cell_type":"code","source":"efficientnetb3 = EfficientNetB3(\n        weights=None,\n        input_shape=(HEIGHT,WIDTH,3),\n        include_top=False\n                   )\n\nefficientnetb3.load_weights(\"../input/efficientnet-keras-weights-b0b5/efficientnet-b3_imagenet_1000_notop.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:15:34.417439Z","iopub.execute_input":"2023-03-21T03:15:34.418140Z","iopub.status.idle":"2023-03-21T03:15:52.030907Z","shell.execute_reply.started":"2023-03-21T03:15:34.418074Z","shell.execute_reply":"2023-03-21T03:15:52.030013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model():\n    model = Sequential()\n    model.add(efficientnetb3)\n    model.add(layers.GlobalAveragePooling2D())\n    model.add(layers.Dropout(0.4))\n    model.add(layers.BatchNormalization())\n    model.add(layers.Dense(5, activation='softmax'))\n    \n    model.compile(\n        loss='binary_crossentropy',\n        #loss=kappa_loss,\n        optimizer=Adam(lr=1e-5,decay=0.5e-6),\n        metrics=['accuracy']\n    )\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:15:52.033732Z","iopub.execute_input":"2023-03-21T03:15:52.034072Z","iopub.status.idle":"2023-03-21T03:15:52.041312Z","shell.execute_reply.started":"2023-03-21T03:15:52.034022Z","shell.execute_reply":"2023-03-21T03:15:52.040592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = build_model()\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:15:52.042932Z","iopub.execute_input":"2023-03-21T03:15:52.043580Z","iopub.status.idle":"2023-03-21T03:15:58.688459Z","shell.execute_reply.started":"2023-03-21T03:15:52.043516Z","shell.execute_reply":"2023-03-21T03:15:58.687603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pretraining with old Data","metadata":{}},{"cell_type":"code","source":"bucket_num = 8\ndiv = round(train_df.shape[0]/bucket_num)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:15:58.689980Z","iopub.execute_input":"2023-03-21T03:15:58.690532Z","iopub.status.idle":"2023-03-21T03:15:58.695516Z","shell.execute_reply.started":"2023-03-21T03:15:58.690472Z","shell.execute_reply":"2023-03-21T03:15:58.694704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_init = {\n    'val_loss': [0.0],\n    'val_acc': [0.0],\n    'loss': [0.0], \n    'acc': [0.0],\n    'bucket': [0.0]\n}\nresults = pd.DataFrame(df_init)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:15:58.697189Z","iopub.execute_input":"2023-03-21T03:15:58.697783Z","iopub.status.idle":"2023-03-21T03:15:58.708511Z","shell.execute_reply.started":"2023-03-21T03:15:58.697732Z","shell.execute_reply":"2023-03-21T03:15:58.707644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{"execution":{"iopub.status.busy":"2023-03-19T18:13:07.638070Z","iopub.execute_input":"2023-03-19T18:13:07.638514Z","iopub.status.idle":"2023-03-19T18:13:07.645982Z","shell.execute_reply.started":"2023-03-19T18:13:07.638333Z","shell.execute_reply":"2023-03-19T18:13:07.644886Z"}}},{"cell_type":"code","source":"#I found that changing the nr. of epochs for each bucket helped in terms of performances\nepochs = [5,5,4,4,5,4,5,4]\nkappa_metrics = Metrics()\nkappa_metrics.val_kappas = []\n\nlearn_control = ReduceLROnPlateau(monitor='val_acc', patience=3,\n                                  verbose=1,factor=.2, min_lr=0.5e-8)\n\ncheckpoint = ModelCheckpoint('val_model.h5', monitor='val_loss', verbose=1, save_best_only=True, mode='min')","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:15:58.709859Z","iopub.execute_input":"2023-03-21T03:15:58.710544Z","iopub.status.idle":"2023-03-21T03:15:58.718214Z","shell.execute_reply.started":"2023-03-21T03:15:58.710494Z","shell.execute_reply":"2023-03-21T03:15:58.717368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(0,bucket_num):\n    if i != (bucket_num-1):\n        print(\"Bucket Nr: {}\".format(i))\n        \n        N = train_df.iloc[i*div:(1+i)*div].shape[0]\n        x_train = np.empty((N, HEIGHT, WIDTH, 3), dtype=np.uint8)\n        for j, image_id in enumerate(tqdm_notebook(train_df.iloc[i*div:(1+i)*div,0])):\n            x_train[j, :, :, :] = preprocess_image(f'{image_id}', height=HEIGHT, width=WIDTH)\n\n        data_generator = create_datagen().flow(x_train, y_train[i*div:(1+i)*div,:], batch_size=BATCH_SIZE, shuffle=False)\n        history = model.fit_generator(\n                        data_generator,\n                        steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                        epochs=epochs[i],\n                        validation_data=(x_val, y_val),\n                        callbacks=[kappa_metrics, learn_control, checkpoint]\n                        )\n        \n        dic = history.history\n        df_model = pd.DataFrame(dic)\n        df_model['bucket'] = i\n    else:\n        print(\"Bucket Nr: {}\".format(i))\n        \n        N = train_df.iloc[i*div:].shape[0]\n        x_train = np.empty((N, HEIGHT, WIDTH, 3), dtype=np.uint8)\n        for j, image_id in enumerate(tqdm_notebook(train_df.iloc[i*div:,0])):\n            x_train[j, :, :, :] = preprocess_image(f'{image_id}', height=HEIGHT, width=WIDTH)\n        data_generator = create_datagen().flow(x_train, y_train[i*div:,:], batch_size=BATCH_SIZE, shuffle=False)\n        \n        history = model.fit_generator(\n                        data_generator,\n                        steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                        epochs=epochs[i],\n                        validation_data=(x_val, y_val),\n                        callbacks=[kappa_metrics, learn_control, checkpoint]\n                        )\n           \n        dic = history.history\n        df_model = pd.DataFrame(dic)\n        df_model['bucket'] = i\n\n    results = results.append(df_model)\n    \n    del data_generator\n    del x_train\n    gc.collect()\n    \n    print('-'*40)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:15:58.719767Z","iopub.execute_input":"2023-03-21T03:15:58.720498Z","iopub.status.idle":"2023-03-21T05:27:39.330112Z","shell.execute_reply.started":"2023-03-21T03:15:58.720292Z","shell.execute_reply":"2023-03-21T05:27:39.329253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = results.iloc[1:]\nresults['kappa'] = kappa_metrics.val_kappas\nresults = results.reset_index()\nresults = results.rename(index=str, columns={\"index\": \"epoch\"})\nresults","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:27:39.331522Z","iopub.execute_input":"2023-03-21T05:27:39.332022Z","iopub.status.idle":"2023-03-21T05:27:39.359472Z","shell.execute_reply.started":"2023-03-21T05:27:39.331960Z","shell.execute_reply":"2023-03-21T05:27:39.358361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results[['loss', 'val_loss']].plot()\nresults[['acc', 'val_acc']].plot()\nresults[['kappa']].plot()\nresults.to_csv('model_results.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:27:39.361060Z","iopub.execute_input":"2023-03-21T05:27:39.361496Z","iopub.status.idle":"2023-03-21T05:27:40.488879Z","shell.execute_reply.started":"2023-03-21T05:27:39.361328Z","shell.execute_reply":"2023-03-21T05:27:40.487988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results[['loss', 'val_loss']].plot(figsize=(20, 16), fontsize=26, linewidth = 5)\n\nplt.savefig('loss_val_loss.png', dpi=300)\nplt.savefig('loss_val_loss.eps', dpi=300)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:27:40.490395Z","iopub.execute_input":"2023-03-21T05:27:40.491954Z","iopub.status.idle":"2023-03-21T05:27:42.861256Z","shell.execute_reply.started":"2023-03-21T05:27:40.491888Z","shell.execute_reply":"2023-03-21T05:27:42.860529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results[['acc', 'val_acc']].plot(figsize=(20, 16), fontsize=26, linewidth = 5)\n\nplt.savefig('acc_val_acc.png', dpi=300)\nplt.savefig('acc_val_acc.eps', dpi=300)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:27:42.862604Z","iopub.execute_input":"2023-03-21T05:27:42.863121Z","iopub.status.idle":"2023-03-21T05:27:45.068249Z","shell.execute_reply.started":"2023-03-21T05:27:42.863066Z","shell.execute_reply":"2023-03-21T05:27:45.067506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results[['kappa']].plot(figsize=(20, 16), fontsize=26, linewidth = 5)\n\nplt.savefig('kappa.png', dpi=300)\nplt.savefig('kappa.eps', dpi=300)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:27:45.069631Z","iopub.execute_input":"2023-03-21T05:27:45.070125Z","iopub.status.idle":"2023-03-21T05:27:47.162316Z","shell.execute_reply.started":"2023-03-21T05:27:45.070073Z","shell.execute_reply":"2023-03-21T05:27:47.161558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fine Tune with new Data\nCreate New Train and Validation Set to finetune our model","metadata":{}},{"cell_type":"code","source":"model.load_weights('val_model.h5')","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:48:07.232155Z","iopub.execute_input":"2023-03-21T05:48:07.232468Z","iopub.status.idle":"2023-03-21T05:48:09.909415Z","shell.execute_reply.started":"2023-03-21T05:48:07.232415Z","shell.execute_reply":"2023-03-21T05:48:09.908596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_val, y_train, y_val = train_test_split(\n    x_val, y_val, \n    test_size=0.2, \n    random_state=nr_seed\n)\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:48:12.601368Z","iopub.execute_input":"2023-03-21T05:48:12.601698Z","iopub.status.idle":"2023-03-21T05:48:13.764477Z","shell.execute_reply.started":"2023-03-21T05:48:12.601641Z","shell.execute_reply":"2023-03-21T05:48:13.763585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data_generator = create_datagen().flow(x_train, y_train, batch_size=BATCH_SIZE, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-03-19T20:37:20.390788Z","iopub.execute_input":"2023-03-19T20:37:20.391271Z","iopub.status.idle":"2023-03-19T20:37:22.851159Z","shell.execute_reply.started":"2023-03-19T20:37:20.391049Z","shell.execute_reply":"2023-03-19T20:37:22.850371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# history = model.fit_generator(\n#                 data_generator,\n#                 steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n#                 epochs=20,\n#                 validation_data=(x_val, y_val),\n#                 callbacks=[kappa_metrics,learn_control,checkpoint]\n#                 )","metadata":{"execution":{"iopub.status.busy":"2023-03-19T20:37:22.854863Z","iopub.execute_input":"2023-03-19T20:37:22.855113Z","iopub.status.idle":"2023-03-19T21:13:08.401334Z","shell.execute_reply.started":"2023-03-19T20:37:22.855068Z","shell.execute_reply":"2023-03-19T21:13:08.400347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_weights('val_model.h5')","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:48:23.332606Z","iopub.execute_input":"2023-03-21T05:48:23.332918Z","iopub.status.idle":"2023-03-21T05:48:23.586323Z","shell.execute_reply.started":"2023-03-21T05:48:23.332866Z","shell.execute_reply":"2023-03-21T05:48:23.585413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"res = model.evaluate(x_val, y_val)\nprint(\"Testing accuracy : \" + str(res[1]))\nprint(\"Testing loss : \" + str(res[0]))","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:48:26.274842Z","iopub.execute_input":"2023-03-21T05:48:26.275138Z","iopub.status.idle":"2023-03-21T05:48:30.848134Z","shell.execute_reply.started":"2023-03-21T05:48:26.275088Z","shell.execute_reply":"2023-03-21T05:48:30.847307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_val = model.predict(x_val)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:48:32.851750Z","iopub.execute_input":"2023-03-21T05:48:32.852065Z","iopub.status.idle":"2023-03-21T05:48:37.014732Z","shell.execute_reply.started":"2023-03-21T05:48:32.852004Z","shell.execute_reply":"2023-03-21T05:48:37.013706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_val","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:48:37.016744Z","iopub.execute_input":"2023-03-21T05:48:37.017034Z","iopub.status.idle":"2023-03-21T05:48:37.023069Z","shell.execute_reply.started":"2023-03-21T05:48:37.016987Z","shell.execute_reply":"2023-03-21T05:48:37.022326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y1 = pred_val > 0.4\ny1 = y1.astype(int).sum(axis=1) - 1","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:48:37.024245Z","iopub.execute_input":"2023-03-21T05:48:37.024711Z","iopub.status.idle":"2023-03-21T05:48:37.033393Z","shell.execute_reply.started":"2023-03-21T05:48:37.024662Z","shell.execute_reply":"2023-03-21T05:48:37.032683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y1.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:48:37.034695Z","iopub.execute_input":"2023-03-21T05:48:37.035143Z","iopub.status.idle":"2023-03-21T05:48:37.043963Z","shell.execute_reply.started":"2023-03-21T05:48:37.035093Z","shell.execute_reply":"2023-03-21T05:48:37.043358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y2 = y_val.sum(axis=1) - 1","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:48:37.046518Z","iopub.execute_input":"2023-03-21T05:48:37.047104Z","iopub.status.idle":"2023-03-21T05:48:37.051420Z","shell.execute_reply.started":"2023-03-21T05:48:37.047050Z","shell.execute_reply":"2023-03-21T05:48:37.050672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y1","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:48:37.053732Z","iopub.execute_input":"2023-03-21T05:48:37.054345Z","iopub.status.idle":"2023-03-21T05:48:37.064277Z","shell.execute_reply.started":"2023-03-21T05:48:37.054298Z","shell.execute_reply":"2023-03-21T05:48:37.063280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y2","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:48:37.065860Z","iopub.execute_input":"2023-03-21T05:48:37.066361Z","iopub.status.idle":"2023-03-21T05:48:37.075481Z","shell.execute_reply.started":"2023-03-21T05:48:37.066178Z","shell.execute_reply":"2023-03-21T05:48:37.074403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nprint(classification_report(y1, y2))","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:48:37.077106Z","iopub.execute_input":"2023-03-21T05:48:37.077634Z","iopub.status.idle":"2023-03-21T05:48:37.089335Z","shell.execute_reply.started":"2023-03-21T05:48:37.077433Z","shell.execute_reply":"2023-03-21T05:48:37.088459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\n\ncf_matrix = confusion_matrix(y1, y2)\nsns.heatmap(cf_matrix, annot=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:48:37.092064Z","iopub.execute_input":"2023-03-21T05:48:37.092288Z","iopub.status.idle":"2023-03-21T05:48:37.689118Z","shell.execute_reply.started":"2023-03-21T05:48:37.092243Z","shell.execute_reply":"2023-03-21T05:48:37.688040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}