{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# EfficientNetB3Trained\n\n\n---","metadata":{}},{"cell_type":"code","source":"# To have reproducible results and compare them\nnr_seed = 1812\nimport numpy as np \nnp.random.seed(nr_seed)\nimport tensorflow as tf\ntf.set_random_seed(nr_seed)","metadata":{"execution":{"iopub.status.busy":"2023-04-27T12:13:03.200347Z","iopub.execute_input":"2023-04-27T12:13:03.200700Z","iopub.status.idle":"2023-04-27T12:13:03.206522Z","shell.execute_reply.started":"2023-04-27T12:13:03.200635Z","shell.execute_reply":"2023-04-27T12:13:03.205078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import libraries\n!pip install -U '../input/install/efficientnet-0.0.3-py2.py3-none-any.whl'\nimport json\nimport math\nfrom tqdm import tqdm, tqdm_notebook\nimport gc\nimport warnings\nimport os\n\nimport cv2\nfrom PIL import Image\n\nimport pandas as pd\nimport scipy\nimport matplotlib.pyplot as plt\n\nfrom keras import backend as K\nfrom keras import layers\nfrom efficientnet import EfficientNetB3\nfrom keras.callbacks import Callback, ModelCheckpoint, ReduceLROnPlateau\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential\nfrom keras.optimizers import Adam\nfrom keras.losses import binary_crossentropy, categorical_crossentropy\nfrom skimage.color import rgb2hsv, lab2lch\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score\n\nwarnings.filterwarnings(\"ignore\")\n\n%matplotlib inline","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-27T12:13:04.958061Z","iopub.execute_input":"2023-04-27T12:13:04.958355Z","iopub.status.idle":"2023-04-27T12:13:32.813811Z","shell.execute_reply.started":"2023-04-27T12:13:04.958306Z","shell.execute_reply":"2023-04-27T12:13:32.812880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image size\nWIDTH= 224\nHEIGHT = 224\n# Batch size\nBATCH_SIZE = 32","metadata":{"execution":{"iopub.status.busy":"2023-04-27T12:13:32.815699Z","iopub.execute_input":"2023-04-27T12:13:32.816005Z","iopub.status.idle":"2023-04-27T12:13:32.827970Z","shell.execute_reply.started":"2023-04-27T12:13:32.815941Z","shell.execute_reply":"2023-04-27T12:13:32.826889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading & Merging","metadata":{}},{"cell_type":"code","source":"# del df\nfrom sklearn.utils import resample\nSEED =1812\ndef append_ext(fn):\n    return fn+\".jpeg\"\n# Đọc file CSV vào DataFrame\ndf = pd.read_csv(\"/kaggle/input/diabetic-retinopathy-resized/trainLabels.csv\", dtype=str)\ndf['image'] = df['image'].apply(append_ext)\n\n# Tách dữ liệu thành 5 tập dữ liệu tương ứng với giá trị của cột \"level\"\ndata_0 = df[df.level == '0']\ndata_1 = df[df.level == '1']\ndata_2 = df[df.level == '2']\ndata_3 = df[df.level == '3']\ndata_4 = df[df.level == '4']\n\n# Lấy số lượng mẫu của lớp thiểu số (giá trị \"1\", \"2\", \"3\", \"4\")\nn_minority = len(data_1) + len(data_2) + len(data_3) + len(data_4)\n\n# Undersampling lớp đa số (giá trị \"0\") bằng cách giảm số lượng mẫu của lớp đa số xuống còn bằng với số lượng mẫu của lớp thiểu số\ndata_0_downsampled = resample(data_0,\n                              replace=False,\n                              n_samples=n_minority,\n                              random_state=SEED)\n# Kết hợp lại các tập dữ liệu (một tập đã undersample và các tập lớp thiểu số)\ndf = pd.concat([data_0_downsampled, data_1, data_2, data_3, data_4])\n\n# In số lượng các giá trị trong cột \"level\" của tập dữ liệu đã cân bằng\nprint(df[\"level\"].value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:00:51.859018Z","iopub.execute_input":"2023-04-26T10:00:51.859653Z","iopub.status.idle":"2023-04-26T10:00:51.960179Z","shell.execute_reply.started":"2023-04-26T10:00:51.859592Z","shell.execute_reply":"2023-04-26T10:00:51.959247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['image'] = '/kaggle/input/diabetic-retinopathy-resized/resized_train/resized_train/' + df['image'].astype(str)\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:00:51.963778Z","iopub.execute_input":"2023-04-26T10:00:51.964338Z","iopub.status.idle":"2023-04-26T10:00:52.010428Z","shell.execute_reply.started":"2023-04-26T10:00:51.964277Z","shell.execute_reply":"2023-04-26T10:00:52.009447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df['image'].iloc[1])","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:00:52.013794Z","iopub.execute_input":"2023-04-26T10:00:52.014467Z","iopub.status.idle":"2023-04-26T10:00:52.022086Z","shell.execute_reply.started":"2023-04-26T10:00:52.014405Z","shell.execute_reply":"2023-04-26T10:00:52.020945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train - Valid split\nUse new Data for validation and Old data for training£","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX = df['image']\ny = df['level']\nX_train, X_valid, y_train, y_valid = train_test_split(X, y, test_size=0.2, random_state=SEED, shuffle=True)\n# X_train, X_test, y_train, y_test = train_test_split(X_train, y_train, test_size=0.1, random_state=SEED, shuffle=True)\ntrain_df = pd.DataFrame({'image': X_train, 'level': y_train}).reset_index(drop=True)\nval_df = pd.DataFrame({'image': X_valid, 'level': y_valid}).reset_index(drop=True)\n# test_df = pd.DataFrame({'image': X_test, 'level': y_test}).reset_index(drop=True)\n\nprint(train_df.shape)\nprint(val_df.shape)\n# print(test_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:00:52.023968Z","iopub.execute_input":"2023-04-26T10:00:52.024659Z","iopub.status.idle":"2023-04-26T10:00:52.044748Z","shell.execute_reply.started":"2023-04-26T10:00:52.024595Z","shell.execute_reply":"2023-04-26T10:00:52.042697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Process Images","metadata":{}},{"cell_type":"markdown","source":"Crop function: https://www.kaggle.com/ratthachat/aptos-updated-preprocessing-ben-s-cropping ","metadata":{}},{"cell_type":"code","source":"def crop_image1(img,tol=7):\n    # img is image data\n    # tol  is tolerance\n        \n    mask = img>tol\n    return img[np.ix_(mask.any(1),mask.any(0))]\n\ndef crop_image_from_gray(img,tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n    #         print(img1.shape,img2.shape,img3.shape)\n            img = np.stack([img1,img2,img3],axis=-1)\n    #         print(img.shape)\n        return img\n\n# Make all images circular (possible data loss)\ndef circle_crop(img, sigmaX=20):   \n    \"\"\"\n    Create circular crop around image centre    \n    \"\"\"    \n    img = crop_image_from_gray(img)    \n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    \n    height, width, depth = img.shape    \n    \n    x = int(width/2)\n    y = int(height/2)\n    r = np.amin((x,y))\n    \n    circle_img = np.zeros((height, width), np.uint8)\n    cv2.circle(circle_img, (x,y), int(r), 1, thickness=-1)\n    img = cv2.bitwise_and(img, img, mask=circle_img)\n    img = crop_image_from_gray(img)\n    img=cv2.addWeighted ( img,4, cv2.GaussianBlur( img , (0,0) , sigmaX) ,-4 ,128)\n    return img \ndef preprocess_image(image_path, width, height, new_data=False):\n    img = cv2.imread(image_path)\n#     img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)    \n#     img = crop_image_from_gray(img)\n    img = circle_crop(img)\n    img = cv2.resize(img, (width,height))\n#     img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), 20) ,-4 ,128)\n\n    return img","metadata":{"execution":{"iopub.status.busy":"2023-04-27T12:15:37.781528Z","iopub.execute_input":"2023-04-27T12:15:37.781848Z","iopub.status.idle":"2023-04-27T12:15:37.798670Z","shell.execute_reply.started":"2023-04-27T12:15:37.781780Z","shell.execute_reply":"2023-04-27T12:15:37.797862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_samples(df, columns=4, rows=3):\n    fig=plt.figure(figsize=(5*columns, 4*rows))\n\n    for i in range(columns*rows):\n        image_path = df.loc[i,'image']\n        print(image_path)\n        image_id = df.loc[i,'level']\n        img = preprocess_image(f'{image_path}', width=WIDTH, height=HEIGHT)\n        fig.add_subplot(rows, columns, i+1)\n        plt.title(image_id)\n        plt.imshow(img)\n    \n    plt.tight_layout()\n\ndisplay_samples(train_df)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-26T10:00:52.069665Z","iopub.execute_input":"2023-04-26T10:00:52.069973Z","iopub.status.idle":"2023-04-26T10:00:57.817144Z","shell.execute_reply.started":"2023-04-26T10:00:52.069911Z","shell.execute_reply":"2023-04-26T10:00:57.815995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Processing Images","metadata":{}},{"cell_type":"markdown","source":"__UPDATE:__ Here we are reading just the validation set. In order to use 320x320 images, we are going to load one bucket at a time only when needed. This will let our code run without memory-related errors.","metadata":{}},{"cell_type":"code","source":"# validation set\nN = val_df.shape[0]\nx_val = np.empty((N, HEIGHT, WIDTH, 3), dtype=np.uint8)\n\nfor i, image_id in enumerate(tqdm_notebook(val_df['image'])):\n    x_val[i, :, :, :] = preprocess_image(\n        f'{image_id}',\n        height=HEIGHT, width=WIDTH, new_data=True\n    )","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:00:57.818766Z","iopub.execute_input":"2023-04-26T10:00:57.819117Z","iopub.status.idle":"2023-04-26T10:09:51.248271Z","shell.execute_reply.started":"2023-04-26T10:00:57.819053Z","shell.execute_reply":"2023-04-26T10:09:51.247432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = pd.get_dummies(train_df['level']).values\ny_val = pd.get_dummies(val_df['level']).values\n\nprint(y_train.shape)\nprint(x_val.shape)\nprint(y_val.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:09:51.249787Z","iopub.execute_input":"2023-04-26T10:09:51.250368Z","iopub.status.idle":"2023-04-26T10:09:51.262659Z","shell.execute_reply.started":"2023-04-26T10:09:51.250314Z","shell.execute_reply":"2023-04-26T10:09:51.261499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_valid = pd.get_dummies(val_df['level']).values","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:09:51.264643Z","iopub.execute_input":"2023-04-26T10:09:51.265074Z","iopub.status.idle":"2023-04-26T10:09:51.271445Z","shell.execute_reply.started":"2023-04-26T10:09:51.264893Z","shell.execute_reply":"2023-04-26T10:09:51.270447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating multilabels\n\nThay vì dự đoán một nhãn đơn lẻ, chúng ta sẽ thay đổi mục tiêu để trở thành một bài toán đa nhãn (multilabel); tức là, nếu nhãn là một lớp nhất định, thì nó bao gồm tất cả các lớp trước đó. Ví dụ, mã hóa một retinopathy của lớp 4 thường là [0, 0, 0, 1], nhưng trong trường hợp của chúng ta, chúng ta sẽ dự đoán [1, 1, 1, 1]. Để biết thêm chi tiết, vui lòng xem Kernel của Lex.","metadata":{}},{"cell_type":"code","source":"y_train_multi = np.empty(y_train.shape, dtype=y_train.dtype)\ny_train_multi[:, 4] = y_train[:, 4]\n\nfor i in range(3, -1, -1):\n    y_train_multi[:, i] = np.logical_or(y_train[:, i], y_train_multi[:, i+1])\n\ny_val_multi = np.empty(y_val.shape, dtype=y_val.dtype)\ny_val_multi[:, 4] = y_val[:, 4]\n\nfor i in range(3, -1, -1):\n    y_val_multi[:, i] = np.logical_or(y_val[:, i], y_val_multi[:, i+1])\n\nprint(\"Y_train multi: {}\".format(y_train_multi.shape))\nprint(\"Y_val multi: {}\".format(y_val_multi.shape))","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:09:51.272938Z","iopub.execute_input":"2023-04-26T10:09:51.273475Z","iopub.status.idle":"2023-04-26T10:09:51.285145Z","shell.execute_reply.started":"2023-04-26T10:09:51.27335Z","shell.execute_reply":"2023-04-26T10:09:51.284534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = y_train_multi\ny_val = y_val_multi","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:09:51.286877Z","iopub.execute_input":"2023-04-26T10:09:51.287312Z","iopub.status.idle":"2023-04-26T10:09:51.296645Z","shell.execute_reply.started":"2023-04-26T10:09:51.287263Z","shell.execute_reply":"2023-04-26T10:09:51.295692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # delete the uneeded df\n# del train\n# del df\n# del val_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:09:51.29976Z","iopub.execute_input":"2023-04-26T10:09:51.300111Z","iopub.status.idle":"2023-04-26T10:09:51.441785Z","shell.execute_reply.started":"2023-04-26T10:09:51.300055Z","shell.execute_reply":"2023-04-26T10:09:51.44095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating keras callback for QWK\n\n---\n\nI had to change this function, in order to consider the best kappa score among all the buckets.","metadata":{}},{"cell_type":"code","source":"class Metrics(Callback):\n\n    def on_epoch_end(self, epoch, logs={}):\n        X_val, y_val = self.validation_data[:2]\n        y_val = y_val.sum(axis=1) - 1\n        \n        y_pred = self.model.predict(X_val) > 0.5\n        y_pred = y_pred.astype(int).sum(axis=1) - 1\n\n        _val_kappa = cohen_kappa_score(\n            y_val,\n            y_pred, \n            weights='quadratic'\n        )\n\n        self.val_kappas.append(_val_kappa)\n\n        print(f\"val_kappa: {_val_kappa:.4f}\")\n        \n        if _val_kappa == max(self.val_kappas):\n            print(\"Validation Kappa has improved. Saving model.\")\n            self.model.save('model.h5')\n\n        return","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:09:51.443398Z","iopub.execute_input":"2023-04-26T10:09:51.443957Z","iopub.status.idle":"2023-04-26T10:09:51.45556Z","shell.execute_reply.started":"2023-04-26T10:09:51.443894Z","shell.execute_reply":"2023-04-26T10:09:51.45466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Generator","metadata":{}},{"cell_type":"code","source":"def create_datagen():\n    return ImageDataGenerator(\n        horizontal_flip=True,\n        vertical_flip=True,\n        zoom_range= 0.3,\n        brightness_range=(0.5, 2),\n        fill_mode='constant',\n        cval=0\n    )","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:09:51.459356Z","iopub.execute_input":"2023-04-26T10:09:51.459626Z","iopub.status.idle":"2023-04-26T10:09:51.469495Z","shell.execute_reply.started":"2023-04-26T10:09:51.459556Z","shell.execute_reply":"2023-04-26T10:09:51.468703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check the differenct kinds of augmentations on the pictures.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 10, figsize=(20, 10))\nax = ax.ravel()\n\nimg = x_val[0].reshape(1,x_val[0].shape[0],x_val[0].shape[1], x_val[0].shape[2])\n\nax[0].imshow(img[0].astype('uint8'))\nax[1].imshow(next(ImageDataGenerator().flow(img))[0].astype('uint8'))\nax[2].imshow(next(ImageDataGenerator(horizontal_flip=True, fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[3].imshow(next(ImageDataGenerator(vertical_flip=True,fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[4].imshow(next(ImageDataGenerator(rotation_range=360, fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[5].imshow(next(ImageDataGenerator(zoom_range= (0.65,1), fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[6].imshow(next(ImageDataGenerator(height_shift_range=0.15, fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[7].imshow(next(ImageDataGenerator(width_shift_range=0.15, fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[8].imshow(next(ImageDataGenerator(brightness_range=(0.5, 2), fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[9].imshow(next(ImageDataGenerator(horizontal_flip=True,\n                                     vertical_flip=True,\n                                     rotation_range=360,zoom_range= (0.65,1),\n                                     brightness_range=(0.5, 2),\n                                     fill_mode='constant',cval=0).flow(img))[0].astype('uint8'))\n","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:09:51.47301Z","iopub.execute_input":"2023-04-26T10:09:51.473279Z","iopub.status.idle":"2023-04-26T10:09:53.030919Z","shell.execute_reply.started":"2023-04-26T10:09:51.473228Z","shell.execute_reply":"2023-04-26T10:09:53.029737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model: EfficientNetB3","metadata":{}},{"cell_type":"code","source":"efficientnetb3 = EfficientNetB3(\n        weights=None,\n        input_shape=(HEIGHT,WIDTH,3),\n        include_top=False\n                   )\n\nefficientnetb3.load_weights(\"../input/efficientnet-keras-weights-b0b5/efficientnet-b3_imagenet_1000_notop.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:29:02.606868Z","iopub.execute_input":"2023-04-26T10:29:02.607211Z","iopub.status.idle":"2023-04-26T10:29:19.894536Z","shell.execute_reply.started":"2023-04-26T10:29:02.607153Z","shell.execute_reply":"2023-04-26T10:29:19.89363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install git+https://github.com/qubvel/segmentation_models\n\nimport segmentation_models as sm\n\nmodel = tf.keras.Sequential([\n        sm.EfficientNetB7(\n            input_shape=(IM_Z, IM_Z, 3),\n            weights='imagenet',\n            include_top=False\n        ),\n        tf.keras.layers.GlobalAveragePooling2D(),\n        tf.keras.layers.Dense(1, activation='sigmoid')\n    ])","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:32:46.825761Z","iopub.execute_input":"2023-04-26T10:32:46.826172Z","iopub.status.idle":"2023-04-26T10:33:29.278988Z","shell.execute_reply.started":"2023-04-26T10:32:46.826106Z","shell.execute_reply":"2023-04-26T10:33:29.277395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model():\n    model = Sequential()\n    model.add(efficientnetb3)\n    model.add(layers.GlobalAveragePooling2D())\n    model.add(layers.Dropout(0.5))\n    model.add(layers.BatchNormalization())\n    model.add(layers.Dense(5, activation='sigmoid'))\n    \n    model.compile(\n        loss='binary_crossentropy',\n        #loss=kappa_loss,\n        optimizer=Adam(lr=1e-4,decay=1e-6),\n        metrics=['accuracy']\n    )\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:36:10.370548Z","iopub.execute_input":"2023-04-26T10:36:10.370933Z","iopub.status.idle":"2023-04-26T10:36:10.377924Z","shell.execute_reply.started":"2023-04-26T10:36:10.370863Z","shell.execute_reply":"2023-04-26T10:36:10.376952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = build_model()\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:36:24.587338Z","iopub.execute_input":"2023-04-26T10:36:24.58767Z","iopub.status.idle":"2023-04-26T10:36:32.363878Z","shell.execute_reply.started":"2023-04-26T10:36:24.587616Z","shell.execute_reply":"2023-04-26T10:36:32.362317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(model, to_file='model.png', show_shapes=True, show_layer_names=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:36:36.625973Z","iopub.execute_input":"2023-04-26T10:36:36.626308Z","iopub.status.idle":"2023-04-26T10:36:39.524434Z","shell.execute_reply.started":"2023-04-26T10:36:36.626253Z","shell.execute_reply":"2023-04-26T10:36:39.523417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pretraining with old Data","metadata":{}},{"cell_type":"code","source":"bucket_num = 8\ndiv = round(train_df.shape[0]/bucket_num)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:36:50.116952Z","iopub.execute_input":"2023-04-26T10:36:50.117312Z","iopub.status.idle":"2023-04-26T10:36:50.122712Z","shell.execute_reply.started":"2023-04-26T10:36:50.117252Z","shell.execute_reply":"2023-04-26T10:36:50.121534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_init = {\n    'val_loss': [0.0],\n    'val_acc': [0.0],\n    'loss': [0.0], \n    'acc': [0.0],\n    'bucket': [0.0]\n}\nresults = pd.DataFrame(df_init)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:36:58.744739Z","iopub.execute_input":"2023-04-26T10:36:58.745093Z","iopub.status.idle":"2023-04-26T10:36:58.751014Z","shell.execute_reply.started":"2023-04-26T10:36:58.745034Z","shell.execute_reply":"2023-04-26T10:36:58.750175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I found that changing the nr. of epochs for each bucket helped in terms of performances\nepochs = [5,5,5,5,5,5,5,5]\nkappa_metrics = Metrics()\nkappa_metrics.val_kappas = []\n\nlearn_control = ReduceLROnPlateau(monitor='val_acc', patience=5,\n                                  verbose=1,factor=.2, min_lr=1e-7)\n\ncheckpoint = ModelCheckpoint('val_model.h5', monitor='val_loss', verbose=1, save_best_only=True, mode='min')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:37:01.392924Z","iopub.execute_input":"2023-04-26T10:37:01.393267Z","iopub.status.idle":"2023-04-26T10:37:01.399524Z","shell.execute_reply.started":"2023-04-26T10:37:01.393209Z","shell.execute_reply":"2023-04-26T10:37:01.398571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nconfig = tf.compat.v1.ConfigProto()\nconfig.gpu_options.allow_growth = True\nsession = tf.compat.v1.Session(config=config)\ntf.compat.v1.keras.backend.set_session(session)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:37:03.967982Z","iopub.execute_input":"2023-04-26T10:37:03.968303Z","iopub.status.idle":"2023-04-26T10:37:03.983042Z","shell.execute_reply.started":"2023-04-26T10:37:03.968247Z","shell.execute_reply":"2023-04-26T10:37:03.982154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(0,bucket_num):\n    if i != (bucket_num-1):\n        print(\"Bucket Nr: {}\".format(i))\n        \n        N = train_df.iloc[i*div:(1+i)*div].shape[0]\n        x_train = np.empty((N, HEIGHT, WIDTH, 3), dtype=np.uint8)\n        for j, image_id in enumerate(tqdm_notebook(train_df.iloc[i*div:(1+i)*div,0])):\n            x_train[j, :, :, :] = preprocess_image(f'{image_id}', height=HEIGHT, width=WIDTH)\n\n        data_generator = create_datagen().flow(x_train, y_train[i*div:(1+i)*div,:], batch_size=BATCH_SIZE)\n        history = model.fit_generator(\n                        data_generator,\n                        steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                        epochs=epochs[i],\n                        validation_data=(x_val, y_val),\n                        callbacks=[kappa_metrics, learn_control, checkpoint]\n                        )\n        \n        dic = history.history\n        df_model = pd.DataFrame(dic)\n        df_model['bucket'] = i\n    else:\n        print(\"Bucket Nr: {}\".format(i))\n        \n        N = train_df.iloc[i*div:].shape[0]\n        x_train = np.empty((N, HEIGHT, WIDTH, 3), dtype=np.uint8)\n        for j, image_id in enumerate(tqdm_notebook(train_df.iloc[i*div:,0])):\n            x_train[j, :, :, :] = preprocess_image(f'{image_id}', height=HEIGHT, width=WIDTH)\n        data_generator = create_datagen().flow(x_train, y_train[i*div:,:], batch_size=BATCH_SIZE)\n        \n        history = model.fit_generator(\n                        data_generator,\n                        steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                        epochs=epochs[i],\n                        validation_data=(x_val, y_val),\n                        callbacks=[kappa_metrics, learn_control, checkpoint]\n                        )\n        \n        dic = history.history\n        df_model = pd.DataFrame(dic)\n        df_model['bucket'] = i\n\n    results = results.append(df_model)\n    \n    del data_generator\n    del x_train\n    gc.collect()\n    \n    print('-'*40)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:41:37.383694Z","iopub.execute_input":"2023-04-26T10:41:37.384085Z","iopub.status.idle":"2023-04-26T12:14:56.474513Z","shell.execute_reply.started":"2023-04-26T10:41:37.384026Z","shell.execute_reply":"2023-04-26T12:14:56.473257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = results.iloc[1:]\nresults['kappa'] = kappa_metrics.val_kappas\nresults = results.reset_index()\nresults = results.rename(index=str, columns={\"index\": \"epoch\"})\nresults","metadata":{"execution":{"iopub.status.busy":"2023-04-26T12:21:43.125076Z","iopub.execute_input":"2023-04-26T12:21:43.125423Z","iopub.status.idle":"2023-04-26T12:21:43.166623Z","shell.execute_reply.started":"2023-04-26T12:21:43.125367Z","shell.execute_reply":"2023-04-26T12:21:43.165775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results[['loss', 'val_loss']].plot()\nresults[['acc', 'val_acc']].plot()\nresults[['kappa']].plot()\nresults.to_csv('model_results.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T12:21:51.453644Z","iopub.execute_input":"2023-04-26T12:21:51.453994Z","iopub.status.idle":"2023-04-26T12:21:52.672496Z","shell.execute_reply.started":"2023-04-26T12:21:51.453935Z","shell.execute_reply":"2023-04-26T12:21:52.671145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Tinh chỉnh với dữ liệu mới\nCreate New Train and Validation Set to finetune our model","metadata":{}},{"cell_type":"code","source":"# model.load_weights('val_model.h5')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:10:14.001182Z","iopub.status.idle":"2023-04-26T10:10:14.001879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_val, y_train, y_val = train_test_split(\n    x_val, y_val, \n    test_size=0.2, \n    random_state=nr_seed\n)\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T12:22:23.514011Z","iopub.execute_input":"2023-04-26T12:22:23.514369Z","iopub.status.idle":"2023-04-26T12:22:24.424922Z","shell.execute_reply.started":"2023-04-26T12:22:23.514314Z","shell.execute_reply":"2023-04-26T12:22:24.423867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_generator = create_datagen().flow(x_train, y_train, batch_size=BATCH_SIZE)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T12:22:27.010932Z","iopub.execute_input":"2023-04-26T12:22:27.011257Z","iopub.status.idle":"2023-04-26T12:22:28.202928Z","shell.execute_reply.started":"2023-04-26T12:22:27.011199Z","shell.execute_reply":"2023-04-26T12:22:28.20211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit_generator(\n                data_generator,\n                steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                epochs=20,\n                validation_data=(x_val, y_val),\n                callbacks=[kappa_metrics,learn_control,checkpoint]\n                )","metadata":{"execution":{"iopub.status.busy":"2023-04-26T12:22:31.083179Z","iopub.execute_input":"2023-04-26T12:22:31.083507Z","iopub.status.idle":"2023-04-26T12:47:34.178892Z","shell.execute_reply.started":"2023-04-26T12:22:31.083453Z","shell.execute_reply":"2023-04-26T12:47:34.177969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_generator = create_datagen().flow(x_train, y_train, batch_size=BATCH_SIZE)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T12:47:34.184096Z","iopub.execute_input":"2023-04-26T12:47:34.186472Z","iopub.status.idle":"2023-04-26T12:47:35.567677Z","shell.execute_reply.started":"2023-04-26T12:47:34.186413Z","shell.execute_reply":"2023-04-26T12:47:35.566696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit_generator(\n                data_generator,\n                steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                epochs=20,\n                validation_data=(x_val, y_val),\n                callbacks=[kappa_metrics,learn_control,checkpoint]\n                )","metadata":{"execution":{"iopub.status.busy":"2023-04-26T12:47:35.569149Z","iopub.execute_input":"2023-04-26T12:47:35.569473Z","iopub.status.idle":"2023-04-26T13:12:26.700373Z","shell.execute_reply.started":"2023-04-26T12:47:35.569409Z","shell.execute_reply":"2023-04-26T13:12:26.699407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_weights('val_model.h5')\npred_val = model.predict(x_val)\nprint(pred_val)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T13:12:26.704543Z","iopub.execute_input":"2023-04-26T13:12:26.706677Z","iopub.status.idle":"2023-04-26T13:12:33.923589Z","shell.execute_reply.started":"2023-04-26T13:12:26.706618Z","shell.execute_reply":"2023-04-26T13:12:33.922572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pred_val_binary = (pred_val > 0.5).astype(int) # threshold at 0.5 to obtain binary predictions\n# accuracy = accuracy_score(y_valid, pred_val)\n# print(\"Accuracy: \", accuracy)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T13:20:57.027308Z","iopub.execute_input":"2023-04-26T13:20:57.027663Z","iopub.status.idle":"2023-04-26T13:20:57.0317Z","shell.execute_reply.started":"2023-04-26T13:20:57.027601Z","shell.execute_reply":"2023-04-26T13:20:57.030698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Tối ưu ngưỡng","metadata":{}},{"cell_type":"code","source":"def compute_score_inv(threshold):\n    y1 = pred_val > threshold\n    y1 = y1.astype(int).sum(axis=1) - 1\n    y2 = y_val.sum(axis=1) - 1\n    score = cohen_kappa_score(y1, y2, weights='quadratic')\n    return 1 - score\nsimplex = scipy.optimize.minimize(compute_score_inv, 0.5, method='nelder-mead')\n\nbest_threshold = simplex['x'][0]\nprint(best_threshold)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T13:18:28.454455Z","iopub.execute_input":"2023-04-26T13:18:28.454779Z","iopub.status.idle":"2023-04-26T13:18:29.184311Z","shell.execute_reply.started":"2023-04-26T13:18:28.454721Z","shell.execute_reply":"2023-04-26T13:18:29.18343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:10:14.018663Z","iopub.status.idle":"2023-04-26T10:10:14.019347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class OptimizedRounder(object):\n    def __init__(self):\n        self.coef_ = 0\n\n    def _kappa_loss(self, coef, X, y):\n        X_p = np.copy(X)\n        for i, pred in enumerate(X_p):\n            if pred < coef[0]:\n                X_p[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                X_p[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                X_p[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                X_p[i] = 3\n            else:\n                X_p[i] = 4\n\n        ll = cohen_kappa_score(y, X_p, weights='quadratic')\n        return -ll\n\n    def fit(self, X, y):\n        loss_partial = partial(self._kappa_loss, X=X, y=y)\n        initial_coef = [0.5, 1.5, 2.5, 3.5]\n        self.coef_ = sp.optimize.minimize(loss_partial, initial_coef, method='nelder-mead')\n        print(-loss_partial(self.coef_['x']))\n\n    def predict(self, X, coef):\n        X_p = np.copy(X)\n        for i, pred in enumerate(X_p):\n            if pred < coef[0]:\n                X_p[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                X_p[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                X_p[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                X_p[i] = 3\n            else:\n                X_p[i] = 4\n        return X_p\n\n    def coefficients(self):\n        return self.coef_['x']","metadata":{"execution":{"iopub.status.busy":"2023-04-26T15:58:34.773931Z","iopub.execute_input":"2023-04-26T15:58:34.774252Z","iopub.status.idle":"2023-04-26T15:58:34.787045Z","shell.execute_reply.started":"2023-04-26T15:58:34.77418Z","shell.execute_reply":"2023-04-26T15:58:34.786152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nmodel.load_weights('/kaggle/input/modelh5')\ny_val_pred = model.predict(x_val)\n\noptR = OptimizedRounder()\noptR.fit(y_val_pred, y_val)\ncoefficients = optR.coefficients()\nprint(f'Coefficients: {coefficients}')\ny_val_pred = optR.predict(y_val_pred, coefficients)\n\nscore = cohen_kappa_score(y_val_pred, y_val, weights='quadratic')\n\nprint('Optimized Validation QWK score: {}'.format(score))\nprint('Not Optimized Validation QWK score: {}'.format(max(results.kappa)))","metadata":{"execution":{"iopub.status.busy":"2023-04-26T16:01:37.326569Z","iopub.execute_input":"2023-04-26T16:01:37.326911Z","iopub.status.idle":"2023-04-26T16:01:37.34016Z","shell.execute_reply.started":"2023-04-26T16:01:37.326855Z","shell.execute_reply":"2023-04-26T16:01:37.338915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    import keras\n    HEIGHT = WIDTH = 224","metadata":{"execution":{"iopub.status.busy":"2023-04-27T12:16:41.312174Z","iopub.execute_input":"2023-04-27T12:16:41.312530Z","iopub.status.idle":"2023-04-27T12:16:41.318660Z","shell.execute_reply.started":"2023-04-27T12:16:41.312469Z","shell.execute_reply":"2023-04-27T12:16:41.315976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = keras.models.load_model('/kaggle/input/val-model-new/val_model.h5')","metadata":{"execution":{"iopub.status.busy":"2023-04-27T12:14:06.664231Z","iopub.execute_input":"2023-04-27T12:14:06.664587Z","iopub.status.idle":"2023-04-27T12:15:05.380204Z","shell.execute_reply.started":"2023-04-27T12:14:06.664515Z","shell.execute_reply":"2023-04-27T12:15:05.379400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = '/kaggle/input/diabetic-retinopathy-resized/resized_train/resized_train/16_left.jpeg'\nx_test = np.empty((1, HEIGHT, WIDTH, 3), dtype=np.uint8)\n\n# Preprocess image and assign to x_test\nx_test[0, :, :, :] = preprocess_image(\n        f'{path}',\n        height=HEIGHT, width=WIDTH\n    )","metadata":{"execution":{"iopub.status.busy":"2023-04-27T13:04:45.119070Z","iopub.execute_input":"2023-04-27T13:04:45.119398Z","iopub.status.idle":"2023-04-27T13:04:45.268793Z","shell.execute_reply.started":"2023-04-27T13:04:45.119339Z","shell.execute_reply":"2023-04-27T13:04:45.268031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x =model.predict(x_test)\nx = x.flatten()\nx","metadata":{"execution":{"iopub.status.busy":"2023-04-27T13:04:47.409239Z","iopub.execute_input":"2023-04-27T13:04:47.409554Z","iopub.status.idle":"2023-04-27T13:04:47.443698Z","shell.execute_reply.started":"2023-04-27T13:04:47.409498Z","shell.execute_reply":"2023-04-27T13:04:47.442796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coef = np.array([0.60513711,1.36421993,2.52844061,3.07940899])\ncoef","metadata":{"execution":{"iopub.status.busy":"2023-04-27T12:43:49.706673Z","iopub.execute_input":"2023-04-27T12:43:49.707026Z","iopub.status.idle":"2023-04-27T12:43:49.716547Z","shell.execute_reply.started":"2023-04-27T12:43:49.706945Z","shell.execute_reply":"2023-04-27T12:43:49.715699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(X, coef):\n    X_p = np.copy(X)\n    for i, pred in enumerate(X_p):\n        if pred < coef[0]:\n            X_p[i] = 0\n        elif pred >= coef[0] and pred < coef[1]:\n            X_p[i] = 1\n        elif pred >= coef[1] and pred < coef[2]:\n            X_p[i] = 2\n        elif pred >= coef[2] and pred < coef[3]:\n            X_p[i] = 3\n        else:\n            X_p[i] = 4\n    return X_p","metadata":{"execution":{"iopub.status.busy":"2023-04-27T12:42:50.511992Z","iopub.execute_input":"2023-04-27T12:42:50.512310Z","iopub.status.idle":"2023-04-27T12:42:50.518810Z","shell.execute_reply.started":"2023-04-27T12:42:50.512259Z","shell.execute_reply":"2023-04-27T12:42:50.517916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    x_p = predict(x ,coef)\n    x_p","metadata":{"execution":{"iopub.status.busy":"2023-04-27T13:04:52.476595Z","iopub.execute_input":"2023-04-27T13:04:52.476920Z","iopub.status.idle":"2023-04-27T13:04:52.483449Z","shell.execute_reply.started":"2023-04-27T13:04:52.476856Z","shell.execute_reply":"2023-04-27T13:04:52.482497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[](http://)","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}