{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# EfficientNetB3Trained\n\n\n---","metadata":{}},{"cell_type":"code","source":"# To have reproducible results and compare them\nnr_seed = 1812\nimport numpy as np \nnp.random.seed(nr_seed)\nimport tensorflow as tf\ntf.set_random_seed(nr_seed)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:00:19.329910Z","iopub.execute_input":"2023-04-26T10:00:19.330211Z","iopub.status.idle":"2023-04-26T10:00:21.330144Z","shell.execute_reply.started":"2023-04-26T10:00:19.330160Z","shell.execute_reply":"2023-04-26T10:00:21.327886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import libraries\n!pip install -U '../input/install/efficientnet-0.0.3-py2.py3-none-any.whl'\nimport json\nimport math\nfrom tqdm import tqdm, tqdm_notebook\nimport gc\nimport warnings\nimport os\n\nimport cv2\nfrom PIL import Image\n\nimport pandas as pd\nimport scipy\nimport matplotlib.pyplot as plt\n\nfrom keras import backend as K\nfrom keras import layers\nfrom efficientnet import EfficientNetB3\nfrom keras.callbacks import Callback, ModelCheckpoint, ReduceLROnPlateau\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential\nfrom keras.optimizers import Adam\nfrom keras.losses import binary_crossentropy, categorical_crossentropy\nfrom skimage.color import rgb2hsv, lab2lch\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score\n\nwarnings.filterwarnings(\"ignore\")\n\n%matplotlib inline","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-26T10:27:24.947751Z","iopub.execute_input":"2023-04-26T10:27:24.948147Z","iopub.status.idle":"2023-04-26T10:27:50.363612Z","shell.execute_reply.started":"2023-04-26T10:27:24.948079Z","shell.execute_reply":"2023-04-26T10:27:50.362778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image size\nWIDTH= 224\nHEIGHT = 224\n# Batch size\nBATCH_SIZE = 32","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:00:51.844528Z","iopub.execute_input":"2023-04-26T10:00:51.844904Z","iopub.status.idle":"2023-04-26T10:00:51.857180Z","shell.execute_reply.started":"2023-04-26T10:00:51.844844Z","shell.execute_reply":"2023-04-26T10:00:51.856230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading & Merging","metadata":{}},{"cell_type":"code","source":"# del df\nfrom sklearn.utils import resample\nSEED =1812\ndef append_ext(fn):\n    return fn+\".jpeg\"\n# Đọc file CSV vào DataFrame\ndf = pd.read_csv(\"/kaggle/input/diabetic-retinopathy-resized/trainLabels.csv\", dtype=str)\ndf['image'] = df['image'].apply(append_ext)\n\n# Tách dữ liệu thành 5 tập dữ liệu tương ứng với giá trị của cột \"level\"\ndata_0 = df[df.level == '0']\ndata_1 = df[df.level == '1']\ndata_2 = df[df.level == '2']\ndata_3 = df[df.level == '3']\ndata_4 = df[df.level == '4']\n\n# Lấy số lượng mẫu của lớp thiểu số (giá trị \"1\", \"2\", \"3\", \"4\")\nn_minority = len(data_1) + len(data_2) + len(data_3) + len(data_4)\n\n# Undersampling lớp đa số (giá trị \"0\") bằng cách giảm số lượng mẫu của lớp đa số xuống còn bằng với số lượng mẫu của lớp thiểu số\ndata_0_downsampled = resample(data_0,\n                              replace=False,\n                              n_samples=n_minority,\n                              random_state=SEED)\n# Kết hợp lại các tập dữ liệu (một tập đã undersample và các tập lớp thiểu số)\ndf = pd.concat([data_0_downsampled, data_1, data_2, data_3, data_4])\n\n# In số lượng các giá trị trong cột \"level\" của tập dữ liệu đã cân bằng\nprint(df[\"level\"].value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:00:51.859018Z","iopub.execute_input":"2023-04-26T10:00:51.859653Z","iopub.status.idle":"2023-04-26T10:00:51.960179Z","shell.execute_reply.started":"2023-04-26T10:00:51.859592Z","shell.execute_reply":"2023-04-26T10:00:51.959247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['image'] = '/kaggle/input/diabetic-retinopathy-resized/resized_train/resized_train/' + df['image'].astype(str)\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:00:51.963778Z","iopub.execute_input":"2023-04-26T10:00:51.964338Z","iopub.status.idle":"2023-04-26T10:00:52.010428Z","shell.execute_reply.started":"2023-04-26T10:00:51.964277Z","shell.execute_reply":"2023-04-26T10:00:52.009447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df['image'].iloc[1])","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:00:52.013794Z","iopub.execute_input":"2023-04-26T10:00:52.014467Z","iopub.status.idle":"2023-04-26T10:00:52.022086Z","shell.execute_reply.started":"2023-04-26T10:00:52.014405Z","shell.execute_reply":"2023-04-26T10:00:52.020945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train - Valid split\nUse new Data for validation and Old data for training£","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX = df['image']\ny = df['level']\nX_train, X_valid, y_train, y_valid = train_test_split(X, y, test_size=0.2, random_state=SEED, shuffle=True)\n# X_train, X_test, y_train, y_test = train_test_split(X_train, y_train, test_size=0.1, random_state=SEED, shuffle=True)\ntrain_df = pd.DataFrame({'image': X_train, 'level': y_train}).reset_index(drop=True)\nval_df = pd.DataFrame({'image': X_valid, 'level': y_valid}).reset_index(drop=True)\n# test_df = pd.DataFrame({'image': X_test, 'level': y_test}).reset_index(drop=True)\n\nprint(train_df.shape)\nprint(val_df.shape)\n# print(test_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:00:52.023968Z","iopub.execute_input":"2023-04-26T10:00:52.024659Z","iopub.status.idle":"2023-04-26T10:00:52.044748Z","shell.execute_reply.started":"2023-04-26T10:00:52.024595Z","shell.execute_reply":"2023-04-26T10:00:52.042697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Process Images","metadata":{}},{"cell_type":"markdown","source":"Crop function: https://www.kaggle.com/ratthachat/aptos-updated-preprocessing-ben-s-cropping ","metadata":{}},{"cell_type":"code","source":"def crop_image1(img,tol=7):\n    # img is image data\n    # tol  is tolerance\n        \n    mask = img>tol\n    return img[np.ix_(mask.any(1),mask.any(0))]\n\ndef crop_image_from_gray(img,tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n    #         print(img1.shape,img2.shape,img3.shape)\n            img = np.stack([img1,img2,img3],axis=-1)\n    #         print(img.shape)\n        return img\n\n# Make all images circular (possible data loss)\ndef circle_crop(img, sigmaX=20):   \n    \"\"\"\n    Create circular crop around image centre    \n    \"\"\"    \n    img = crop_image_from_gray(img)    \n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    \n    height, width, depth = img.shape    \n    \n    x = int(width/2)\n    y = int(height/2)\n    r = np.amin((x,y))\n    \n    circle_img = np.zeros((height, width), np.uint8)\n    cv2.circle(circle_img, (x,y), int(r), 1, thickness=-1)\n    img = cv2.bitwise_and(img, img, mask=circle_img)\n    img = crop_image_from_gray(img)\n    img=cv2.addWeighted ( img,4, cv2.GaussianBlur( img , (0,0) , sigmaX) ,-4 ,128)\n    return img \ndef preprocess_image(image_path, width, height, new_data=False):\n    img = cv2.imread(image_path)\n#     img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)    \n#     img = crop_image_from_gray(img)\n    img = circle_crop(img)\n    img = cv2.resize(img, (width,height))\n#     img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), 20) ,-4 ,128)\n\n    return img","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:00:52.046722Z","iopub.execute_input":"2023-04-26T10:00:52.047091Z","iopub.status.idle":"2023-04-26T10:00:52.068069Z","shell.execute_reply.started":"2023-04-26T10:00:52.047035Z","shell.execute_reply":"2023-04-26T10:00:52.066602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_samples(df, columns=4, rows=3):\n    fig=plt.figure(figsize=(5*columns, 4*rows))\n\n    for i in range(columns*rows):\n        image_path = df.loc[i,'image']\n        print(image_path)\n        image_id = df.loc[i,'level']\n        img = preprocess_image(f'{image_path}', width=WIDTH, height=HEIGHT)\n        fig.add_subplot(rows, columns, i+1)\n        plt.title(image_id)\n        plt.imshow(img)\n    \n    plt.tight_layout()\n\ndisplay_samples(train_df)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-26T10:00:52.069665Z","iopub.execute_input":"2023-04-26T10:00:52.069973Z","iopub.status.idle":"2023-04-26T10:00:57.817144Z","shell.execute_reply.started":"2023-04-26T10:00:52.069911Z","shell.execute_reply":"2023-04-26T10:00:57.815995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Processing Images","metadata":{}},{"cell_type":"markdown","source":"__UPDATE:__ Here we are reading just the validation set. In order to use 320x320 images, we are going to load one bucket at a time only when needed. This will let our code run without memory-related errors.","metadata":{}},{"cell_type":"code","source":"# validation set\nN = val_df.shape[0]\nx_val = np.empty((N, HEIGHT, WIDTH, 3), dtype=np.uint8)\n\nfor i, image_id in enumerate(tqdm_notebook(val_df['image'])):\n    x_val[i, :, :, :] = preprocess_image(\n        f'{image_id}',\n        height=HEIGHT, width=WIDTH, new_data=True\n    )","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:00:57.818766Z","iopub.execute_input":"2023-04-26T10:00:57.819117Z","iopub.status.idle":"2023-04-26T10:09:51.248271Z","shell.execute_reply.started":"2023-04-26T10:00:57.819053Z","shell.execute_reply":"2023-04-26T10:09:51.247432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = pd.get_dummies(train_df['level']).values\ny_val = pd.get_dummies(val_df['level']).values\n\nprint(y_train.shape)\nprint(x_val.shape)\nprint(y_val.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:09:51.249787Z","iopub.execute_input":"2023-04-26T10:09:51.250368Z","iopub.status.idle":"2023-04-26T10:09:51.262659Z","shell.execute_reply.started":"2023-04-26T10:09:51.250314Z","shell.execute_reply":"2023-04-26T10:09:51.261499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_valid = pd.get_dummies(val_df['level']).values","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:09:51.264643Z","iopub.execute_input":"2023-04-26T10:09:51.265074Z","iopub.status.idle":"2023-04-26T10:09:51.271445Z","shell.execute_reply.started":"2023-04-26T10:09:51.264893Z","shell.execute_reply":"2023-04-26T10:09:51.270447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating multilabels\n\nThay vì dự đoán một nhãn đơn lẻ, chúng ta sẽ thay đổi mục tiêu để trở thành một bài toán đa nhãn (multilabel); tức là, nếu nhãn là một lớp nhất định, thì nó bao gồm tất cả các lớp trước đó. Ví dụ, mã hóa một retinopathy của lớp 4 thường là [0, 0, 0, 1], nhưng trong trường hợp của chúng ta, chúng ta sẽ dự đoán [1, 1, 1, 1]. Để biết thêm chi tiết, vui lòng xem Kernel của Lex.","metadata":{}},{"cell_type":"code","source":"y_train_multi = np.empty(y_train.shape, dtype=y_train.dtype)\ny_train_multi[:, 4] = y_train[:, 4]\n\nfor i in range(3, -1, -1):\n    y_train_multi[:, i] = np.logical_or(y_train[:, i], y_train_multi[:, i+1])\n\ny_val_multi = np.empty(y_val.shape, dtype=y_val.dtype)\ny_val_multi[:, 4] = y_val[:, 4]\n\nfor i in range(3, -1, -1):\n    y_val_multi[:, i] = np.logical_or(y_val[:, i], y_val_multi[:, i+1])\n\nprint(\"Y_train multi: {}\".format(y_train_multi.shape))\nprint(\"Y_val multi: {}\".format(y_val_multi.shape))","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:09:51.272938Z","iopub.execute_input":"2023-04-26T10:09:51.273475Z","iopub.status.idle":"2023-04-26T10:09:51.285145Z","shell.execute_reply.started":"2023-04-26T10:09:51.273350Z","shell.execute_reply":"2023-04-26T10:09:51.284534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = y_train_multi\ny_val = y_val_multi","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:09:51.286877Z","iopub.execute_input":"2023-04-26T10:09:51.287312Z","iopub.status.idle":"2023-04-26T10:09:51.296645Z","shell.execute_reply.started":"2023-04-26T10:09:51.287263Z","shell.execute_reply":"2023-04-26T10:09:51.295692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # delete the uneeded df\n# del train\n# del df\n# del val_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:09:51.299760Z","iopub.execute_input":"2023-04-26T10:09:51.300111Z","iopub.status.idle":"2023-04-26T10:09:51.441785Z","shell.execute_reply.started":"2023-04-26T10:09:51.300055Z","shell.execute_reply":"2023-04-26T10:09:51.440950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating keras callback for QWK\n\n---\n\nI had to change this function, in order to consider the best kappa score among all the buckets.","metadata":{}},{"cell_type":"code","source":"class Metrics(Callback):\n\n    def on_epoch_end(self, epoch, logs={}):\n        X_val, y_val = self.validation_data[:2]\n        y_val = y_val.sum(axis=1) - 1\n        \n        y_pred = self.model.predict(X_val) > 0.5\n        y_pred = y_pred.astype(int).sum(axis=1) - 1\n\n        _val_kappa = cohen_kappa_score(\n            y_val,\n            y_pred, \n            weights='quadratic'\n        )\n\n        self.val_kappas.append(_val_kappa)\n\n        print(f\"val_kappa: {_val_kappa:.4f}\")\n        \n        if _val_kappa == max(self.val_kappas):\n            print(\"Validation Kappa has improved. Saving model.\")\n            self.model.save('model.h5')\n\n        return","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:09:51.443398Z","iopub.execute_input":"2023-04-26T10:09:51.443957Z","iopub.status.idle":"2023-04-26T10:09:51.455560Z","shell.execute_reply.started":"2023-04-26T10:09:51.443894Z","shell.execute_reply":"2023-04-26T10:09:51.454660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Generator","metadata":{}},{"cell_type":"code","source":"def create_datagen():\n    return ImageDataGenerator(\n        horizontal_flip=True,\n        vertical_flip=True,\n        zoom_range= 0.3,\n        brightness_range=(0.5, 2),\n        fill_mode='constant',\n        cval=0\n    )","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:09:51.459356Z","iopub.execute_input":"2023-04-26T10:09:51.459626Z","iopub.status.idle":"2023-04-26T10:09:51.469495Z","shell.execute_reply.started":"2023-04-26T10:09:51.459556Z","shell.execute_reply":"2023-04-26T10:09:51.468703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check the differenct kinds of augmentations on the pictures.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 10, figsize=(20, 10))\nax = ax.ravel()\n\nimg = x_val[0].reshape(1,x_val[0].shape[0],x_val[0].shape[1], x_val[0].shape[2])\n\nax[0].imshow(img[0].astype('uint8'))\nax[1].imshow(next(ImageDataGenerator().flow(img))[0].astype('uint8'))\nax[2].imshow(next(ImageDataGenerator(horizontal_flip=True, fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[3].imshow(next(ImageDataGenerator(vertical_flip=True,fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[4].imshow(next(ImageDataGenerator(rotation_range=360, fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[5].imshow(next(ImageDataGenerator(zoom_range= (0.65,1), fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[6].imshow(next(ImageDataGenerator(height_shift_range=0.15, fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[7].imshow(next(ImageDataGenerator(width_shift_range=0.15, fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[8].imshow(next(ImageDataGenerator(brightness_range=(0.5, 2), fill_mode='constant', cval=0).flow(img))[0].astype('uint8'))\nax[9].imshow(next(ImageDataGenerator(horizontal_flip=True,\n                                     vertical_flip=True,\n                                     rotation_range=360,zoom_range= (0.65,1),\n                                     brightness_range=(0.5, 2),\n                                     fill_mode='constant',cval=0).flow(img))[0].astype('uint8'))\n","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:09:51.473010Z","iopub.execute_input":"2023-04-26T10:09:51.473279Z","iopub.status.idle":"2023-04-26T10:09:53.030919Z","shell.execute_reply.started":"2023-04-26T10:09:51.473228Z","shell.execute_reply":"2023-04-26T10:09:53.029737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model: EfficientNetB3","metadata":{}},{"cell_type":"code","source":"efficientnetb3 = EfficientNetB3(\n        weights=None,\n        input_shape=(HEIGHT,WIDTH,3),\n        include_top=False\n                   )\n\nefficientnetb3.load_weights(\"../input/efficientnet-keras-weights-b0b5/efficientnet-b3_imagenet_1000_notop.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:29:02.606868Z","iopub.execute_input":"2023-04-26T10:29:02.607211Z","iopub.status.idle":"2023-04-26T10:29:19.894536Z","shell.execute_reply.started":"2023-04-26T10:29:02.607153Z","shell.execute_reply":"2023-04-26T10:29:19.893630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install git+https://github.com/qubvel/segmentation_models\n\nimport segmentation_models as sm\n\nmodel = tf.keras.Sequential([\n        sm.EfficientNetB7(\n            input_shape=(IM_Z, IM_Z, 3),\n            weights='imagenet',\n            include_top=False\n        ),\n        tf.keras.layers.GlobalAveragePooling2D(),\n        tf.keras.layers.Dense(1, activation='sigmoid')\n    ])","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:32:46.825761Z","iopub.execute_input":"2023-04-26T10:32:46.826172Z","iopub.status.idle":"2023-04-26T10:33:29.278988Z","shell.execute_reply.started":"2023-04-26T10:32:46.826106Z","shell.execute_reply":"2023-04-26T10:33:29.277395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model():\n    model = Sequential()\n    model.add(efficientnetb3)\n    model.add(layers.GlobalAveragePooling2D())\n    model.add(layers.Dropout(0.5))\n    model.add(layers.BatchNormalization())\n    model.add(layers.Dense(5, activation='sigmoid'))\n    \n    model.compile(\n        loss='binary_crossentropy',\n        #loss=kappa_loss,\n        optimizer=Adam(lr=1e-4,decay=1e-6),\n        metrics=['accuracy']\n    )\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:36:10.370548Z","iopub.execute_input":"2023-04-26T10:36:10.370933Z","iopub.status.idle":"2023-04-26T10:36:10.377924Z","shell.execute_reply.started":"2023-04-26T10:36:10.370863Z","shell.execute_reply":"2023-04-26T10:36:10.376952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = build_model()\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:36:24.587338Z","iopub.execute_input":"2023-04-26T10:36:24.587670Z","iopub.status.idle":"2023-04-26T10:36:32.363878Z","shell.execute_reply.started":"2023-04-26T10:36:24.587616Z","shell.execute_reply":"2023-04-26T10:36:32.362317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(model, to_file='model.png', show_shapes=True, show_layer_names=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:36:36.625973Z","iopub.execute_input":"2023-04-26T10:36:36.626308Z","iopub.status.idle":"2023-04-26T10:36:39.524434Z","shell.execute_reply.started":"2023-04-26T10:36:36.626253Z","shell.execute_reply":"2023-04-26T10:36:39.523417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pretraining with old Data","metadata":{}},{"cell_type":"code","source":"bucket_num = 8\ndiv = round(train_df.shape[0]/bucket_num)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:36:50.116952Z","iopub.execute_input":"2023-04-26T10:36:50.117312Z","iopub.status.idle":"2023-04-26T10:36:50.122712Z","shell.execute_reply.started":"2023-04-26T10:36:50.117252Z","shell.execute_reply":"2023-04-26T10:36:50.121534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_init = {\n    'val_loss': [0.0],\n    'val_acc': [0.0],\n    'loss': [0.0], \n    'acc': [0.0],\n    'bucket': [0.0]\n}\nresults = pd.DataFrame(df_init)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:36:58.744739Z","iopub.execute_input":"2023-04-26T10:36:58.745093Z","iopub.status.idle":"2023-04-26T10:36:58.751014Z","shell.execute_reply.started":"2023-04-26T10:36:58.745034Z","shell.execute_reply":"2023-04-26T10:36:58.750175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I found that changing the nr. of epochs for each bucket helped in terms of performances\nepochs = [5,5,5,5,5,5,5,5]\nkappa_metrics = Metrics()\nkappa_metrics.val_kappas = []\n\nlearn_control = ReduceLROnPlateau(monitor='val_acc', patience=5,\n                                  verbose=1,factor=.2, min_lr=1e-7)\n\ncheckpoint = ModelCheckpoint('val_model.h5', monitor='val_loss', verbose=1, save_best_only=True, mode='min')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:37:01.392924Z","iopub.execute_input":"2023-04-26T10:37:01.393267Z","iopub.status.idle":"2023-04-26T10:37:01.399524Z","shell.execute_reply.started":"2023-04-26T10:37:01.393209Z","shell.execute_reply":"2023-04-26T10:37:01.398571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nconfig = tf.compat.v1.ConfigProto()\nconfig.gpu_options.allow_growth = True\nsession = tf.compat.v1.Session(config=config)\ntf.compat.v1.keras.backend.set_session(session)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:37:03.967982Z","iopub.execute_input":"2023-04-26T10:37:03.968303Z","iopub.status.idle":"2023-04-26T10:37:03.983042Z","shell.execute_reply.started":"2023-04-26T10:37:03.968247Z","shell.execute_reply":"2023-04-26T10:37:03.982154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(0,bucket_num):\n    if i != (bucket_num-1):\n        print(\"Bucket Nr: {}\".format(i))\n        \n        N = train_df.iloc[i*div:(1+i)*div].shape[0]\n        x_train = np.empty((N, HEIGHT, WIDTH, 3), dtype=np.uint8)\n        for j, image_id in enumerate(tqdm_notebook(train_df.iloc[i*div:(1+i)*div,0])):\n            x_train[j, :, :, :] = preprocess_image(f'{image_id}', height=HEIGHT, width=WIDTH)\n\n        data_generator = create_datagen().flow(x_train, y_train[i*div:(1+i)*div,:], batch_size=BATCH_SIZE)\n        history = model.fit_generator(\n                        data_generator,\n                        steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                        epochs=epochs[i],\n                        validation_data=(x_val, y_val),\n                        callbacks=[kappa_metrics, learn_control, checkpoint]\n                        )\n        \n        dic = history.history\n        df_model = pd.DataFrame(dic)\n        df_model['bucket'] = i\n    else:\n        print(\"Bucket Nr: {}\".format(i))\n        \n        N = train_df.iloc[i*div:].shape[0]\n        x_train = np.empty((N, HEIGHT, WIDTH, 3), dtype=np.uint8)\n        for j, image_id in enumerate(tqdm_notebook(train_df.iloc[i*div:,0])):\n            x_train[j, :, :, :] = preprocess_image(f'{image_id}', height=HEIGHT, width=WIDTH)\n        data_generator = create_datagen().flow(x_train, y_train[i*div:,:], batch_size=BATCH_SIZE)\n        \n        history = model.fit_generator(\n                        data_generator,\n                        steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                        epochs=epochs[i],\n                        validation_data=(x_val, y_val),\n                        callbacks=[kappa_metrics, learn_control, checkpoint]\n                        )\n        \n        dic = history.history\n        df_model = pd.DataFrame(dic)\n        df_model['bucket'] = i\n\n    results = results.append(df_model)\n    \n    del data_generator\n    del x_train\n    gc.collect()\n    \n    print('-'*40)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:41:37.383694Z","iopub.execute_input":"2023-04-26T10:41:37.384085Z","iopub.status.idle":"2023-04-26T12:14:56.474513Z","shell.execute_reply.started":"2023-04-26T10:41:37.384026Z","shell.execute_reply":"2023-04-26T12:14:56.473257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = results.iloc[1:]\nresults['kappa'] = kappa_metrics.val_kappas\nresults = results.reset_index()\nresults = results.rename(index=str, columns={\"index\": \"epoch\"})\nresults","metadata":{"execution":{"iopub.status.busy":"2023-04-26T12:21:43.125076Z","iopub.execute_input":"2023-04-26T12:21:43.125423Z","iopub.status.idle":"2023-04-26T12:21:43.166623Z","shell.execute_reply.started":"2023-04-26T12:21:43.125367Z","shell.execute_reply":"2023-04-26T12:21:43.165775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results[['loss', 'val_loss']].plot()\nresults[['acc', 'val_acc']].plot()\nresults[['kappa']].plot()\nresults.to_csv('model_results.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T12:21:51.453644Z","iopub.execute_input":"2023-04-26T12:21:51.453994Z","iopub.status.idle":"2023-04-26T12:21:52.672496Z","shell.execute_reply.started":"2023-04-26T12:21:51.453935Z","shell.execute_reply":"2023-04-26T12:21:52.671145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Tinh chỉnh với dữ liệu mới\nCreate New Train and Validation Set to finetune our model","metadata":{}},{"cell_type":"code","source":"# model.load_weights('val_model.h5')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:10:14.001182Z","iopub.status.idle":"2023-04-26T10:10:14.001879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_val, y_train, y_val = train_test_split(\n    x_val, y_val, \n    test_size=0.2, \n    random_state=nr_seed\n)\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T12:22:23.514011Z","iopub.execute_input":"2023-04-26T12:22:23.514369Z","iopub.status.idle":"2023-04-26T12:22:24.424922Z","shell.execute_reply.started":"2023-04-26T12:22:23.514314Z","shell.execute_reply":"2023-04-26T12:22:24.423867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_generator = create_datagen().flow(x_train, y_train, batch_size=BATCH_SIZE)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T12:22:27.010932Z","iopub.execute_input":"2023-04-26T12:22:27.011257Z","iopub.status.idle":"2023-04-26T12:22:28.202928Z","shell.execute_reply.started":"2023-04-26T12:22:27.011199Z","shell.execute_reply":"2023-04-26T12:22:28.202110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit_generator(\n                data_generator,\n                steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                epochs=20,\n                validation_data=(x_val, y_val),\n                callbacks=[kappa_metrics,learn_control,checkpoint]\n                )","metadata":{"execution":{"iopub.status.busy":"2023-04-26T12:22:31.083179Z","iopub.execute_input":"2023-04-26T12:22:31.083507Z","iopub.status.idle":"2023-04-26T12:47:34.178892Z","shell.execute_reply.started":"2023-04-26T12:22:31.083453Z","shell.execute_reply":"2023-04-26T12:47:34.177969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_generator = create_datagen().flow(x_train, y_train, batch_size=BATCH_SIZE)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T12:47:34.184096Z","iopub.execute_input":"2023-04-26T12:47:34.186472Z","iopub.status.idle":"2023-04-26T12:47:35.567677Z","shell.execute_reply.started":"2023-04-26T12:47:34.186413Z","shell.execute_reply":"2023-04-26T12:47:35.566696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit_generator(\n                data_generator,\n                steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                epochs=20,\n                validation_data=(x_val, y_val),\n                callbacks=[kappa_metrics,learn_control,checkpoint]\n                )","metadata":{"execution":{"iopub.status.busy":"2023-04-26T12:47:35.569149Z","iopub.execute_input":"2023-04-26T12:47:35.569473Z","iopub.status.idle":"2023-04-26T13:12:26.700373Z","shell.execute_reply.started":"2023-04-26T12:47:35.569409Z","shell.execute_reply":"2023-04-26T13:12:26.699407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_weights('val_model.h5')\npred_val = model.predict(x_val)\nprint(pred_val)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T13:12:26.704543Z","iopub.execute_input":"2023-04-26T13:12:26.706677Z","iopub.status.idle":"2023-04-26T13:12:33.923589Z","shell.execute_reply.started":"2023-04-26T13:12:26.706618Z","shell.execute_reply":"2023-04-26T13:12:33.922572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pred_val_binary = (pred_val > 0.5).astype(int) # threshold at 0.5 to obtain binary predictions\n# accuracy = accuracy_score(y_valid, pred_val)\n# print(\"Accuracy: \", accuracy)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T13:20:57.027308Z","iopub.execute_input":"2023-04-26T13:20:57.027663Z","iopub.status.idle":"2023-04-26T13:20:57.031700Z","shell.execute_reply.started":"2023-04-26T13:20:57.027601Z","shell.execute_reply":"2023-04-26T13:20:57.030698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Tối ưu ngưỡng","metadata":{}},{"cell_type":"code","source":"def compute_score_inv(threshold):\n    y1 = pred_val > threshold\n    y1 = y1.astype(int).sum(axis=1) - 1\n    y2 = y_val.sum(axis=1) - 1\n    score = cohen_kappa_score(y1, y2, weights='quadratic')\n    return 1 - score\nsimplex = scipy.optimize.minimize(compute_score_inv, 0.5, method='nelder-mead')\n\nbest_threshold = simplex['x'][0]\nprint(best_threshold)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T13:18:28.454455Z","iopub.execute_input":"2023-04-26T13:18:28.454779Z","iopub.status.idle":"2023-04-26T13:18:29.184311Z","shell.execute_reply.started":"2023-04-26T13:18:28.454721Z","shell.execute_reply":"2023-04-26T13:18:29.183430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test","metadata":{"execution":{"iopub.status.busy":"2023-04-26T10:10:14.018663Z","iopub.status.idle":"2023-04-26T10:10:14.019347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Load pre-trained model\n# # loaded_model = model.load_weights('val_model.h5')\n\n# # Define image path and size\n# image_id = '/kaggle/input/diabetic-retinopathy-resized/resized_train/resized_train/12334_left.jpeg'\n\n\n# # Initialize x_test array\n# x_test = np.empty((1, HEIGHT, WIDTH, 3), dtype=np.uint8)\n\n# # Preprocess image and assign to x_test\n# x_test[0, :, :, :] = preprocess_image(\n#         f'{image_id}',\n#         height=HEIGHT, width=WIDTH, new_data=True\n#     )\n\n# # Make prediction with loaded model\n\n# model.load_weights('model.h5')\n# prev_test = model.predict(x_test)\n\n\n\n# # Display prediction results\n# print('Kết quả dự đoán:', prev_test)\n# max_index = np.argmax(prev_test)\n\n# print(\"giá trị lớn nhất:\", max_index)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-26T13:19:29.725545Z","iopub.execute_input":"2023-04-26T13:19:29.725919Z","iopub.status.idle":"2023-04-26T13:19:29.778645Z","shell.execute_reply.started":"2023-04-26T13:19:29.725856Z","shell.execute_reply":"2023-04-26T13:19:29.777205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class OptimizedRounder(object):\n    def __init__(self):\n        self.coef_ = 0\n\n    def _kappa_loss(self, coef, X, y):\n        X_p = np.copy(X)\n        for i, pred in enumerate(X_p):\n            if pred < coef[0]:\n                X_p[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                X_p[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                X_p[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                X_p[i] = 3\n            else:\n                X_p[i] = 4\n\n        ll = cohen_kappa_score(y, X_p, weights='quadratic')\n        return -ll\n\n    def fit(self, X, y):\n        loss_partial = partial(self._kappa_loss, X=X, y=y)\n        initial_coef = [0.5, 1.5, 2.5, 3.5]\n        self.coef_ = sp.optimize.minimize(loss_partial, initial_coef, method='nelder-mead')\n        print(-loss_partial(self.coef_['x']))\n\n    def predict(self, X, coef):\n        X_p = np.copy(X)\n        for i, pred in enumerate(X_p):\n            if pred < coef[0]:\n                X_p[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                X_p[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                X_p[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                X_p[i] = 3\n            else:\n                X_p[i] = 4\n        return X_p\n\n    def coefficients(self):\n        return self.coef_['x']","metadata":{"execution":{"iopub.status.busy":"2023-04-26T15:58:34.773931Z","iopub.execute_input":"2023-04-26T15:58:34.774252Z","iopub.status.idle":"2023-04-26T15:58:34.787045Z","shell.execute_reply.started":"2023-04-26T15:58:34.774180Z","shell.execute_reply":"2023-04-26T15:58:34.786152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nmodel.load_weights('/kaggle/input/modelh5')\ny_val_pred = model.predict(x_val)\n\noptR = OptimizedRounder()\noptR.fit(y_val_pred, y_val)\ncoefficients = optR.coefficients()\nprint(f'Coefficients: {coefficients}')\ny_val_pred = optR.predict(y_val_pred, coefficients)\n\nscore = cohen_kappa_score(y_val_pred, y_val, weights='quadratic')\n\nprint('Optimized Validation QWK score: {}'.format(score))\nprint('Not Optimized Validation QWK score: {}'.format(max(results.kappa)))","metadata":{"execution":{"iopub.status.busy":"2023-04-26T16:01:37.326569Z","iopub.execute_input":"2023-04-26T16:01:37.326911Z","iopub.status.idle":"2023-04-26T16:01:37.340160Z","shell.execute_reply.started":"2023-04-26T16:01:37.326855Z","shell.execute_reply":"2023-04-26T16:01:37.338915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(/kaggle/input/modelh5)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T16:02:04.698402Z","iopub.execute_input":"2023-04-26T16:02:04.698761Z","iopub.status.idle":"2023-04-26T16:02:04.704383Z","shell.execute_reply.started":"2023-04-26T16:02:04.698700Z","shell.execute_reply":"2023-04-26T16:02:04.703108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[](http://)","metadata":{}}]}