{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"},{"sourceId":418031,"sourceType":"datasetVersion","datasetId":131128},{"sourceId":2812287,"sourceType":"datasetVersion","datasetId":1719146},{"sourceId":2819730,"sourceType":"datasetVersion","datasetId":1723812},{"sourceId":2823796,"sourceType":"datasetVersion","datasetId":1727005},{"sourceId":5579822,"sourceType":"datasetVersion","datasetId":3211385},{"sourceId":5579826,"sourceType":"datasetVersion","datasetId":3211388}],"dockerImageVersionId":30628,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# To have reproducible results and compare them\nnr_seed = 2023\nimport numpy as np \nnp.random.seed(nr_seed)\nimport tensorflow as tf\ntf.random.set_seed(nr_seed)\nprint(tf.__version__)\n\n# Libraries\nimport json\nimport math\nimport os\n\n\nimport scipy as sp\nfrom functools import partial\nfrom collections import Counter\nimport json\n\nfrom PIL import Image\nimport numpy as np\nfrom keras import backend as K\nfrom keras import layers\nfrom tensorflow.keras.applications import EfficientNetV2B3\nfrom keras.callbacks import Callback, ModelCheckpoint\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential\nfrom keras.optimizers import Adam\nfrom tensorflow.keras.utils import plot_model\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport gc\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-17T01:35:47.724441Z","iopub.execute_input":"2023-12-17T01:35:47.725153Z","iopub.status.idle":"2023-12-17T01:36:02.993419Z","shell.execute_reply.started":"2023-12-17T01:35:47.725114Z","shell.execute_reply":"2023-12-17T01:36:02.992631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install scikit-learn","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:36:02.994968Z","iopub.execute_input":"2023-12-17T01:36:02.995466Z","iopub.status.idle":"2023-12-17T01:36:06.718358Z","shell.execute_reply.started":"2023-12-17T01:36:02.995433Z","shell.execute_reply":"2023-12-17T01:36:06.717410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import cohen_kappa_score","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:36:06.719530Z","iopub.execute_input":"2023-12-17T01:36:06.719801Z","iopub.status.idle":"2023-12-17T01:36:07.292884Z","shell.execute_reply.started":"2023-12-17T01:36:06.719773Z","shell.execute_reply":"2023-12-17T01:36:07.292190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# detect and init the TPU\n#tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\n\n# instantiate a distribution strategy\n#tpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)\n\ntpu = tf.distribute.cluster_resolver.TPUClusterResolver()\ntf.config.experimental_connect_to_cluster(tpu)\ntf.tpu.experimental.initialize_tpu_system(tpu)\nstrategy = tf.distribute.experimental.TPUStrategy(tpu)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:36:07.294563Z","iopub.execute_input":"2023-12-17T01:36:07.295463Z","iopub.status.idle":"2023-12-17T01:36:15.461055Z","shell.execute_reply.started":"2023-12-17T01:36:07.295432Z","shell.execute_reply":"2023-12-17T01:36:15.460256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image size\nim_size = 224\n# Batch size\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:36:15.462054Z","iopub.execute_input":"2023-12-17T01:36:15.462302Z","iopub.status.idle":"2023-12-17T01:36:15.465950Z","shell.execute_reply.started":"2023-12-17T01:36:15.462275Z","shell.execute_reply":"2023-12-17T01:36:15.465247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:36:15.466900Z","iopub.execute_input":"2023-12-17T01:36:15.467149Z","iopub.status.idle":"2023-12-17T01:36:15.492291Z","shell.execute_reply.started":"2023-12-17T01:36:15.467123Z","shell.execute_reply":"2023-12-17T01:36:15.491591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/train-df/train_df_withoutINDEX.csv')\nval_df = pd.read_csv('../input/val-df/val_df_withoutINDEX.csv')\n\n\n\nprint(train_df.shape)\nprint(val_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:36:15.493215Z","iopub.execute_input":"2023-12-17T01:36:15.493467Z","iopub.status.idle":"2023-12-17T01:36:15.609259Z","shell.execute_reply.started":"2023-12-17T01:36:15.493441Z","shell.execute_reply":"2023-12-17T01:36:15.608403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!apt-get update && apt-get install -y python3-opencv\n!pip install opencv-python\n!apt-get update && apt-get install -y python3-tqdm\n!pip install tqdm","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:36:15.610210Z","iopub.execute_input":"2023-12-17T01:36:15.610465Z","iopub.status.idle":"2023-12-17T01:38:20.720159Z","shell.execute_reply.started":"2023-12-17T01:36:15.610439Z","shell.execute_reply":"2023-12-17T01:38:20.718904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport tqdm","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:38:20.721753Z","iopub.execute_input":"2023-12-17T01:38:20.722049Z","iopub.status.idle":"2023-12-17T01:38:21.510061Z","shell.execute_reply.started":"2023-12-17T01:38:20.722020Z","shell.execute_reply":"2023-12-17T01:38:21.509117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_samples(df, columns=4, rows=3):\n    fig=plt.figure(figsize=(5*columns, 4*rows))\n\n    for i in range(columns*rows):\n        image_path = df.loc[i,'id_code']\n        image_id = df.loc[i,'diagnosis']\n        img = cv2.imread(f'{image_path}')\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        #img = crop_image_from_gray(img)\n        img = cv2.resize(img, (im_size,im_size))\n        img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), im_size/40) ,-4 ,128)\n        \n        fig.add_subplot(rows, columns, i+1)\n        plt.title(image_id)\n        plt.imshow(img)\n    \n    plt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:38:21.513094Z","iopub.execute_input":"2023-12-17T01:38:21.513365Z","iopub.status.idle":"2023-12-17T01:38:21.519975Z","shell.execute_reply.started":"2023-12-17T01:38:21.513338Z","shell.execute_reply":"2023-12-17T01:38:21.519089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display train images\ndisplay_samples(train_df)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:38:21.520871Z","iopub.execute_input":"2023-12-17T01:38:21.521095Z","iopub.status.idle":"2023-12-17T01:38:24.746713Z","shell.execute_reply.started":"2023-12-17T01:38:21.521072Z","shell.execute_reply":"2023-12-17T01:38:24.745733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display train images\ndisplay_samples(val_df)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:38:24.748916Z","iopub.execute_input":"2023-12-17T01:38:24.749212Z","iopub.status.idle":"2023-12-17T01:38:29.499446Z","shell.execute_reply.started":"2023-12-17T01:38:24.749175Z","shell.execute_reply":"2023-12-17T01:38:29.498385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = BATCH_SIZE*2","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:38:29.500970Z","iopub.execute_input":"2023-12-17T01:38:29.501250Z","iopub.status.idle":"2023-12-17T01:38:29.505503Z","shell.execute_reply.started":"2023-12-17T01:38:29.501221Z","shell.execute_reply":"2023-12-17T01:38:29.504577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:38:29.506598Z","iopub.execute_input":"2023-12-17T01:38:29.506979Z","iopub.status.idle":"2023-12-17T01:38:29.525942Z","shell.execute_reply.started":"2023-12-17T01:38:29.506950Z","shell.execute_reply":"2023-12-17T01:38:29.524998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:38:29.526974Z","iopub.execute_input":"2023-12-17T01:38:29.527241Z","iopub.status.idle":"2023-12-17T01:38:29.544391Z","shell.execute_reply.started":"2023-12-17T01:38:29.527206Z","shell.execute_reply":"2023-12-17T01:38:29.543515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def crop_image_from_gray(img,tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n            img = np.stack([img1,img2,img3],axis=-1)\n        return img\n\n    \n    \ndef preprocess_image(image_path, desired_size=224):\n    img = cv2.imread(image_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = crop_image_from_gray(img)\n    img = cv2.resize(img, (desired_size,desired_size))\n    img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), desired_size/30) ,-4 ,128)\n    return img\n\n\ndef preprocess_image_old(image_path, desired_size=224):\n    img = cv2.imread(image_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    #img = crop_image_from_gray(img)\n    img = cv2.resize(img, (desired_size,desired_size))\n    img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), desired_size/40) ,-4 ,128)\n    return img","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:38:29.545414Z","iopub.execute_input":"2023-12-17T01:38:29.545667Z","iopub.status.idle":"2023-12-17T01:38:29.574208Z","shell.execute_reply.started":"2023-12-17T01:38:29.545641Z","shell.execute_reply":"2023-12-17T01:38:29.573393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# validation settebook\nN = val_df.shape[0]\nx_val = np.empty((N, im_size, im_size, 3), dtype=np.uint8)\n\nfor i, image_id in tqdm(enumerate(val_df['id_code']), total=N):\n    x_val[i, :, :, :] = preprocess_image(\n        f'{image_id}',\n        desired_size=im_size\n    )","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:38:29.575195Z","iopub.execute_input":"2023-12-17T01:38:29.575457Z","iopub.status.idle":"2023-12-17T01:49:38.606788Z","shell.execute_reply.started":"2023-12-17T01:38:29.575430Z","shell.execute_reply":"2023-12-17T01:49:38.605839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = train_df['diagnosis'].values\ny_val = val_df['diagnosis'].values\n\nprint(y_train.shape)\nprint(x_val.shape)\nprint(y_val.shape)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:49:38.607955Z","iopub.execute_input":"2023-12-17T01:49:38.608291Z","iopub.status.idle":"2023-12-17T01:49:38.614018Z","shell.execute_reply.started":"2023-12-17T01:49:38.608257Z","shell.execute_reply":"2023-12-17T01:49:38.613211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Metrics(Callback):\n    def __init__(self, validation_data=()):\n        super().__init__()\n        self.validation_data = validation_data\n\n    def on_epoch_end(self, epoch, logs=None):\n        x_val, y_val = self.validation_data[0], self.validation_data[1]\n        \n        y_pred = self.model.predict(x_val)\n        \n        coef = [0.5, 1.5, 2.5, 3.5]\n\n        for i, pred in enumerate(y_pred):\n            if pred < coef[0]:\n                y_pred[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                y_pred[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                y_pred[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                y_pred[i] = 3\n            else:\n                y_pred[i] = 4\n\n        _val_kappa = cohen_kappa_score(\n            y_val,\n            y_pred, \n            weights='quadratic')\n        self.val_kappas.append(_val_kappa)\n        print(f\"val_kappa: {_val_kappa:.4f}\")\n\n\n        if _val_kappa == max(self.val_kappas):\n            print(\"Validation Kappa has improved. Saving model.\")\n            self.model.save('test204.h5')\n\n        return","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:49:38.615010Z","iopub.execute_input":"2023-12-17T01:49:38.615282Z","iopub.status.idle":"2023-12-17T01:49:38.625929Z","shell.execute_reply.started":"2023-12-17T01:49:38.615254Z","shell.execute_reply":"2023-12-17T01:49:38.625175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_datagen():\n    return ImageDataGenerator(\n        horizontal_flip = True,\n        vertical_flip = True,\n        rotation_range = 160,\n        zoom_range=0.35\n    )","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:49:38.626861Z","iopub.execute_input":"2023-12-17T01:49:38.627121Z","iopub.status.idle":"2023-12-17T01:49:38.637432Z","shell.execute_reply.started":"2023-12-17T01:49:38.627094Z","shell.execute_reply":"2023-12-17T01:49:38.636733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model():\n    effnet = EfficientNetV2B3(\n    input_shape=(im_size,im_size,3),\n    weights='imagenet',\n    include_top=False)\n    model = Sequential()\n    model.add(effnet)\n    model.add(layers.Dropout(0.25))\n    model.add(layers.Dense(2048))\n    model.add(layers.LeakyReLU())\n    model.add(layers.GlobalAveragePooling2D())\n    model.add(layers.Dropout(0.5))\n    model.add(layers.Dense(1, activation='linear'))\n    \n    \n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:49:38.638268Z","iopub.execute_input":"2023-12-17T01:49:38.638543Z","iopub.status.idle":"2023-12-17T01:49:38.646858Z","shell.execute_reply.started":"2023-12-17T01:49:38.638507Z","shell.execute_reply":"2023-12-17T01:49:38.646053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflow","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:49:38.647674Z","iopub.execute_input":"2023-12-17T01:49:38.647921Z","iopub.status.idle":"2023-12-17T01:49:42.135665Z","shell.execute_reply.started":"2023-12-17T01:49:38.647895Z","shell.execute_reply":"2023-12-17T01:49:42.134381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:49:42.137058Z","iopub.execute_input":"2023-12-17T01:49:42.137349Z","iopub.status.idle":"2023-12-17T01:49:42.141791Z","shell.execute_reply.started":"2023-12-17T01:49:42.137320Z","shell.execute_reply":"2023-12-17T01:49:42.141131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    model = build_model()\n    model.compile(\n        loss='mean_squared_error',\n        #optimizer=Adam(lr=0.001,decay=1e-6),\n        optimizer = tf.keras.optimizers.Adam(learning_rate=0.0001),\n        metrics=['mae']\n    )","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:49:42.142678Z","iopub.execute_input":"2023-12-17T01:49:42.142934Z","iopub.status.idle":"2023-12-17T01:50:17.929953Z","shell.execute_reply.started":"2023-12-17T01:49:42.142909Z","shell.execute_reply":"2023-12-17T01:50:17.928760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:50:17.931102Z","iopub.execute_input":"2023-12-17T01:50:17.931379Z","iopub.status.idle":"2023-12-17T01:50:17.976166Z","shell.execute_reply.started":"2023-12-17T01:50:17.931351Z","shell.execute_reply":"2023-12-17T01:50:17.975376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_bucket = 4\ndiv = round(train_df.shape[0]/num_bucket)\ndiv","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:50:17.977122Z","iopub.execute_input":"2023-12-17T01:50:17.977384Z","iopub.status.idle":"2023-12-17T01:50:17.982778Z","shell.execute_reply.started":"2023-12-17T01:50:17.977357Z","shell.execute_reply":"2023-12-17T01:50:17.981979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:50:17.983761Z","iopub.execute_input":"2023-12-17T01:50:17.984030Z","iopub.status.idle":"2023-12-17T01:50:17.992579Z","shell.execute_reply.started":"2023-12-17T01:50:17.984006Z","shell.execute_reply":"2023-12-17T01:50:17.991818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = pd.DataFrame({\n                        'val_loss': [0.0],\n                        'val_mean_absolute_error': [0.0],\n                        'loss': [0.0], \n                        'mean_absolute_error': [0.0],\n                        'bucket': [0.0]\n                        })","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:50:17.996658Z","iopub.execute_input":"2023-12-17T01:50:17.996917Z","iopub.status.idle":"2023-12-17T01:50:18.001909Z","shell.execute_reply.started":"2023-12-17T01:50:17.996892Z","shell.execute_reply":"2023-12-17T01:50:18.001123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Epochs\nepochs = [4,4,4,4]\nkappa_metrics = Metrics(validation_data=(x_val, y_val))\nkappa_metrics.val_kappas = []","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:50:18.002748Z","iopub.execute_input":"2023-12-17T01:50:18.002995Z","iopub.status.idle":"2023-12-17T01:50:18.010409Z","shell.execute_reply.started":"2023-12-17T01:50:18.002969Z","shell.execute_reply":"2023-12-17T01:50:18.009726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(epochs)  ","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:50:18.011188Z","iopub.execute_input":"2023-12-17T01:50:18.011397Z","iopub.status.idle":"2023-12-17T01:50:18.019757Z","shell.execute_reply.started":"2023-12-17T01:50:18.011375Z","shell.execute_reply":"2023-12-17T01:50:18.019012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(0,num_bucket):\n    if i != (num_bucket-1):\n        print(\"Bucket Nr: {}\".format(i))\n        \n        N = train_df.iloc[i*div:(1+i)*div].shape[0]\n        x_train = np.empty((N, im_size, im_size, 3), dtype=np.uint8)\n        for j, image_id in enumerate(tqdm(train_df.iloc[i*div:(1+i)*div,0])):\n            x_train[j, :, :, :] = preprocess_image_old(f'{image_id}', desired_size = im_size)\n\n        data_generator = create_datagen().flow(x_train, y_train[i*div:(1+i)*div], batch_size=BATCH_SIZE)\n        history = model.fit_generator(\n                        data_generator,\n                        steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                        epochs=epochs[i],\n                        validation_data=(x_val, y_val),\n                        callbacks=[kappa_metrics]\n                        )\n        \n        dic = history.history\n        df_model = pd.DataFrame(dic)\n        df_model['bucket'] = i\n    else:\n        print(\"Bucket Nr: {}\".format(i))\n        N = train_df.iloc[i*div:].shape[0]\n        x_train = np.empty((N, im_size, im_size, 3), dtype=np.uint8)\n        for j, image_id in enumerate(tqdm(train_df.iloc[i*div:,0])):\n            x_train[j, :, :, :] = preprocess_image_old(f'{image_id}', desired_size = im_size)\n            \n        data_generator = create_datagen().flow(x_train, y_train[i*div:], batch_size=BATCH_SIZE)\n        history = model.fit_generator(\n                        data_generator,\n                        steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                        epochs=epochs[i],\n                        validation_data=(x_val, y_val),\n                        callbacks=[kappa_metrics]\n                        )\n        \n        dic = history.history\n        df_model = pd.DataFrame(dic)\n        df_model['bucket'] = i\n\n    results = pd.concat([results, df_model], ignore_index=True)\n    del data_generator\n    del x_train\n    gc.collect()\n    print('-'*40)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T01:50:18.020618Z","iopub.execute_input":"2023-12-17T01:50:18.020867Z","iopub.status.idle":"2023-12-17T02:36:04.924856Z","shell.execute_reply.started":"2023-12-17T01:50:18.020843Z","shell.execute_reply":"2023-12-17T02:36:04.923849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = results.iloc[1:]\nkappa_vals = kappa_metrics.val_kappas[:len(results)]\nresults['kappa'] = kappa_vals\nresults = results.reset_index(drop=True)\nresults = results.rename(index=str, columns={\"index\": \"epoch\"})\nprint(max(results.kappa))","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:36:04.926905Z","iopub.execute_input":"2023-12-17T02:36:04.927221Z","iopub.status.idle":"2023-12-17T02:36:04.937009Z","shell.execute_reply.started":"2023-12-17T02:36:04.927190Z","shell.execute_reply":"2023-12-17T02:36:04.936092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results[['loss', 'val_loss']].plot()\nresults[['mae', 'val_mae']].plot()\nresults[['kappa']].plot()\nresults.to_csv('model_results.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:36:04.938009Z","iopub.execute_input":"2023-12-17T02:36:04.938285Z","iopub.status.idle":"2023-12-17T02:36:05.515705Z","shell.execute_reply.started":"2023-12-17T02:36:04.938257Z","shell.execute_reply":"2023-12-17T02:36:05.514797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class OptimizedRounder(object):\n    def __init__(self):\n        self.coef_ = 0\n\n    def _kappa_loss(self, coef, X, y):\n        X_p = np.copy(X)\n        for i, pred in enumerate(X_p):\n            if pred < coef[0]:\n                X_p[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                X_p[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                X_p[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                X_p[i] = 3\n            else:\n                X_p[i] = 4\n\n        ll = cohen_kappa_score(y, X_p, weights='quadratic')\n        return -ll\n\n    def fit(self, X, y):\n        loss_partial = partial(self._kappa_loss, X=X, y=y)\n        initial_coef = [0.5, 1.5, 2.5, 3.5]\n        self.coef_ = sp.optimize.minimize(loss_partial, initial_coef, method='nelder-mead')\n        print(-loss_partial(self.coef_['x']))\n\n    def predict(self, X, coef):\n        X_p = np.copy(X)\n        for i, pred in enumerate(X_p):\n            if pred < coef[0]:\n                 X_p[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                X_p[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                X_p[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                X_p[i] = 3\n            else:\n                X_p[i] = 4\n        return X_p\n\n    def coefficients(self):\n        return self.coef_['x']","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:36:05.516779Z","iopub.execute_input":"2023-12-17T02:36:05.517111Z","iopub.status.idle":"2023-12-17T02:36:05.527328Z","shell.execute_reply.started":"2023-12-17T02:36:05.517077Z","shell.execute_reply":"2023-12-17T02:36:05.526433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_weights('test204.h5')  \ny_val_pred = model.predict(x_val)\n\noptR = OptimizedRounder()\noptR.fit(y_val_pred, y_val)\ncoefficients = optR.coefficients()\nprint(f'Coefficients: {coefficients}')\ny_val_pred = optR.predict(y_val_pred, coefficients)\n\nscore = cohen_kappa_score(y_val_pred, y_val, weights='quadratic')\n\nprint('Optimized Validation QWK score: {}'.format(score))\nprint('Not Optimized Validation QWK score: {}'.format(max(results.kappa)))","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:36:05.528421Z","iopub.execute_input":"2023-12-17T02:36:05.528744Z","iopub.status.idle":"2023-12-17T02:36:22.482955Z","shell.execute_reply.started":"2023-12-17T02:36:05.528712Z","shell.execute_reply":"2023-12-17T02:36:22.481718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('EfficientNetV2B3_Blanced_test204.h5')","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:36:22.484123Z","iopub.execute_input":"2023-12-17T02:36:22.484407Z","iopub.status.idle":"2023-12-17T02:36:25.474505Z","shell.execute_reply.started":"2023-12-17T02:36:22.484379Z","shell.execute_reply":"2023-12-17T02:36:25.473246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!apt-get update && apt-get install -y python3-seaborn\n!pip install seaborn\n!apt-get update && apt-get install -y python3-statsmodel\n!pip install statsmodel","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:36:25.476195Z","iopub.execute_input":"2023-12-17T02:36:25.476463Z","iopub.status.idle":"2023-12-17T02:38:29.966953Z","shell.execute_reply.started":"2023-12-17T02:36:25.476437Z","shell.execute_reply":"2023-12-17T02:38:29.965422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install statsmodels","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:38:29.968812Z","iopub.execute_input":"2023-12-17T02:38:29.969167Z","iopub.status.idle":"2023-12-17T02:38:38.069104Z","shell.execute_reply.started":"2023-12-17T02:38:29.969132Z","shell.execute_reply":"2023-12-17T02:38:38.067972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, roc_curve, auc\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport statsmodels.api as sm\n\n# Load the model and predict on the validation data\nmodel.load_weights('test204.h5')\ny_val_pred = model.predict(x_val)\n\n# Instantiate an OptimizedRounder object and fit on the validation data\noptR = OptimizedRounder()\noptR.fit(y_val_pred, y_val)\ncoefficients = optR.coefficients()\nprint(f'Coefficients: {coefficients}')\n\n# Get the predicted labels using the optimized coefficients\ny_val_pred_rounded = optR.predict(y_val_pred, coefficients)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:38:38.070590Z","iopub.execute_input":"2023-12-17T02:38:38.070954Z","iopub.status.idle":"2023-12-17T02:38:55.004540Z","shell.execute_reply.started":"2023-12-17T02:38:38.070920Z","shell.execute_reply":"2023-12-17T02:38:55.003565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute confusion matrix\ncm = confusion_matrix(y_val, y_val_pred_rounded)\n\n# Plot confusion matrix\nfig, ax = plt.subplots(figsize=(8, 6))\nsns.heatmap(cm, annot=True, cmap='Blues', fmt='g', ax=ax)\n\n# Add labels, title, and axis ticks\nax.set_xlabel('Predicted labels')\nax.set_ylabel('True labels')\nax.set_title('Confusion Matrix')\nax.xaxis.set_ticklabels(['No-DR', 'Mild NPDR', 'Moderate NPDR', 'Severe NPDR', 'PDR'])\nax.yaxis.set_ticklabels(['No-DR', 'Mild NPDR', 'Moderate NPDR', 'Severe NPDR', 'PDR'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:38:55.005646Z","iopub.execute_input":"2023-12-17T02:38:55.006177Z","iopub.status.idle":"2023-12-17T02:38:55.346378Z","shell.execute_reply.started":"2023-12-17T02:38:55.006145Z","shell.execute_reply":"2023-12-17T02:38:55.345566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:38:55.347287Z","iopub.execute_input":"2023-12-17T02:38:55.347525Z","iopub.status.idle":"2023-12-17T02:38:55.351137Z","shell.execute_reply.started":"2023-12-17T02:38:55.347500Z","shell.execute_reply":"2023-12-17T02:38:55.350466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy = accuracy_score(y_val, y_val_pred_rounded)\n\nprint('Accuracy:', accuracy)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:38:55.351952Z","iopub.execute_input":"2023-12-17T02:38:55.352209Z","iopub.status.idle":"2023-12-17T02:38:55.362969Z","shell.execute_reply.started":"2023-12-17T02:38:55.352182Z","shell.execute_reply":"2023-12-17T02:38:55.362171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report\n\n# assuming y_val and y_val_pred_rounded are the true and predicted labels respectively\ncm = confusion_matrix(y_val, y_val_pred_rounded)\n\n# Calculate precision, recall, and accuracy\nreport = classification_report(y_val, y_val_pred_rounded, target_names=['No-DR', 'Mild NPDR', 'Moderate NPDR', 'Severe NPDR', 'PDR'])\n\n# Print confusion matrix, precision, recall, and accuracy\nprint('Confusion Matrix:\\n', cm)\nprint('Classification Report:\\n', report)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:38:55.363874Z","iopub.execute_input":"2023-12-17T02:38:55.364136Z","iopub.status.idle":"2023-12-17T02:38:55.386878Z","shell.execute_reply.started":"2023-12-17T02:38:55.364112Z","shell.execute_reply":"2023-12-17T02:38:55.386235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Map 0 to 'No DR' and 1,2,3,4 to 'DR' in y_true\ny_true_binary = np.where(y_val == 0, 0, 1)\ny_val_pred_rounded_binary = np.where(y_val_pred_rounded == 0, 0, 1)\n\n# Compute ROC curve and AUC\nfpr, tpr, thresholds = roc_curve(y_true_binary, y_val_pred_rounded_binary, pos_label=1)\nroc_auc = auc(fpr, tpr)\n\nplt.plot(fpr, tpr, color='darkorange', lw=2, label='ROC curve (area = %0.2f)' % roc_auc)\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver operating characteristic curve')\nplt.legend(loc=\"lower right\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:38:55.387700Z","iopub.execute_input":"2023-12-17T02:38:55.387937Z","iopub.status.idle":"2023-12-17T02:38:55.553426Z","shell.execute_reply.started":"2023-12-17T02:38:55.387913Z","shell.execute_reply":"2023-12-17T02:38:55.552605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc = accuracy_score(y_true_binary, y_val_pred_rounded_binary)\n\nprint('Accuracy for binary classification:', acc)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:38:55.554354Z","iopub.execute_input":"2023-12-17T02:38:55.554606Z","iopub.status.idle":"2023-12-17T02:38:55.559732Z","shell.execute_reply.started":"2023-12-17T02:38:55.554580Z","shell.execute_reply":"2023-12-17T02:38:55.558981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute confusion matrix\ncm = confusion_matrix(y_true_binary, y_val_pred_rounded_binary)\n\n# Calculate specificity and sensitivity\ntn, fp, fn, tp = cm.ravel()\nspecificity = tn / (tn + fp)\nsensitivity = tp / (tp + fn)\n\n# Display confusion matrix with percentages\nfig, ax = plt.subplots(figsize=(8, 6))\nsns.heatmap(cm, annot=True, cmap='Blues', fmt='g', ax=ax)\n\n# Add labels, title, and axis ticks\nax.set_xlabel('Predicted labels')\nax.set_ylabel('True labels')\nax.set_title('Confusion Matrix (Specificity={:.2f}, Sensitivity={:.2f})'.format(specificity, sensitivity))\nax.xaxis.set_ticklabels(['No-DR', 'DR'])\nax.yaxis.set_ticklabels(['No-DR', 'DR'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:38:55.560589Z","iopub.execute_input":"2023-12-17T02:38:55.560872Z","iopub.status.idle":"2023-12-17T02:38:55.749462Z","shell.execute_reply.started":"2023-12-17T02:38:55.560843Z","shell.execute_reply":"2023-12-17T02:38:55.748708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}