{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"},{"sourceId":5579826,"sourceType":"datasetVersion","datasetId":3211388},{"sourceId":5579822,"sourceType":"datasetVersion","datasetId":3211385},{"sourceId":418031,"sourceType":"datasetVersion","datasetId":131128},{"sourceId":2812287,"sourceType":"datasetVersion","datasetId":1719146},{"sourceId":2823796,"sourceType":"datasetVersion","datasetId":1727005},{"sourceId":2819730,"sourceType":"datasetVersion","datasetId":1723812}],"dockerImageVersionId":30628,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# To have reproducible results and compare them\nnr_seed = 2023\nimport numpy as np \nnp.random.seed(nr_seed)\nimport tensorflow as tf\ntf.random.set_seed(nr_seed)\nprint(tf.__version__)\n\n# Libraries\nimport json\nimport math\nimport os\n\n\nimport scipy as sp\nfrom functools import partial\nfrom collections import Counter\nimport json\n\nfrom PIL import Image\nimport numpy as np\nfrom keras import backend as K\nfrom keras import layers\nfrom tensorflow.keras.applications import EfficientNetV2B3\nfrom keras.callbacks import Callback, ModelCheckpoint\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential\nfrom keras.optimizers import Adam\nfrom tensorflow.keras.utils import plot_model\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport gc\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-24T10:08:05.294345Z","iopub.execute_input":"2023-12-24T10:08:05.294633Z","iopub.status.idle":"2023-12-24T10:08:21.185394Z","shell.execute_reply.started":"2023-12-24T10:08:05.294582Z","shell.execute_reply":"2023-12-24T10:08:21.184409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install scikit-learn","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:08:21.187018Z","iopub.execute_input":"2023-12-24T10:08:21.187476Z","iopub.status.idle":"2023-12-24T10:08:25.071715Z","shell.execute_reply.started":"2023-12-24T10:08:21.187446Z","shell.execute_reply":"2023-12-24T10:08:25.070456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import cohen_kappa_score","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:08:25.073220Z","iopub.execute_input":"2023-12-24T10:08:25.073516Z","iopub.status.idle":"2023-12-24T10:08:25.532643Z","shell.execute_reply.started":"2023-12-24T10:08:25.073485Z","shell.execute_reply":"2023-12-24T10:08:25.531628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# detect and init the TPU\n#tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\n\n# instantiate a distribution strategy\n#tpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)\n\ntpu = tf.distribute.cluster_resolver.TPUClusterResolver()\ntf.config.experimental_connect_to_cluster(tpu)\ntf.tpu.experimental.initialize_tpu_system(tpu)\nstrategy = tf.distribute.experimental.TPUStrategy(tpu)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:08:25.534690Z","iopub.execute_input":"2023-12-24T10:08:25.535392Z","iopub.status.idle":"2023-12-24T10:08:33.060507Z","shell.execute_reply.started":"2023-12-24T10:08:25.535361Z","shell.execute_reply":"2023-12-24T10:08:33.059759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image size\nim_size = 224\n# Batch size\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:08:33.061581Z","iopub.execute_input":"2023-12-24T10:08:33.061872Z","iopub.status.idle":"2023-12-24T10:08:33.065745Z","shell.execute_reply.started":"2023-12-24T10:08:33.061836Z","shell.execute_reply":"2023-12-24T10:08:33.064959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:08:33.066545Z","iopub.execute_input":"2023-12-24T10:08:33.066788Z","iopub.status.idle":"2023-12-24T10:08:33.081398Z","shell.execute_reply.started":"2023-12-24T10:08:33.066763Z","shell.execute_reply":"2023-12-24T10:08:33.080643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/train-df/train_df_withoutINDEX.csv')\nval_df = pd.read_csv('../input/val-df/val_df_withoutINDEX.csv')\n\n\n\nprint(train_df.shape)\nprint(val_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:08:33.082241Z","iopub.execute_input":"2023-12-24T10:08:33.082487Z","iopub.status.idle":"2023-12-24T10:08:33.201433Z","shell.execute_reply.started":"2023-12-24T10:08:33.082460Z","shell.execute_reply":"2023-12-24T10:08:33.200572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!apt-get update && apt-get install -y python3-opencv\n!pip install opencv-python\n!apt-get update && apt-get install -y python3-tqdm\n!pip install tqdm","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:08:33.202372Z","iopub.execute_input":"2023-12-24T10:08:33.202660Z","iopub.status.idle":"2023-12-24T10:10:09.617967Z","shell.execute_reply.started":"2023-12-24T10:08:33.202630Z","shell.execute_reply":"2023-12-24T10:10:09.616706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport tqdm","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:10:09.619370Z","iopub.execute_input":"2023-12-24T10:10:09.619726Z","iopub.status.idle":"2023-12-24T10:10:10.382846Z","shell.execute_reply.started":"2023-12-24T10:10:09.619691Z","shell.execute_reply":"2023-12-24T10:10:10.382035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_samples(df, columns=4, rows=3):\n    fig=plt.figure(figsize=(5*columns, 4*rows))\n\n    for i in range(columns*rows):\n        image_path = df.loc[i,'id_code']\n        image_id = df.loc[i,'diagnosis']\n        img = cv2.imread(f'{image_path}')\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        #img = crop_image_from_gray(img)\n        img = cv2.resize(img, (im_size,im_size))\n        img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), im_size/40) ,-4 ,128)\n        \n        fig.add_subplot(rows, columns, i+1)\n        plt.title(image_id)\n        plt.imshow(img)\n    \n    plt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:10:10.385803Z","iopub.execute_input":"2023-12-24T10:10:10.386067Z","iopub.status.idle":"2023-12-24T10:10:10.392105Z","shell.execute_reply.started":"2023-12-24T10:10:10.386040Z","shell.execute_reply":"2023-12-24T10:10:10.391415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display train images\ndisplay_samples(train_df)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:10:10.392844Z","iopub.execute_input":"2023-12-24T10:10:10.393067Z","iopub.status.idle":"2023-12-24T10:10:13.504305Z","shell.execute_reply.started":"2023-12-24T10:10:10.393043Z","shell.execute_reply":"2023-12-24T10:10:13.502934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display train images\ndisplay_samples(val_df)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:10:13.505627Z","iopub.execute_input":"2023-12-24T10:10:13.505897Z","iopub.status.idle":"2023-12-24T10:10:18.412529Z","shell.execute_reply.started":"2023-12-24T10:10:13.505870Z","shell.execute_reply":"2023-12-24T10:10:18.408918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = BATCH_SIZE*2","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:10:18.415285Z","iopub.execute_input":"2023-12-24T10:10:18.415575Z","iopub.status.idle":"2023-12-24T10:10:18.419650Z","shell.execute_reply.started":"2023-12-24T10:10:18.415547Z","shell.execute_reply":"2023-12-24T10:10:18.418884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:10:18.420633Z","iopub.execute_input":"2023-12-24T10:10:18.420922Z","iopub.status.idle":"2023-12-24T10:10:18.435819Z","shell.execute_reply.started":"2023-12-24T10:10:18.420893Z","shell.execute_reply":"2023-12-24T10:10:18.435115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:10:18.436748Z","iopub.execute_input":"2023-12-24T10:10:18.436975Z","iopub.status.idle":"2023-12-24T10:10:18.445706Z","shell.execute_reply.started":"2023-12-24T10:10:18.436952Z","shell.execute_reply":"2023-12-24T10:10:18.444840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def crop_image_from_gray(img,tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n            img = np.stack([img1,img2,img3],axis=-1)\n        return img\n\n    \n    \ndef preprocess_image(image_path, desired_size=224):\n    img = cv2.imread(image_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = crop_image_from_gray(img)\n    img = cv2.resize(img, (desired_size,desired_size))\n    img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), desired_size/30) ,-4 ,128)\n    return img\n\n\ndef preprocess_image_old(image_path, desired_size=224):\n    img = cv2.imread(image_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    #img = crop_image_from_gray(img)\n    img = cv2.resize(img, (desired_size,desired_size))\n    img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), desired_size/40) ,-4 ,128)\n    return img","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:10:18.446744Z","iopub.execute_input":"2023-12-24T10:10:18.446987Z","iopub.status.idle":"2023-12-24T10:10:18.458925Z","shell.execute_reply.started":"2023-12-24T10:10:18.446962Z","shell.execute_reply":"2023-12-24T10:10:18.458170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# validation settebook\nN = val_df.shape[0]\nx_val = np.empty((N, im_size, im_size, 3), dtype=np.uint8)\n\nfor i, image_id in tqdm(enumerate(val_df['id_code']), total=N):\n    x_val[i, :, :, :] = preprocess_image(\n        f'{image_id}',\n        desired_size=im_size\n    )","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:10:18.459911Z","iopub.execute_input":"2023-12-24T10:10:18.460222Z","iopub.status.idle":"2023-12-24T10:22:08.652516Z","shell.execute_reply.started":"2023-12-24T10:10:18.460194Z","shell.execute_reply":"2023-12-24T10:22:08.651236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = train_df['diagnosis'].values\ny_val = val_df['diagnosis'].values\n\nprint(y_train.shape)\nprint(x_val.shape)\nprint(y_val.shape)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:22:08.653700Z","iopub.execute_input":"2023-12-24T10:22:08.653975Z","iopub.status.idle":"2023-12-24T10:22:08.659522Z","shell.execute_reply.started":"2023-12-24T10:22:08.653947Z","shell.execute_reply":"2023-12-24T10:22:08.658633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Metrics(Callback):\n    def __init__(self, validation_data=()):\n        super().__init__()\n        self.validation_data = validation_data\n\n    def on_epoch_end(self, epoch, logs=None):\n        x_val, y_val = self.validation_data[0], self.validation_data[1]\n        \n        y_pred = self.model.predict(x_val)\n        \n        coef = [0.5, 1.5, 2.5, 3.5]\n\n        for i, pred in enumerate(y_pred):\n            if pred < coef[0]:\n                y_pred[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                y_pred[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                y_pred[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                y_pred[i] = 3\n            else:\n                y_pred[i] = 4\n\n        _val_kappa = cohen_kappa_score(\n            y_val,\n            y_pred, \n            weights='quadratic')\n        self.val_kappas.append(_val_kappa)\n        print(f\"val_kappa: {_val_kappa:.4f}\")\n\n\n        if _val_kappa == max(self.val_kappas):\n            print(\"Validation Kappa has improved. Saving model.\")\n            self.model.save('testnew204.h5')\n\n        return","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:22:08.660467Z","iopub.execute_input":"2023-12-24T10:22:08.660720Z","iopub.status.idle":"2023-12-24T10:22:08.670935Z","shell.execute_reply.started":"2023-12-24T10:22:08.660695Z","shell.execute_reply":"2023-12-24T10:22:08.670296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_datagen():\n    return ImageDataGenerator(\n        horizontal_flip = True,\n        vertical_flip = True,\n        rotation_range = 160,\n        zoom_range=0.35\n    )","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:22:08.671780Z","iopub.execute_input":"2023-12-24T10:22:08.672029Z","iopub.status.idle":"2023-12-24T10:22:08.682829Z","shell.execute_reply.started":"2023-12-24T10:22:08.672001Z","shell.execute_reply":"2023-12-24T10:22:08.682119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model():\n    effnet = EfficientNetV2B3(\n    input_shape=(im_size,im_size,3),\n    weights='imagenet',\n    include_top=False)\n    model = Sequential()\n    model.add(effnet)\n    model.add(layers.Dropout(0.25))\n    model.add(layers.Dense(2048))\n    model.add(layers.LeakyReLU())\n    model.add(layers.GlobalAveragePooling2D())\n    model.add(layers.Dropout(0.5))\n    model.add(layers.Dense(1, activation='linear'))\n    \n    \n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:22:08.683700Z","iopub.execute_input":"2023-12-24T10:22:08.683946Z","iopub.status.idle":"2023-12-24T10:22:08.693075Z","shell.execute_reply.started":"2023-12-24T10:22:08.683921Z","shell.execute_reply":"2023-12-24T10:22:08.692336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflow","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:22:08.693921Z","iopub.execute_input":"2023-12-24T10:22:08.694148Z","iopub.status.idle":"2023-12-24T10:22:12.010756Z","shell.execute_reply.started":"2023-12-24T10:22:08.694124Z","shell.execute_reply":"2023-12-24T10:22:12.009580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:22:12.012152Z","iopub.execute_input":"2023-12-24T10:22:12.012472Z","iopub.status.idle":"2023-12-24T10:22:12.016763Z","shell.execute_reply.started":"2023-12-24T10:22:12.012437Z","shell.execute_reply":"2023-12-24T10:22:12.015994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    model = build_model()\n    model.compile(\n        loss='mean_squared_error',\n        #optimizer=Adam(lr=0.001,decay=1e-6),\n        optimizer = tf.keras.optimizers.Adam(learning_rate=0.0001),\n        metrics=['mae']\n    )","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:22:12.017664Z","iopub.execute_input":"2023-12-24T10:22:12.017896Z","iopub.status.idle":"2023-12-24T10:22:48.474757Z","shell.execute_reply.started":"2023-12-24T10:22:12.017871Z","shell.execute_reply":"2023-12-24T10:22:48.473684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:22:48.476084Z","iopub.execute_input":"2023-12-24T10:22:48.476396Z","iopub.status.idle":"2023-12-24T10:22:48.529218Z","shell.execute_reply.started":"2023-12-24T10:22:48.476364Z","shell.execute_reply":"2023-12-24T10:22:48.528088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_bucket = 4\ndiv = round(train_df.shape[0]/num_bucket)\ndiv","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:22:48.530170Z","iopub.execute_input":"2023-12-24T10:22:48.530410Z","iopub.status.idle":"2023-12-24T10:22:48.536508Z","shell.execute_reply.started":"2023-12-24T10:22:48.530384Z","shell.execute_reply":"2023-12-24T10:22:48.535359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:22:48.537741Z","iopub.execute_input":"2023-12-24T10:22:48.537985Z","iopub.status.idle":"2023-12-24T10:22:48.545890Z","shell.execute_reply.started":"2023-12-24T10:22:48.537960Z","shell.execute_reply":"2023-12-24T10:22:48.544941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = pd.DataFrame({\n                        'val_loss': [0.0],\n                        'val_mean_absolute_error': [0.0],\n                        'loss': [0.0], \n                        'mean_absolute_error': [0.0],\n                        'bucket': [0.0]\n                        })","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:22:48.550099Z","iopub.execute_input":"2023-12-24T10:22:48.550513Z","iopub.status.idle":"2023-12-24T10:22:48.555570Z","shell.execute_reply.started":"2023-12-24T10:22:48.550487Z","shell.execute_reply":"2023-12-24T10:22:48.554760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Epochs\nepochs = [20,20,20,20]\nkappa_metrics = Metrics(validation_data=(x_val, y_val))\nkappa_metrics.val_kappas = []","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:22:48.556437Z","iopub.execute_input":"2023-12-24T10:22:48.556712Z","iopub.status.idle":"2023-12-24T10:22:48.565163Z","shell.execute_reply.started":"2023-12-24T10:22:48.556684Z","shell.execute_reply":"2023-12-24T10:22:48.564374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(epochs)  ","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:22:48.566083Z","iopub.execute_input":"2023-12-24T10:22:48.566339Z","iopub.status.idle":"2023-12-24T10:22:48.575795Z","shell.execute_reply.started":"2023-12-24T10:22:48.566311Z","shell.execute_reply":"2023-12-24T10:22:48.574906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(0,num_bucket):\n    if i != (num_bucket-1):\n        print(\"Bucket Nr: {}\".format(i))\n        \n        N = train_df.iloc[i*div:(1+i)*div].shape[0]\n        x_train = np.empty((N, im_size, im_size, 3), dtype=np.uint8)\n        for j, image_id in enumerate(tqdm(train_df.iloc[i*div:(1+i)*div,0])):\n            x_train[j, :, :, :] = preprocess_image(f'{image_id}', desired_size = im_size)\n\n        data_generator = create_datagen().flow(x_train, y_train[i*div:(1+i)*div], batch_size=BATCH_SIZE)\n        history = model.fit_generator(\n                        data_generator,\n                        steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                        epochs=epochs[i],\n                        validation_data=(x_val, y_val),\n                        callbacks=[kappa_metrics]\n                        )\n        \n        dic = history.history\n        df_model = pd.DataFrame(dic)\n        df_model['bucket'] = i\n    else:\n        print(\"Bucket Nr: {}\".format(i))\n        N = train_df.iloc[i*div:].shape[0]\n        x_train = np.empty((N, im_size, im_size, 3), dtype=np.uint8)\n        for j, image_id in enumerate(tqdm(train_df.iloc[i*div:,0])):\n            x_train[j, :, :, :] = preprocess_image(f'{image_id}', desired_size = im_size)\n            \n        data_generator = create_datagen().flow(x_train, y_train[i*div:], batch_size=BATCH_SIZE)\n        history = model.fit_generator(\n                        data_generator,\n                        steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                        epochs=epochs[i],\n                        validation_data=(x_val, y_val),\n                        callbacks=[kappa_metrics]\n                        )\n        \n        dic = history.history\n        df_model = pd.DataFrame(dic)\n        df_model['bucket'] = i\n\n    results = pd.concat([results, df_model], ignore_index=True)\n    del data_generator\n    del x_train\n    gc.collect()\n    print('-'*40)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T10:22:48.576749Z","iopub.execute_input":"2023-12-24T10:22:48.577013Z","iopub.status.idle":"2023-12-24T13:18:05.087243Z","shell.execute_reply.started":"2023-12-24T10:22:48.576986Z","shell.execute_reply":"2023-12-24T13:18:05.085813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = results.iloc[1:]\nkappa_vals = kappa_metrics.val_kappas[:len(results)]\nresults['kappa'] = kappa_vals\nresults = results.reset_index(drop=True)\nresults = results.rename(index=str, columns={\"index\": \"epoch\"})\nprint(max(results.kappa))","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:18:05.110593Z","iopub.execute_input":"2023-12-24T13:18:05.110939Z","iopub.status.idle":"2023-12-24T13:18:05.120790Z","shell.execute_reply.started":"2023-12-24T13:18:05.110909Z","shell.execute_reply":"2023-12-24T13:18:05.119842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results[['loss', 'val_loss']].plot()\nresults[['mae', 'val_mae']].plot()\nresults[['kappa']].plot()\nresults.to_csv('model_results.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:18:05.121752Z","iopub.execute_input":"2023-12-24T13:18:05.121992Z","iopub.status.idle":"2023-12-24T13:18:05.731968Z","shell.execute_reply.started":"2023-12-24T13:18:05.121966Z","shell.execute_reply":"2023-12-24T13:18:05.730973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class OptimizedRounder(object):\n    def __init__(self):\n        self.coef_ = 0\n\n    def _kappa_loss(self, coef, X, y):\n        X_p = np.copy(X)\n        for i, pred in enumerate(X_p):\n            if pred < coef[0]:\n                X_p[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                X_p[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                X_p[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                X_p[i] = 3\n            else:\n                X_p[i] = 4\n\n        ll = cohen_kappa_score(y, X_p, weights='quadratic')\n        return -ll\n\n    def fit(self, X, y):\n        loss_partial = partial(self._kappa_loss, X=X, y=y)\n        initial_coef = [0.5, 1.5, 2.5, 3.5]\n        self.coef_ = sp.optimize.minimize(loss_partial, initial_coef, method='nelder-mead')\n        print(-loss_partial(self.coef_['x']))\n\n    def predict(self, X, coef):\n        X_p = np.copy(X)\n        for i, pred in enumerate(X_p):\n            if pred < coef[0]:\n                 X_p[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                X_p[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                X_p[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                X_p[i] = 3\n            else:\n                X_p[i] = 4\n        return X_p\n\n    def coefficients(self):\n        return self.coef_['x']","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:18:05.733128Z","iopub.execute_input":"2023-12-24T13:18:05.733387Z","iopub.status.idle":"2023-12-24T13:18:05.743242Z","shell.execute_reply.started":"2023-12-24T13:18:05.733360Z","shell.execute_reply":"2023-12-24T13:18:05.742441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_weights('testnew204.h5')  \ny_val_pred = model.predict(x_val)\n\noptR = OptimizedRounder()\noptR.fit(y_val_pred, y_val)\ncoefficients = optR.coefficients()\nprint(f'Coefficients: {coefficients}')\ny_val_pred = optR.predict(y_val_pred, coefficients)\n\nscore = cohen_kappa_score(y_val_pred, y_val, weights='quadratic')\n\nprint('Optimized Validation QWK score: {}'.format(score))\nprint('Not Optimized Validation QWK score: {}'.format(max(results.kappa)))","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:18:05.744191Z","iopub.execute_input":"2023-12-24T13:18:05.744426Z","iopub.status.idle":"2023-12-24T13:18:21.896180Z","shell.execute_reply.started":"2023-12-24T13:18:05.744401Z","shell.execute_reply":"2023-12-24T13:18:21.894952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('EfficientNetV2B3_Blanced_testnew204.h5')","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:18:21.897287Z","iopub.execute_input":"2023-12-24T13:18:21.897557Z","iopub.status.idle":"2023-12-24T13:18:24.890490Z","shell.execute_reply.started":"2023-12-24T13:18:21.897529Z","shell.execute_reply":"2023-12-24T13:18:24.889392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!apt-get update && apt-get install -y python3-seaborn\n!pip install seaborn\n!apt-get update && apt-get install -y python3-statsmodel\n!pip install statsmodel","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:18:24.892504Z","iopub.execute_input":"2023-12-24T13:18:24.892793Z","iopub.status.idle":"2023-12-24T13:19:56.434696Z","shell.execute_reply.started":"2023-12-24T13:18:24.892764Z","shell.execute_reply":"2023-12-24T13:19:56.433334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install statsmodels","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:19:56.436150Z","iopub.execute_input":"2023-12-24T13:19:56.436433Z","iopub.status.idle":"2023-12-24T13:20:05.061137Z","shell.execute_reply.started":"2023-12-24T13:19:56.436403Z","shell.execute_reply":"2023-12-24T13:20:05.059967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, roc_curve, auc\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport statsmodels.api as sm\n\n# Load the model and predict on the validation data\nmodel.load_weights('testnew204.h5')\ny_val_pred = model.predict(x_val)\n\n# Instantiate an OptimizedRounder object and fit on the validation data\noptR = OptimizedRounder()\noptR.fit(y_val_pred, y_val)\ncoefficients = optR.coefficients()\nprint(f'Coefficients: {coefficients}')\n\n# Get the predicted labels using the optimized coefficients\ny_val_pred_rounded = optR.predict(y_val_pred, coefficients)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:20:05.062547Z","iopub.execute_input":"2023-12-24T13:20:05.062861Z","iopub.status.idle":"2023-12-24T13:20:20.879221Z","shell.execute_reply.started":"2023-12-24T13:20:05.062828Z","shell.execute_reply":"2023-12-24T13:20:20.878188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute confusion matrix\ncm = confusion_matrix(y_val, y_val_pred_rounded)\n\n# Plot confusion matrix\nfig, ax = plt.subplots(figsize=(8, 6))\nsns.heatmap(cm, annot=True, cmap='Blues', fmt='g', ax=ax)\n\n# Add labels, title, and axis ticks\nax.set_xlabel('Predicted labels')\nax.set_ylabel('True labels')\nax.set_title('Confusion Matrix')\nax.xaxis.set_ticklabels(['No-DR', 'Mild NPDR', 'Moderate NPDR', 'Severe NPDR', 'PDR'])\nax.yaxis.set_ticklabels(['No-DR', 'Mild NPDR', 'Moderate NPDR', 'Severe NPDR', 'PDR'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:20:20.880395Z","iopub.execute_input":"2023-12-24T13:20:20.881023Z","iopub.status.idle":"2023-12-24T13:20:21.218910Z","shell.execute_reply.started":"2023-12-24T13:20:20.880993Z","shell.execute_reply":"2023-12-24T13:20:21.218000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:20:21.219958Z","iopub.execute_input":"2023-12-24T13:20:21.220249Z","iopub.status.idle":"2023-12-24T13:20:21.223811Z","shell.execute_reply.started":"2023-12-24T13:20:21.220220Z","shell.execute_reply":"2023-12-24T13:20:21.223086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy = accuracy_score(y_val, y_val_pred_rounded)\n\nprint('Accuracy:', accuracy)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:20:21.224690Z","iopub.execute_input":"2023-12-24T13:20:21.224991Z","iopub.status.idle":"2023-12-24T13:20:21.236058Z","shell.execute_reply.started":"2023-12-24T13:20:21.224958Z","shell.execute_reply":"2023-12-24T13:20:21.235322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report\n\n# assuming y_val and y_val_pred_rounded are the true and predicted labels respectively\ncm = confusion_matrix(y_val, y_val_pred_rounded)\n\n# Calculate precision, recall, and accuracy\nreport = classification_report(y_val, y_val_pred_rounded, target_names=['No-DR', 'Mild NPDR', 'Moderate NPDR', 'Severe NPDR', 'PDR'])\n\n# Print confusion matrix, precision, recall, and accuracy\nprint('Confusion Matrix:\\n', cm)\nprint('Classification Report:\\n', report)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:20:21.236977Z","iopub.execute_input":"2023-12-24T13:20:21.237248Z","iopub.status.idle":"2023-12-24T13:20:21.261121Z","shell.execute_reply.started":"2023-12-24T13:20:21.237220Z","shell.execute_reply":"2023-12-24T13:20:21.260357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Map 0 to 'No DR' and 1,2,3,4 to 'DR' in y_true\ny_true_binary = np.where(y_val == 0, 0, 1)\ny_val_pred_rounded_binary = np.where(y_val_pred_rounded == 0, 0, 1)\n\n# Compute ROC curve and AUC\nfpr, tpr, thresholds = roc_curve(y_true_binary, y_val_pred_rounded_binary, pos_label=1)\nroc_auc = auc(fpr, tpr)\n\nplt.plot(fpr, tpr, color='darkorange', lw=2, label='ROC curve (area = %0.2f)' % roc_auc)\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver operating characteristic curve')\nplt.legend(loc=\"lower right\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:20:21.262019Z","iopub.execute_input":"2023-12-24T13:20:21.262268Z","iopub.status.idle":"2023-12-24T13:20:21.413411Z","shell.execute_reply.started":"2023-12-24T13:20:21.262243Z","shell.execute_reply":"2023-12-24T13:20:21.412648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc = accuracy_score(y_true_binary, y_val_pred_rounded_binary)\n\nprint('Accuracy for binary classification:', acc)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:20:21.414309Z","iopub.execute_input":"2023-12-24T13:20:21.414559Z","iopub.status.idle":"2023-12-24T13:20:21.419600Z","shell.execute_reply.started":"2023-12-24T13:20:21.414532Z","shell.execute_reply":"2023-12-24T13:20:21.418801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute confusion matrix\ncm = confusion_matrix(y_true_binary, y_val_pred_rounded_binary)\n\n# Calculate specificity and sensitivity\ntn, fp, fn, tp = cm.ravel()\nspecificity = tn / (tn + fp)\nsensitivity = tp / (tp + fn)\n\n# Display confusion matrix with percentages\nfig, ax = plt.subplots(figsize=(8, 6))\nsns.heatmap(cm, annot=True, cmap='Blues', fmt='g', ax=ax)\n\n# Add labels, title, and axis ticks\nax.set_xlabel('Predicted labels')\nax.set_ylabel('True labels')\nax.set_title('Confusion Matrix (Specificity={:.2f}, Sensitivity={:.2f})'.format(specificity, sensitivity))\nax.xaxis.set_ticklabels(['No-DR', 'DR'])\nax.yaxis.set_ticklabels(['No-DR', 'DR'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-24T13:20:21.420635Z","iopub.execute_input":"2023-12-24T13:20:21.420935Z","iopub.status.idle":"2023-12-24T13:20:21.601428Z","shell.execute_reply.started":"2023-12-24T13:20:21.420907Z","shell.execute_reply":"2023-12-24T13:20:21.600548Z"},"trusted":true},"execution_count":null,"outputs":[]}]}