{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"},{"sourceId":418031,"sourceType":"datasetVersion","datasetId":131128},{"sourceId":2812287,"sourceType":"datasetVersion","datasetId":1719146},{"sourceId":2819730,"sourceType":"datasetVersion","datasetId":1723812},{"sourceId":2823796,"sourceType":"datasetVersion","datasetId":1727005},{"sourceId":5579822,"sourceType":"datasetVersion","datasetId":3211385},{"sourceId":5579826,"sourceType":"datasetVersion","datasetId":3211388}],"dockerImageVersionId":30628,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# To have reproducible results and compare them\nnr_seed = 2023\nimport numpy as np \nnp.random.seed(nr_seed)\nimport tensorflow as tf\ntf.random.set_seed(nr_seed)\nprint(tf.__version__)\n\n# Libraries\nimport json\nimport math\nimport os\n\n\nimport scipy as sp\nfrom functools import partial\nfrom collections import Counter\nimport json\n\nfrom PIL import Image\nimport numpy as np\nfrom keras import backend as K\nfrom keras import layers\nfrom tensorflow.keras.applications import EfficientNetV2B3\nfrom keras.callbacks import Callback, ModelCheckpoint\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential\nfrom keras.optimizers import Adam\nfrom tensorflow.keras.utils import plot_model\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport gc\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-24T05:51:42.431846Z","iopub.execute_input":"2023-12-24T05:51:42.432130Z","iopub.status.idle":"2023-12-24T05:51:56.821539Z","shell.execute_reply.started":"2023-12-24T05:51:42.432101Z","shell.execute_reply":"2023-12-24T05:51:56.820839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install scikit-learn","metadata":{"execution":{"iopub.status.busy":"2023-12-24T05:51:56.822855Z","iopub.execute_input":"2023-12-24T05:51:56.823296Z","iopub.status.idle":"2023-12-24T05:52:00.594173Z","shell.execute_reply.started":"2023-12-24T05:51:56.823266Z","shell.execute_reply":"2023-12-24T05:52:00.593088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import cohen_kappa_score","metadata":{"execution":{"iopub.status.busy":"2023-12-24T05:52:00.595360Z","iopub.execute_input":"2023-12-24T05:52:00.595626Z","iopub.status.idle":"2023-12-24T05:52:01.047142Z","shell.execute_reply.started":"2023-12-24T05:52:00.595599Z","shell.execute_reply":"2023-12-24T05:52:01.046318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# detect and init the TPU\n#tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\n\n# instantiate a distribution strategy\n#tpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)\n\ntpu = tf.distribute.cluster_resolver.TPUClusterResolver()\ntf.config.experimental_connect_to_cluster(tpu)\ntf.tpu.experimental.initialize_tpu_system(tpu)\nstrategy = tf.distribute.experimental.TPUStrategy(tpu)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T05:52:01.048959Z","iopub.execute_input":"2023-12-24T05:52:01.049635Z","iopub.status.idle":"2023-12-24T05:52:08.494040Z","shell.execute_reply.started":"2023-12-24T05:52:01.049604Z","shell.execute_reply":"2023-12-24T05:52:08.493321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image size\nim_size = 224\n# Batch size\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync","metadata":{"execution":{"iopub.status.busy":"2023-12-24T05:52:08.494981Z","iopub.execute_input":"2023-12-24T05:52:08.495243Z","iopub.status.idle":"2023-12-24T05:52:08.498636Z","shell.execute_reply.started":"2023-12-24T05:52:08.495216Z","shell.execute_reply":"2023-12-24T05:52:08.497886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2023-12-24T05:52:08.499534Z","iopub.execute_input":"2023-12-24T05:52:08.499783Z","iopub.status.idle":"2023-12-24T05:52:08.514380Z","shell.execute_reply.started":"2023-12-24T05:52:08.499759Z","shell.execute_reply":"2023-12-24T05:52:08.513787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/train-df/train_df_withoutINDEX.csv')\nval_df = pd.read_csv('../input/val-df/val_df_withoutINDEX.csv')\n\n\n\nprint(train_df.shape)\nprint(val_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T05:52:08.515171Z","iopub.execute_input":"2023-12-24T05:52:08.515393Z","iopub.status.idle":"2023-12-24T05:52:08.623042Z","shell.execute_reply.started":"2023-12-24T05:52:08.515369Z","shell.execute_reply":"2023-12-24T05:52:08.622230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!apt-get update && apt-get install -y python3-opencv\n!pip install opencv-python\n!apt-get update && apt-get install -y python3-tqdm\n!pip install tqdm","metadata":{"execution":{"iopub.status.busy":"2023-12-24T05:52:08.624033Z","iopub.execute_input":"2023-12-24T05:52:08.624284Z","iopub.status.idle":"2023-12-24T05:53:36.613967Z","shell.execute_reply.started":"2023-12-24T05:52:08.624260Z","shell.execute_reply":"2023-12-24T05:53:36.612647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport tqdm","metadata":{"execution":{"iopub.status.busy":"2023-12-24T05:53:36.615721Z","iopub.execute_input":"2023-12-24T05:53:36.616101Z","iopub.status.idle":"2023-12-24T05:53:37.377947Z","shell.execute_reply.started":"2023-12-24T05:53:36.616062Z","shell.execute_reply":"2023-12-24T05:53:37.377077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_samples(df, columns=4, rows=3):\n    fig=plt.figure(figsize=(5*columns, 4*rows))\n\n    for i in range(columns*rows):\n        image_path = df.loc[i,'id_code']\n        image_id = df.loc[i,'diagnosis']\n        img = cv2.imread(f'{image_path}')\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        #img = crop_image_from_gray(img)\n        img = cv2.resize(img, (im_size,im_size))\n        img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), im_size/40) ,-4 ,128)\n        \n        fig.add_subplot(rows, columns, i+1)\n        plt.title(image_id)\n        plt.imshow(img)\n    \n    plt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-12-24T05:53:37.380980Z","iopub.execute_input":"2023-12-24T05:53:37.381267Z","iopub.status.idle":"2023-12-24T05:53:37.387738Z","shell.execute_reply.started":"2023-12-24T05:53:37.381240Z","shell.execute_reply":"2023-12-24T05:53:37.387062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display train images\ndisplay_samples(train_df)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T05:53:37.388586Z","iopub.execute_input":"2023-12-24T05:53:37.388811Z","iopub.status.idle":"2023-12-24T05:53:40.605305Z","shell.execute_reply.started":"2023-12-24T05:53:37.388786Z","shell.execute_reply":"2023-12-24T05:53:40.604181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display train images\ndisplay_samples(val_df)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T05:53:40.607626Z","iopub.execute_input":"2023-12-24T05:53:40.607921Z","iopub.status.idle":"2023-12-24T05:53:45.422116Z","shell.execute_reply.started":"2023-12-24T05:53:40.607883Z","shell.execute_reply":"2023-12-24T05:53:45.420906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = BATCH_SIZE*2","metadata":{"execution":{"iopub.status.busy":"2023-12-24T05:53:45.423373Z","iopub.execute_input":"2023-12-24T05:53:45.423664Z","iopub.status.idle":"2023-12-24T05:53:45.427966Z","shell.execute_reply.started":"2023-12-24T05:53:45.423633Z","shell.execute_reply":"2023-12-24T05:53:45.426989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2023-12-24T05:53:45.428956Z","iopub.execute_input":"2023-12-24T05:53:45.429204Z","iopub.status.idle":"2023-12-24T05:53:45.439276Z","shell.execute_reply.started":"2023-12-24T05:53:45.429180Z","shell.execute_reply":"2023-12-24T05:53:45.438479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-12-24T05:53:45.440334Z","iopub.execute_input":"2023-12-24T05:53:45.440802Z","iopub.status.idle":"2023-12-24T05:53:45.446563Z","shell.execute_reply.started":"2023-12-24T05:53:45.440758Z","shell.execute_reply":"2023-12-24T05:53:45.445772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def crop_image_from_gray(img,tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n            img = np.stack([img1,img2,img3],axis=-1)\n        return img\n\n    \n    \ndef preprocess_image(image_path, desired_size=224):\n    img = cv2.imread(image_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = crop_image_from_gray(img)\n    img = cv2.resize(img, (desired_size,desired_size))\n    img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), desired_size/30) ,-4 ,128)\n    return img\n\n\ndef preprocess_image_old(image_path, desired_size=224):\n    img = cv2.imread(image_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    #img = crop_image_from_gray(img)\n    img = cv2.resize(img, (desired_size,desired_size))\n    img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), desired_size/40) ,-4 ,128)\n    return img","metadata":{"execution":{"iopub.status.busy":"2023-12-24T05:53:45.447570Z","iopub.execute_input":"2023-12-24T05:53:45.447811Z","iopub.status.idle":"2023-12-24T05:53:45.456982Z","shell.execute_reply.started":"2023-12-24T05:53:45.447786Z","shell.execute_reply":"2023-12-24T05:53:45.456166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# validation settebook\nN = val_df.shape[0]\nx_val = np.empty((N, im_size, im_size, 3), dtype=np.uint8)\n\nfor i, image_id in tqdm(enumerate(val_df['id_code']), total=N):\n    x_val[i, :, :, :] = preprocess_image(\n        f'{image_id}',\n        desired_size=im_size\n    )","metadata":{"execution":{"iopub.status.busy":"2023-12-24T05:53:45.457977Z","iopub.execute_input":"2023-12-24T05:53:45.458305Z","iopub.status.idle":"2023-12-24T06:05:02.484788Z","shell.execute_reply.started":"2023-12-24T05:53:45.458279Z","shell.execute_reply":"2023-12-24T06:05:02.483740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = train_df['diagnosis'].values\ny_val = val_df['diagnosis'].values\n\nprint(y_train.shape)\nprint(x_val.shape)\nprint(y_val.shape)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T06:05:02.485996Z","iopub.execute_input":"2023-12-24T06:05:02.486307Z","iopub.status.idle":"2023-12-24T06:05:02.491603Z","shell.execute_reply.started":"2023-12-24T06:05:02.486276Z","shell.execute_reply":"2023-12-24T06:05:02.490852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Metrics(Callback):\n    def __init__(self, validation_data=()):\n        super().__init__()\n        self.validation_data = validation_data\n\n    def on_epoch_end(self, epoch, logs=None):\n        x_val, y_val = self.validation_data[0], self.validation_data[1]\n        \n        y_pred = self.model.predict(x_val)\n        \n        coef = [0.5, 1.5, 2.5, 3.5]\n\n        for i, pred in enumerate(y_pred):\n            if pred < coef[0]:\n                y_pred[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                y_pred[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                y_pred[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                y_pred[i] = 3\n            else:\n                y_pred[i] = 4\n\n        _val_kappa = cohen_kappa_score(\n            y_val,\n            y_pred, \n            weights='quadratic')\n        self.val_kappas.append(_val_kappa)\n        print(f\"val_kappa: {_val_kappa:.4f}\")\n\n\n        if _val_kappa == max(self.val_kappas):\n            print(\"Validation Kappa has improved. Saving model.\")\n            self.model.save('testnew44.h5')\n\n        return","metadata":{"execution":{"iopub.status.busy":"2023-12-24T06:05:02.492484Z","iopub.execute_input":"2023-12-24T06:05:02.492722Z","iopub.status.idle":"2023-12-24T06:05:02.503448Z","shell.execute_reply.started":"2023-12-24T06:05:02.492697Z","shell.execute_reply":"2023-12-24T06:05:02.502725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_datagen():\n    return ImageDataGenerator(\n        horizontal_flip = True,\n        vertical_flip = True,\n        rotation_range = 160,\n        zoom_range=0.35\n    )","metadata":{"execution":{"iopub.status.busy":"2023-12-24T06:05:02.504377Z","iopub.execute_input":"2023-12-24T06:05:02.504637Z","iopub.status.idle":"2023-12-24T06:05:02.515524Z","shell.execute_reply.started":"2023-12-24T06:05:02.504609Z","shell.execute_reply":"2023-12-24T06:05:02.514882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model():\n    effnet = EfficientNetV2B3(\n    input_shape=(im_size,im_size,3),\n    weights='imagenet',\n    include_top=False)\n    model = Sequential()\n    model.add(effnet)\n    model.add(layers.Dropout(0.25))\n    model.add(layers.Dense(2048))\n    model.add(layers.LeakyReLU())\n    model.add(layers.GlobalAveragePooling2D())\n    model.add(layers.Dropout(0.5))\n    model.add(layers.Dense(1, activation='linear'))\n    \n    \n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-12-24T06:05:02.516381Z","iopub.execute_input":"2023-12-24T06:05:02.516616Z","iopub.status.idle":"2023-12-24T06:05:02.524743Z","shell.execute_reply.started":"2023-12-24T06:05:02.516591Z","shell.execute_reply":"2023-12-24T06:05:02.524066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflow","metadata":{"execution":{"iopub.status.busy":"2023-12-24T06:05:02.525583Z","iopub.execute_input":"2023-12-24T06:05:02.525805Z","iopub.status.idle":"2023-12-24T06:05:05.872080Z","shell.execute_reply.started":"2023-12-24T06:05:02.525781Z","shell.execute_reply":"2023-12-24T06:05:05.870618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2023-12-24T06:05:05.873725Z","iopub.execute_input":"2023-12-24T06:05:05.874084Z","iopub.status.idle":"2023-12-24T06:05:05.878784Z","shell.execute_reply.started":"2023-12-24T06:05:05.874048Z","shell.execute_reply":"2023-12-24T06:05:05.878051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    model = build_model()\n    model.compile(\n        loss='mean_squared_error',\n        #optimizer=Adam(lr=0.001,decay=1e-6),\n        optimizer = tf.keras.optimizers.Adam(learning_rate=0.0001),\n        metrics=['mae']\n    )","metadata":{"execution":{"iopub.status.busy":"2023-12-24T06:05:05.879668Z","iopub.execute_input":"2023-12-24T06:05:05.879944Z","iopub.status.idle":"2023-12-24T06:05:41.630680Z","shell.execute_reply.started":"2023-12-24T06:05:05.879917Z","shell.execute_reply":"2023-12-24T06:05:41.629537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-12-24T06:05:41.631844Z","iopub.execute_input":"2023-12-24T06:05:41.632120Z","iopub.status.idle":"2023-12-24T06:05:41.676647Z","shell.execute_reply.started":"2023-12-24T06:05:41.632091Z","shell.execute_reply":"2023-12-24T06:05:41.675784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_bucket = 4\ndiv = round(train_df.shape[0]/num_bucket)\ndiv","metadata":{"execution":{"iopub.status.busy":"2023-12-24T06:05:41.677766Z","iopub.execute_input":"2023-12-24T06:05:41.678055Z","iopub.status.idle":"2023-12-24T06:05:41.683198Z","shell.execute_reply.started":"2023-12-24T06:05:41.678010Z","shell.execute_reply":"2023-12-24T06:05:41.682453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2023-12-24T06:05:41.684256Z","iopub.execute_input":"2023-12-24T06:05:41.684484Z","iopub.status.idle":"2023-12-24T06:05:41.692493Z","shell.execute_reply.started":"2023-12-24T06:05:41.684460Z","shell.execute_reply":"2023-12-24T06:05:41.691764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = pd.DataFrame({\n                        'val_loss': [0.0],\n                        'val_mean_absolute_error': [0.0],\n                        'loss': [0.0], \n                        'mean_absolute_error': [0.0],\n                        'bucket': [0.0]\n                        })","metadata":{"execution":{"iopub.status.busy":"2023-12-24T06:05:41.696250Z","iopub.execute_input":"2023-12-24T06:05:41.696534Z","iopub.status.idle":"2023-12-24T06:05:41.701531Z","shell.execute_reply.started":"2023-12-24T06:05:41.696506Z","shell.execute_reply":"2023-12-24T06:05:41.700778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Epochs\nepochs = [4,4,4,4]\nkappa_metrics = Metrics(validation_data=(x_val, y_val))\nkappa_metrics.val_kappas = []","metadata":{"execution":{"iopub.status.busy":"2023-12-24T06:05:41.702459Z","iopub.execute_input":"2023-12-24T06:05:41.702780Z","iopub.status.idle":"2023-12-24T06:05:41.710235Z","shell.execute_reply.started":"2023-12-24T06:05:41.702752Z","shell.execute_reply":"2023-12-24T06:05:41.709468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(epochs)  ","metadata":{"execution":{"iopub.status.busy":"2023-12-24T06:05:41.711083Z","iopub.execute_input":"2023-12-24T06:05:41.711302Z","iopub.status.idle":"2023-12-24T06:05:41.720102Z","shell.execute_reply.started":"2023-12-24T06:05:41.711279Z","shell.execute_reply":"2023-12-24T06:05:41.719400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(0,num_bucket):\n    if i != (num_bucket-1):\n        print(\"Bucket Nr: {}\".format(i))\n        \n        N = train_df.iloc[i*div:(1+i)*div].shape[0]\n        x_train = np.empty((N, im_size, im_size, 3), dtype=np.uint8)\n        for j, image_id in enumerate(tqdm(train_df.iloc[i*div:(1+i)*div,0])):\n            x_train[j, :, :, :] = preprocess_image(f'{image_id}', desired_size = im_size)\n\n        data_generator = create_datagen().flow(x_train, y_train[i*div:(1+i)*div], batch_size=BATCH_SIZE)\n        history = model.fit_generator(\n                        data_generator,\n                        steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                        epochs=epochs[i],\n                        validation_data=(x_val, y_val),\n                        callbacks=[kappa_metrics]\n                        )\n        \n        dic = history.history\n        df_model = pd.DataFrame(dic)\n        df_model['bucket'] = i\n    else:\n        print(\"Bucket Nr: {}\".format(i))\n        N = train_df.iloc[i*div:].shape[0]\n        x_train = np.empty((N, im_size, im_size, 3), dtype=np.uint8)\n        for j, image_id in enumerate(tqdm(train_df.iloc[i*div:,0])):\n            x_train[j, :, :, :] = preprocess_image(f'{image_id}', desired_size = im_size)\n            \n        data_generator = create_datagen().flow(x_train, y_train[i*div:], batch_size=BATCH_SIZE)\n        history = model.fit_generator(\n                        data_generator,\n                        steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                        epochs=epochs[i],\n                        validation_data=(x_val, y_val),\n                        callbacks=[kappa_metrics]\n                        )\n        \n        dic = history.history\n        df_model = pd.DataFrame(dic)\n        df_model['bucket'] = i\n\n    results = pd.concat([results, df_model], ignore_index=True)\n    del data_generator\n    del x_train\n    gc.collect()\n    print('-'*40)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T06:05:41.721008Z","iopub.execute_input":"2023-12-24T06:05:41.721251Z","iopub.status.idle":"2023-12-24T07:07:14.463035Z","shell.execute_reply.started":"2023-12-24T06:05:41.721227Z","shell.execute_reply":"2023-12-24T07:07:14.461991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = results.iloc[1:]\nkappa_vals = kappa_metrics.val_kappas[:len(results)]\nresults['kappa'] = kappa_vals\nresults = results.reset_index(drop=True)\nresults = results.rename(index=str, columns={\"index\": \"epoch\"})\nprint(max(results.kappa))","metadata":{"execution":{"iopub.status.busy":"2023-12-24T07:07:14.464886Z","iopub.execute_input":"2023-12-24T07:07:14.465180Z","iopub.status.idle":"2023-12-24T07:07:14.472620Z","shell.execute_reply.started":"2023-12-24T07:07:14.465152Z","shell.execute_reply":"2023-12-24T07:07:14.471769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results[['loss', 'val_loss']].plot()\nresults[['mae', 'val_mae']].plot()\nresults[['kappa']].plot()\nresults.to_csv('model_results.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T07:07:14.473634Z","iopub.execute_input":"2023-12-24T07:07:14.473910Z","iopub.status.idle":"2023-12-24T07:07:15.061682Z","shell.execute_reply.started":"2023-12-24T07:07:14.473884Z","shell.execute_reply":"2023-12-24T07:07:15.060846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class OptimizedRounder(object):\n    def __init__(self):\n        self.coef_ = 0\n\n    def _kappa_loss(self, coef, X, y):\n        X_p = np.copy(X)\n        for i, pred in enumerate(X_p):\n            if pred < coef[0]:\n                X_p[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                X_p[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                X_p[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                X_p[i] = 3\n            else:\n                X_p[i] = 4\n\n        ll = cohen_kappa_score(y, X_p, weights='quadratic')\n        return -ll\n\n    def fit(self, X, y):\n        loss_partial = partial(self._kappa_loss, X=X, y=y)\n        initial_coef = [0.5, 1.5, 2.5, 3.5]\n        self.coef_ = sp.optimize.minimize(loss_partial, initial_coef, method='nelder-mead')\n        print(-loss_partial(self.coef_['x']))\n\n    def predict(self, X, coef):\n        X_p = np.copy(X)\n        for i, pred in enumerate(X_p):\n            if pred < coef[0]:\n                 X_p[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                X_p[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                X_p[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                X_p[i] = 3\n            else:\n                X_p[i] = 4\n        return X_p\n\n    def coefficients(self):\n        return self.coef_['x']","metadata":{"execution":{"iopub.status.busy":"2023-12-24T07:07:15.062839Z","iopub.execute_input":"2023-12-24T07:07:15.063165Z","iopub.status.idle":"2023-12-24T07:07:15.072406Z","shell.execute_reply.started":"2023-12-24T07:07:15.063133Z","shell.execute_reply":"2023-12-24T07:07:15.071537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_weights('testnew44.h5')  \ny_val_pred = model.predict(x_val)\n\noptR = OptimizedRounder()\noptR.fit(y_val_pred, y_val)\ncoefficients = optR.coefficients()\nprint(f'Coefficients: {coefficients}')\ny_val_pred = optR.predict(y_val_pred, coefficients)\n\nscore = cohen_kappa_score(y_val_pred, y_val, weights='quadratic')\n\nprint('Optimized Validation QWK score: {}'.format(score))\nprint('Not Optimized Validation QWK score: {}'.format(max(results.kappa)))","metadata":{"execution":{"iopub.status.busy":"2023-12-24T07:07:15.073437Z","iopub.execute_input":"2023-12-24T07:07:15.073738Z","iopub.status.idle":"2023-12-24T07:07:31.352951Z","shell.execute_reply.started":"2023-12-24T07:07:15.073707Z","shell.execute_reply":"2023-12-24T07:07:31.351676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('EfficientNetV2B3_Blanced_testnew44.h5')","metadata":{"execution":{"iopub.status.busy":"2023-12-24T07:07:31.354333Z","iopub.execute_input":"2023-12-24T07:07:31.354679Z","iopub.status.idle":"2023-12-24T07:07:34.354268Z","shell.execute_reply.started":"2023-12-24T07:07:31.354643Z","shell.execute_reply":"2023-12-24T07:07:34.352981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!apt-get update && apt-get install -y python3-seaborn\n!pip install seaborn\n!apt-get update && apt-get install -y python3-statsmodel\n!pip install statsmodel","metadata":{"execution":{"iopub.status.busy":"2023-12-24T07:07:34.356287Z","iopub.execute_input":"2023-12-24T07:07:34.356574Z","iopub.status.idle":"2023-12-24T07:09:03.228932Z","shell.execute_reply.started":"2023-12-24T07:07:34.356546Z","shell.execute_reply":"2023-12-24T07:09:03.227631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install statsmodels","metadata":{"execution":{"iopub.status.busy":"2023-12-24T07:09:03.230428Z","iopub.execute_input":"2023-12-24T07:09:03.230737Z","iopub.status.idle":"2023-12-24T07:09:11.257660Z","shell.execute_reply.started":"2023-12-24T07:09:03.230696Z","shell.execute_reply":"2023-12-24T07:09:11.256464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, roc_curve, auc\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport statsmodels.api as sm\n\n# Load the model and predict on the validation data\nmodel.load_weights('testnew44.h5')\ny_val_pred = model.predict(x_val)\n\n# Instantiate an OptimizedRounder object and fit on the validation data\noptR = OptimizedRounder()\noptR.fit(y_val_pred, y_val)\ncoefficients = optR.coefficients()\nprint(f'Coefficients: {coefficients}')\n\n# Get the predicted labels using the optimized coefficients\ny_val_pred_rounded = optR.predict(y_val_pred, coefficients)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T07:09:11.258989Z","iopub.execute_input":"2023-12-24T07:09:11.259284Z","iopub.status.idle":"2023-12-24T07:09:26.959789Z","shell.execute_reply.started":"2023-12-24T07:09:11.259254Z","shell.execute_reply":"2023-12-24T07:09:26.958922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute confusion matrix\ncm = confusion_matrix(y_val, y_val_pred_rounded)\n\n# Plot confusion matrix\nfig, ax = plt.subplots(figsize=(8, 6))\nsns.heatmap(cm, annot=True, cmap='Blues', fmt='g', ax=ax)\n\n# Add labels, title, and axis ticks\nax.set_xlabel('Predicted labels')\nax.set_ylabel('True labels')\nax.set_title('Confusion Matrix')\nax.xaxis.set_ticklabels(['No-DR', 'Mild NPDR', 'Moderate NPDR', 'Severe NPDR', 'PDR'])\nax.yaxis.set_ticklabels(['No-DR', 'Mild NPDR', 'Moderate NPDR', 'Severe NPDR', 'PDR'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-24T07:09:26.960767Z","iopub.execute_input":"2023-12-24T07:09:26.961318Z","iopub.status.idle":"2023-12-24T07:09:27.290327Z","shell.execute_reply.started":"2023-12-24T07:09:26.961288Z","shell.execute_reply":"2023-12-24T07:09:27.289496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2023-12-24T07:09:27.291288Z","iopub.execute_input":"2023-12-24T07:09:27.291547Z","iopub.status.idle":"2023-12-24T07:09:27.295154Z","shell.execute_reply.started":"2023-12-24T07:09:27.291519Z","shell.execute_reply":"2023-12-24T07:09:27.294355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy = accuracy_score(y_val, y_val_pred_rounded)\n\nprint('Accuracy:', accuracy)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T07:09:27.296189Z","iopub.execute_input":"2023-12-24T07:09:27.296437Z","iopub.status.idle":"2023-12-24T07:09:27.306636Z","shell.execute_reply.started":"2023-12-24T07:09:27.296410Z","shell.execute_reply":"2023-12-24T07:09:27.305974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report\n\n# assuming y_val and y_val_pred_rounded are the true and predicted labels respectively\ncm = confusion_matrix(y_val, y_val_pred_rounded)\n\n# Calculate precision, recall, and accuracy\nreport = classification_report(y_val, y_val_pred_rounded, target_names=['No-DR', 'Mild NPDR', 'Moderate NPDR', 'Severe NPDR', 'PDR'])\n\n# Print confusion matrix, precision, recall, and accuracy\nprint('Confusion Matrix:\\n', cm)\nprint('Classification Report:\\n', report)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T07:09:27.307459Z","iopub.execute_input":"2023-12-24T07:09:27.307706Z","iopub.status.idle":"2023-12-24T07:09:27.328679Z","shell.execute_reply.started":"2023-12-24T07:09:27.307680Z","shell.execute_reply":"2023-12-24T07:09:27.328039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Map 0 to 'No DR' and 1,2,3,4 to 'DR' in y_true\ny_true_binary = np.where(y_val == 0, 0, 1)\ny_val_pred_rounded_binary = np.where(y_val_pred_rounded == 0, 0, 1)\n\n# Compute ROC curve and AUC\nfpr, tpr, thresholds = roc_curve(y_true_binary, y_val_pred_rounded_binary, pos_label=1)\nroc_auc = auc(fpr, tpr)\n\nplt.plot(fpr, tpr, color='darkorange', lw=2, label='ROC curve (area = %0.2f)' % roc_auc)\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver operating characteristic curve')\nplt.legend(loc=\"lower right\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-24T07:09:27.329499Z","iopub.execute_input":"2023-12-24T07:09:27.329725Z","iopub.status.idle":"2023-12-24T07:09:27.479887Z","shell.execute_reply.started":"2023-12-24T07:09:27.329701Z","shell.execute_reply":"2023-12-24T07:09:27.479248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc = accuracy_score(y_true_binary, y_val_pred_rounded_binary)\n\nprint('Accuracy for binary classification:', acc)","metadata":{"execution":{"iopub.status.busy":"2023-12-24T07:09:27.480653Z","iopub.execute_input":"2023-12-24T07:09:27.480877Z","iopub.status.idle":"2023-12-24T07:09:27.485673Z","shell.execute_reply.started":"2023-12-24T07:09:27.480853Z","shell.execute_reply":"2023-12-24T07:09:27.484918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute confusion matrix\ncm = confusion_matrix(y_true_binary, y_val_pred_rounded_binary)\n\n# Calculate specificity and sensitivity\ntn, fp, fn, tp = cm.ravel()\nspecificity = tn / (tn + fp)\nsensitivity = tp / (tp + fn)\n\n# Display confusion matrix with percentages\nfig, ax = plt.subplots(figsize=(8, 6))\nsns.heatmap(cm, annot=True, cmap='Blues', fmt='g', ax=ax)\n\n# Add labels, title, and axis ticks\nax.set_xlabel('Predicted labels')\nax.set_ylabel('True labels')\nax.set_title('Confusion Matrix (Specificity={:.2f}, Sensitivity={:.2f})'.format(specificity, sensitivity))\nax.xaxis.set_ticklabels(['No-DR', 'DR'])\nax.yaxis.set_ticklabels(['No-DR', 'DR'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-24T07:09:27.486536Z","iopub.execute_input":"2023-12-24T07:09:27.487095Z","iopub.status.idle":"2023-12-24T07:09:27.655874Z","shell.execute_reply.started":"2023-12-24T07:09:27.487069Z","shell.execute_reply":"2023-12-24T07:09:27.655171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}