{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"},{"sourceId":418031,"sourceType":"datasetVersion","datasetId":131128},{"sourceId":2812287,"sourceType":"datasetVersion","datasetId":1719146},{"sourceId":2819730,"sourceType":"datasetVersion","datasetId":1723812},{"sourceId":2823796,"sourceType":"datasetVersion","datasetId":1727005},{"sourceId":5579252,"sourceType":"datasetVersion","datasetId":3211156},{"sourceId":5579268,"sourceType":"datasetVersion","datasetId":3211163}],"dockerImageVersionId":30628,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# To have reproducible results and compare them\nnr_seed = 2023\nimport numpy as np \nnp.random.seed(nr_seed)\nimport tensorflow as tf\ntf.random.set_seed(nr_seed)\nprint(tf.__version__)\n\n# Libraries\nimport json\nimport math\nimport os\n\n\nimport scipy as sp\nfrom functools import partial\nfrom collections import Counter\nimport json\n\nfrom PIL import Image\nimport numpy as np\nfrom keras import backend as K\nfrom keras import layers\nfrom tensorflow.keras.applications import EfficientNetV2B3\nfrom keras.callbacks import Callback, ModelCheckpoint\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential\nfrom keras.optimizers import Adam\nfrom tensorflow.keras.utils import plot_model\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport gc\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-19T13:42:13.800665Z","iopub.execute_input":"2023-12-19T13:42:13.800888Z","iopub.status.idle":"2023-12-19T13:42:29.320028Z","shell.execute_reply.started":"2023-12-19T13:42:13.800863Z","shell.execute_reply":"2023-12-19T13:42:29.319324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install scikit-learn","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:42:29.320959Z","iopub.execute_input":"2023-12-19T13:42:29.321397Z","iopub.status.idle":"2023-12-19T13:42:33.191982Z","shell.execute_reply.started":"2023-12-19T13:42:29.321369Z","shell.execute_reply":"2023-12-19T13:42:33.191108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import cohen_kappa_score","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:42:33.193931Z","iopub.execute_input":"2023-12-19T13:42:33.194331Z","iopub.status.idle":"2023-12-19T13:42:33.769857Z","shell.execute_reply.started":"2023-12-19T13:42:33.194299Z","shell.execute_reply":"2023-12-19T13:42:33.769194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# detect and init the TPU\n#tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\n\n# instantiate a distribution strategy\n#tpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)\n\ntpu = tf.distribute.cluster_resolver.TPUClusterResolver()\ntf.config.experimental_connect_to_cluster(tpu)\ntf.tpu.experimental.initialize_tpu_system(tpu)\nstrategy = tf.distribute.experimental.TPUStrategy(tpu)","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:42:33.770744Z","iopub.execute_input":"2023-12-19T13:42:33.771425Z","iopub.status.idle":"2023-12-19T13:42:42.301626Z","shell.execute_reply.started":"2023-12-19T13:42:33.771394Z","shell.execute_reply":"2023-12-19T13:42:42.300651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image size\nim_size = 224\n# Batch size\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:42:42.302474Z","iopub.execute_input":"2023-12-19T13:42:42.302702Z","iopub.status.idle":"2023-12-19T13:42:42.306018Z","shell.execute_reply.started":"2023-12-19T13:42:42.302677Z","shell.execute_reply":"2023-12-19T13:42:42.305312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:42:42.306899Z","iopub.execute_input":"2023-12-19T13:42:42.307168Z","iopub.status.idle":"2023-12-19T13:42:42.319944Z","shell.execute_reply.started":"2023-12-19T13:42:42.307140Z","shell.execute_reply":"2023-12-19T13:42:42.319228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/trywithoutindex/train_df_withoutINDEX.csv')\nval_df = pd.read_csv('../input/trywithoutindex2/val_df_withoutINDEX.csv')\n\n\n\n\nprint(train_df.shape)\nprint(val_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:42:42.320805Z","iopub.execute_input":"2023-12-19T13:42:42.321062Z","iopub.status.idle":"2023-12-19T13:42:42.467081Z","shell.execute_reply.started":"2023-12-19T13:42:42.321021Z","shell.execute_reply":"2023-12-19T13:42:42.466202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport tqdm","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:42:42.469892Z","iopub.execute_input":"2023-12-19T13:42:42.470223Z","iopub.status.idle":"2023-12-19T13:42:43.429268Z","shell.execute_reply.started":"2023-12-19T13:42:42.470191Z","shell.execute_reply":"2023-12-19T13:42:43.428382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_samples(df, columns=4, rows=3):\n    fig=plt.figure(figsize=(5*columns, 4*rows))\n\n    for i in range(columns*rows):\n        image_path = df.loc[i,'id_code']\n        image_id = df.loc[i,'diagnosis']\n        img = cv2.imread(f'{image_path}')\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        #img = crop_image_from_gray(img)\n        img = cv2.resize(img, (im_size,im_size))\n        img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), im_size/40) ,-4 ,128)\n        \n        fig.add_subplot(rows, columns, i+1)\n        plt.title(image_id)\n        plt.imshow(img)\n    \n    plt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:42:43.430278Z","iopub.execute_input":"2023-12-19T13:42:43.430544Z","iopub.status.idle":"2023-12-19T13:42:43.436521Z","shell.execute_reply.started":"2023-12-19T13:42:43.430517Z","shell.execute_reply":"2023-12-19T13:42:43.435748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display train images\ndisplay_samples(train_df)","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:42:43.437514Z","iopub.execute_input":"2023-12-19T13:42:43.437972Z","iopub.status.idle":"2023-12-19T13:42:46.558287Z","shell.execute_reply.started":"2023-12-19T13:42:43.437933Z","shell.execute_reply":"2023-12-19T13:42:46.557243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display train images\ndisplay_samples(val_df)","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:42:46.560385Z","iopub.execute_input":"2023-12-19T13:42:46.560644Z","iopub.status.idle":"2023-12-19T13:42:51.386129Z","shell.execute_reply.started":"2023-12-19T13:42:46.560617Z","shell.execute_reply":"2023-12-19T13:42:51.384820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = BATCH_SIZE*2","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:42:51.387445Z","iopub.execute_input":"2023-12-19T13:42:51.387754Z","iopub.status.idle":"2023-12-19T13:42:51.392032Z","shell.execute_reply.started":"2023-12-19T13:42:51.387722Z","shell.execute_reply":"2023-12-19T13:42:51.391202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:42:51.392987Z","iopub.execute_input":"2023-12-19T13:42:51.393292Z","iopub.status.idle":"2023-12-19T13:42:51.409253Z","shell.execute_reply.started":"2023-12-19T13:42:51.393253Z","shell.execute_reply":"2023-12-19T13:42:51.408364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:42:51.410257Z","iopub.execute_input":"2023-12-19T13:42:51.410505Z","iopub.status.idle":"2023-12-19T13:42:51.418925Z","shell.execute_reply.started":"2023-12-19T13:42:51.410480Z","shell.execute_reply":"2023-12-19T13:42:51.418086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def crop_image_from_gray(img,tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n            img = np.stack([img1,img2,img3],axis=-1)\n        return img\n\n    \n    \ndef preprocess_image(image_path, desired_size=224):\n    img = cv2.imread(image_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = crop_image_from_gray(img)\n    img = cv2.resize(img, (desired_size,desired_size))\n    img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), desired_size/30) ,-4 ,128)\n    return img\n\n\ndef preprocess_image_old(image_path, desired_size=224):\n    img = cv2.imread(image_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    #img = crop_image_from_gray(img)\n    img = cv2.resize(img, (desired_size,desired_size))\n    img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), desired_size/40) ,-4 ,128)\n    return img","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:42:51.419967Z","iopub.execute_input":"2023-12-19T13:42:51.420243Z","iopub.status.idle":"2023-12-19T13:42:51.431635Z","shell.execute_reply.started":"2023-12-19T13:42:51.420218Z","shell.execute_reply":"2023-12-19T13:42:51.430891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# validation settebook\nN = val_df.shape[0]\nx_val = np.empty((N, im_size, im_size, 3), dtype=np.uint8)\n\nfor i, image_id in tqdm(enumerate(val_df['id_code']), total=N):\n    x_val[i, :, :, :] = preprocess_image(\n        f'{image_id}',\n        desired_size=im_size\n    )","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:42:51.432526Z","iopub.execute_input":"2023-12-19T13:42:51.432746Z","iopub.status.idle":"2023-12-19T13:53:55.939456Z","shell.execute_reply.started":"2023-12-19T13:42:51.432723Z","shell.execute_reply":"2023-12-19T13:53:55.938551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = train_df['diagnosis'].values\ny_val = val_df['diagnosis'].values\n\nprint(y_train.shape)\nprint(x_val.shape)\nprint(y_val.shape)","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:53:55.941085Z","iopub.execute_input":"2023-12-19T13:53:55.941382Z","iopub.status.idle":"2023-12-19T13:53:55.946230Z","shell.execute_reply.started":"2023-12-19T13:53:55.941351Z","shell.execute_reply":"2023-12-19T13:53:55.945603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Metrics(Callback):\n    def __init__(self, validation_data=()):\n        super().__init__()\n        self.validation_data = validation_data\n\n    def on_epoch_end(self, epoch, logs=None):\n        x_val, y_val = self.validation_data[0], self.validation_data[1]\n        \n        y_pred = self.model.predict(x_val)\n        \n        coef = [0.5, 1.5, 2.5, 3.5]\n\n        for i, pred in enumerate(y_pred):\n            if pred < coef[0]:\n                y_pred[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                y_pred[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                y_pred[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                y_pred[i] = 3\n            else:\n                y_pred[i] = 4\n\n        _val_kappa = cohen_kappa_score(\n            y_val,\n            y_pred, \n            weights='quadratic')\n        self.val_kappas.append(_val_kappa)\n        print(f\"val_kappa: {_val_kappa:.4f}\")\n\n\n        if _val_kappa == max(self.val_kappas):\n            print(\"Validation Kappa has improved. Saving model.\")\n            self.model.save('mintest1.h5')\n\n        return","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:53:55.947125Z","iopub.execute_input":"2023-12-19T13:53:55.947359Z","iopub.status.idle":"2023-12-19T13:53:55.960417Z","shell.execute_reply.started":"2023-12-19T13:53:55.947334Z","shell.execute_reply":"2023-12-19T13:53:55.959808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_datagen():\n    return ImageDataGenerator(\n        horizontal_flip = True,\n        vertical_flip = True,\n        rotation_range = 160,\n        zoom_range=0.35\n    )","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:53:55.961216Z","iopub.execute_input":"2023-12-19T13:53:55.961437Z","iopub.status.idle":"2023-12-19T13:53:55.974931Z","shell.execute_reply.started":"2023-12-19T13:53:55.961414Z","shell.execute_reply":"2023-12-19T13:53:55.974337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model():\n    effnet = EfficientNetV2B3(\n    input_shape=(im_size,im_size,3),\n    weights='imagenet',\n    include_top=False)\n    model = Sequential()\n    model.add(effnet)\n    model.add(layers.Dropout(0.25))\n    model.add(layers.Dense(2048))\n    model.add(layers.LeakyReLU())\n    model.add(layers.GlobalAveragePooling2D())\n    model.add(layers.Dropout(0.5))\n    model.add(layers.Dense(1, activation='linear'))\n    \n    \n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:53:55.975721Z","iopub.execute_input":"2023-12-19T13:53:55.975950Z","iopub.status.idle":"2023-12-19T13:53:55.985766Z","shell.execute_reply.started":"2023-12-19T13:53:55.975927Z","shell.execute_reply":"2023-12-19T13:53:55.985164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflow","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:53:55.986583Z","iopub.execute_input":"2023-12-19T13:53:55.986801Z","iopub.status.idle":"2023-12-19T13:53:59.446143Z","shell.execute_reply.started":"2023-12-19T13:53:55.986778Z","shell.execute_reply":"2023-12-19T13:53:59.445027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:53:59.447610Z","iopub.execute_input":"2023-12-19T13:53:59.447925Z","iopub.status.idle":"2023-12-19T13:53:59.452316Z","shell.execute_reply.started":"2023-12-19T13:53:59.447892Z","shell.execute_reply":"2023-12-19T13:53:59.451660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    model = build_model()\n    model.compile(\n        loss='mean_squared_error',\n        #optimizer=Adam(lr=0.001,decay=1e-6),\n        optimizer = tf.keras.optimizers.Adam(learning_rate=0.0001),\n        metrics=['mae']\n    )","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:53:59.457042Z","iopub.execute_input":"2023-12-19T13:53:59.457295Z","iopub.status.idle":"2023-12-19T13:54:34.768935Z","shell.execute_reply.started":"2023-12-19T13:53:59.457269Z","shell.execute_reply":"2023-12-19T13:54:34.767889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:54:34.769945Z","iopub.execute_input":"2023-12-19T13:54:34.770348Z","iopub.status.idle":"2023-12-19T13:54:34.817973Z","shell.execute_reply.started":"2023-12-19T13:54:34.770317Z","shell.execute_reply":"2023-12-19T13:54:34.817097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_bucket = 4\ndiv = round(train_df.shape[0]/num_bucket)\ndiv","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:54:34.818957Z","iopub.execute_input":"2023-12-19T13:54:34.819222Z","iopub.status.idle":"2023-12-19T13:54:34.824314Z","shell.execute_reply.started":"2023-12-19T13:54:34.819195Z","shell.execute_reply":"2023-12-19T13:54:34.823537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:54:34.825213Z","iopub.execute_input":"2023-12-19T13:54:34.825480Z","iopub.status.idle":"2023-12-19T13:54:34.835130Z","shell.execute_reply.started":"2023-12-19T13:54:34.825454Z","shell.execute_reply":"2023-12-19T13:54:34.834413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = pd.DataFrame({\n                        'val_loss': [0.0],\n                        'val_mean_absolute_error': [0.0],\n                        'loss': [0.0], \n                        'mean_absolute_error': [0.0],\n                        'bucket': [0.0]\n                        })","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:54:34.836040Z","iopub.execute_input":"2023-12-19T13:54:34.836310Z","iopub.status.idle":"2023-12-19T13:54:34.844650Z","shell.execute_reply.started":"2023-12-19T13:54:34.836283Z","shell.execute_reply":"2023-12-19T13:54:34.843959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Epochs\nepochs = [24,24,24,24]\nkappa_metrics = Metrics(validation_data=(x_val, y_val))\nkappa_metrics.val_kappas = []","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:54:34.845643Z","iopub.execute_input":"2023-12-19T13:54:34.845913Z","iopub.status.idle":"2023-12-19T13:54:34.855998Z","shell.execute_reply.started":"2023-12-19T13:54:34.845877Z","shell.execute_reply":"2023-12-19T13:54:34.855089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(epochs)  ","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:54:34.857020Z","iopub.execute_input":"2023-12-19T13:54:34.857299Z","iopub.status.idle":"2023-12-19T13:54:34.867196Z","shell.execute_reply.started":"2023-12-19T13:54:34.857272Z","shell.execute_reply":"2023-12-19T13:54:34.866348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(0,num_bucket):\n    if i != (num_bucket-1):\n        print(\"Bucket Nr: {}\".format(i))\n        \n        N = train_df.iloc[i*div:(1+i)*div].shape[0]\n        x_train = np.empty((N, im_size, im_size, 3), dtype=np.uint8)\n        for j, image_id in enumerate(tqdm(train_df.iloc[i*div:(1+i)*div,0])):\n            x_train[j, :, :, :] = preprocess_image_old(f'{image_id}', desired_size = im_size)\n\n        data_generator = create_datagen().flow(x_train, y_train[i*div:(1+i)*div], batch_size=BATCH_SIZE)\n        history = model.fit_generator(\n                        data_generator,\n                        steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                        epochs=epochs[i],\n                        validation_data=(x_val, y_val),\n                        callbacks=[kappa_metrics]\n                        )\n        \n        dic = history.history\n        df_model = pd.DataFrame(dic)\n        df_model['bucket'] = i\n    else:\n        print(\"Bucket Nr: {}\".format(i))\n        N = train_df.iloc[i*div:].shape[0]\n        x_train = np.empty((N, im_size, im_size, 3), dtype=np.uint8)\n        for j, image_id in enumerate(tqdm(train_df.iloc[i*div:,0])):\n            x_train[j, :, :, :] = preprocess_image_old(f'{image_id}', desired_size = im_size)\n            \n        data_generator = create_datagen().flow(x_train, y_train[i*div:], batch_size=BATCH_SIZE)\n        history = model.fit_generator(\n                        data_generator,\n                        steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                        epochs=epochs[i],\n                        validation_data=(x_val, y_val),\n                        callbacks=[kappa_metrics]\n                        )\n        \n        dic = history.history\n        df_model = pd.DataFrame(dic)\n        df_model['bucket'] = i\n\n    results = pd.concat([results, df_model], ignore_index=True)\n    del data_generator\n    del x_train\n    gc.collect()\n    print('-'*40)","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:54:34.868455Z","iopub.execute_input":"2023-12-19T13:54:34.868730Z","iopub.status.idle":"2023-12-19T17:06:00.138251Z","shell.execute_reply.started":"2023-12-19T13:54:34.868702Z","shell.execute_reply":"2023-12-19T17:06:00.137091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = results.iloc[1:]\nkappa_vals = kappa_metrics.val_kappas[:len(results)]\nresults['kappa'] = kappa_vals\nresults = results.reset_index(drop=True)\nresults = results.rename(index=str, columns={\"index\": \"epoch\"})\nprint(max(results.kappa))","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:06:00.142152Z","iopub.execute_input":"2023-12-19T17:06:00.142438Z","iopub.status.idle":"2023-12-19T17:06:00.153901Z","shell.execute_reply.started":"2023-12-19T17:06:00.142409Z","shell.execute_reply":"2023-12-19T17:06:00.153088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results[['loss', 'val_loss']].plot()\nresults[['mae', 'val_mae']].plot()\nresults[['kappa']].plot()\nresults.to_csv('model_results.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:06:00.154894Z","iopub.execute_input":"2023-12-19T17:06:00.155200Z","iopub.status.idle":"2023-12-19T17:06:00.685619Z","shell.execute_reply.started":"2023-12-19T17:06:00.155168Z","shell.execute_reply":"2023-12-19T17:06:00.684723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class OptimizedRounder(object):\n    def __init__(self):\n        self.coef_ = 0\n\n    def _kappa_loss(self, coef, X, y):\n        X_p = np.copy(X)\n        for i, pred in enumerate(X_p):\n            if pred < coef[0]:\n                X_p[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                X_p[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                X_p[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                X_p[i] = 3\n            else:\n                X_p[i] = 4\n\n        ll = cohen_kappa_score(y, X_p, weights='quadratic')\n        return -ll\n\n    def fit(self, X, y):\n        loss_partial = partial(self._kappa_loss, X=X, y=y)\n        initial_coef = [0.5, 1.5, 2.5, 3.5]\n        self.coef_ = sp.optimize.minimize(loss_partial, initial_coef, method='nelder-mead')\n        print(-loss_partial(self.coef_['x']))\n\n    def predict(self, X, coef):\n        X_p = np.copy(X)\n        for i, pred in enumerate(X_p):\n            if pred < coef[0]:\n                 X_p[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                X_p[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                X_p[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                X_p[i] = 3\n            else:\n                X_p[i] = 4\n        return X_p\n\n    def coefficients(self):\n        return self.coef_['x']","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:06:00.686604Z","iopub.execute_input":"2023-12-19T17:06:00.686866Z","iopub.status.idle":"2023-12-19T17:06:00.696282Z","shell.execute_reply.started":"2023-12-19T17:06:00.686840Z","shell.execute_reply":"2023-12-19T17:06:00.695511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_weights('mintest1.h5')  \ny_val_pred = model.predict(x_val)\n\noptR = OptimizedRounder()\noptR.fit(y_val_pred, y_val)\ncoefficients = optR.coefficients()\nprint(f'Coefficients: {coefficients}')\ny_val_pred = optR.predict(y_val_pred, coefficients)\n\nscore = cohen_kappa_score(y_val_pred, y_val, weights='quadratic')\n\nprint('Optimized Validation QWK score: {}'.format(score))\nprint('Not Optimized Validation QWK score: {}'.format(max(results.kappa)))","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:06:00.697201Z","iopub.execute_input":"2023-12-19T17:06:00.697430Z","iopub.status.idle":"2023-12-19T17:06:16.684191Z","shell.execute_reply.started":"2023-12-19T17:06:00.697406Z","shell.execute_reply":"2023-12-19T17:06:16.683089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('EfficientNetV2B3_Blanced_ss.h5')","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:06:16.685375Z","iopub.execute_input":"2023-12-19T17:06:16.685712Z","iopub.status.idle":"2023-12-19T17:06:19.638796Z","shell.execute_reply.started":"2023-12-19T17:06:16.685678Z","shell.execute_reply":"2023-12-19T17:06:19.637563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!apt-get update && apt-get install -y python3-seaborn\n!pip install seaborn\n!apt-get update && apt-get install -y python3-statsmodel\n!pip install statsmodel","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:06:19.640875Z","iopub.execute_input":"2023-12-19T17:06:19.641163Z","iopub.status.idle":"2023-12-19T17:08:29.503073Z","shell.execute_reply.started":"2023-12-19T17:06:19.641136Z","shell.execute_reply":"2023-12-19T17:08:29.501613Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install statsmodels","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:08:29.504766Z","iopub.execute_input":"2023-12-19T17:08:29.505085Z","iopub.status.idle":"2023-12-19T17:08:38.033032Z","shell.execute_reply.started":"2023-12-19T17:08:29.505040Z","shell.execute_reply":"2023-12-19T17:08:38.031793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, roc_curve, auc\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport statsmodels.api as sm\n\n# Load the model and predict on the validation data\nmodel.load_weights('mintest1.h5')\ny_val_pred = model.predict(x_val)\n\n# Instantiate an OptimizedRounder object and fit on the validation data\noptR = OptimizedRounder()\noptR.fit(y_val_pred, y_val)\ncoefficients = optR.coefficients()\nprint(f'Coefficients: {coefficients}')\n\n# Get the predicted labels using the optimized coefficients\ny_val_pred_rounded = optR.predict(y_val_pred, coefficients)","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:08:38.034508Z","iopub.execute_input":"2023-12-19T17:08:38.034816Z","iopub.status.idle":"2023-12-19T17:08:53.886872Z","shell.execute_reply.started":"2023-12-19T17:08:38.034787Z","shell.execute_reply":"2023-12-19T17:08:53.885611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute confusion matrix\ncm = confusion_matrix(y_val, y_val_pred_rounded)\n\n# Plot confusion matrix\nfig, ax = plt.subplots(figsize=(8, 6))\nsns.heatmap(cm, annot=True, cmap='Blues', fmt='g', ax=ax)\n\n# Add labels, title, and axis ticks\nax.set_xlabel('Predicted labels')\nax.set_ylabel('True labels')\nax.set_title('Confusion Matrix')\nax.xaxis.set_ticklabels(['No-DR', 'Mild NPDR', 'Moderate NPDR', 'Severe NPDR', 'PDR'])\nax.yaxis.set_ticklabels(['No-DR', 'Mild NPDR', 'Moderate NPDR', 'Severe NPDR', 'PDR'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:08:53.888251Z","iopub.execute_input":"2023-12-19T17:08:53.888929Z","iopub.status.idle":"2023-12-19T17:08:54.221972Z","shell.execute_reply.started":"2023-12-19T17:08:53.888890Z","shell.execute_reply":"2023-12-19T17:08:54.220844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:08:54.223121Z","iopub.execute_input":"2023-12-19T17:08:54.223402Z","iopub.status.idle":"2023-12-19T17:08:54.227539Z","shell.execute_reply.started":"2023-12-19T17:08:54.223374Z","shell.execute_reply":"2023-12-19T17:08:54.226741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy = accuracy_score(y_val, y_val_pred_rounded)\n\nprint('Accuracy:', accuracy)","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:08:54.228614Z","iopub.execute_input":"2023-12-19T17:08:54.228894Z","iopub.status.idle":"2023-12-19T17:08:54.238897Z","shell.execute_reply.started":"2023-12-19T17:08:54.228866Z","shell.execute_reply":"2023-12-19T17:08:54.237999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report\n\n# assuming y_val and y_val_pred_rounded are the true and predicted labels respectively\ncm = confusion_matrix(y_val, y_val_pred_rounded)\n\n# Calculate precision, recall, and accuracy\nreport = classification_report(y_val, y_val_pred_rounded, target_names=['No-DR', 'Mild NPDR', 'Moderate NPDR', 'Severe NPDR', 'PDR'])\n\n# Print confusion matrix, precision, recall, and accuracy\nprint('Confusion Matrix:\\n', cm)\nprint('Classification Report:\\n', report)","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:08:54.240083Z","iopub.execute_input":"2023-12-19T17:08:54.240343Z","iopub.status.idle":"2023-12-19T17:08:54.262368Z","shell.execute_reply.started":"2023-12-19T17:08:54.240317Z","shell.execute_reply":"2023-12-19T17:08:54.261605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Map 0 to 'No DR' and 1,2,3,4 to 'DR' in y_true\ny_true_binary = np.where(y_val == 0, 0, 1)\ny_val_pred_rounded_binary = np.where(y_val_pred_rounded == 0, 0, 1)\n\n# Compute ROC curve and AUC\nfpr, tpr, thresholds = roc_curve(y_true_binary, y_val_pred_rounded_binary, pos_label=1)\nroc_auc = auc(fpr, tpr)\n\nplt.plot(fpr, tpr, color='darkorange', lw=2, label='ROC curve (area = %0.2f)' % roc_auc)\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver operating characteristic curve')\nplt.legend(loc=\"lower right\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:08:54.263299Z","iopub.execute_input":"2023-12-19T17:08:54.263565Z","iopub.status.idle":"2023-12-19T17:08:54.413712Z","shell.execute_reply.started":"2023-12-19T17:08:54.263539Z","shell.execute_reply":"2023-12-19T17:08:54.412831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc = accuracy_score(y_true_binary, y_val_pred_rounded_binary)\n\nprint('Accuracy for binary classification:', acc)","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:08:54.414839Z","iopub.execute_input":"2023-12-19T17:08:54.415163Z","iopub.status.idle":"2023-12-19T17:08:54.420674Z","shell.execute_reply.started":"2023-12-19T17:08:54.415135Z","shell.execute_reply":"2023-12-19T17:08:54.419734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute confusion matrix\ncm = confusion_matrix(y_true_binary, y_val_pred_rounded_binary)\n\n# Calculate specificity and sensitivity\ntn, fp, fn, tp = cm.ravel()\nspecificity = tn / (tn + fp)\nsensitivity = tp / (tp + fn)\n\n# Display confusion matrix with percentages\nfig, ax = plt.subplots(figsize=(8, 6))\nsns.heatmap(cm, annot=True, cmap='Blues', fmt='g', ax=ax)\n\n# Add labels, title, and axis ticks\nax.set_xlabel('Predicted labels')\nax.set_ylabel('True labels')\nax.set_title('Confusion Matrix (Specificity={:.2f}, Sensitivity={:.2f})'.format(specificity, sensitivity))\nax.xaxis.set_ticklabels(['No-DR', 'DR'])\nax.yaxis.set_ticklabels(['No-DR', 'DR'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:08:54.421627Z","iopub.execute_input":"2023-12-19T17:08:54.421891Z","iopub.status.idle":"2023-12-19T17:08:54.603954Z","shell.execute_reply.started":"2023-12-19T17:08:54.421865Z","shell.execute_reply":"2023-12-19T17:08:54.603085Z"},"trusted":true},"execution_count":null,"outputs":[]}]}