{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"},{"sourceId":5579822,"sourceType":"datasetVersion","datasetId":3211385},{"sourceId":5579826,"sourceType":"datasetVersion","datasetId":3211388},{"sourceId":418031,"sourceType":"datasetVersion","datasetId":131128},{"sourceId":2812287,"sourceType":"datasetVersion","datasetId":1719146},{"sourceId":2819730,"sourceType":"datasetVersion","datasetId":1723812},{"sourceId":2823796,"sourceType":"datasetVersion","datasetId":1727005}],"dockerImageVersionId":30628,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# To have reproducible results and compare them\nnr_seed = 2023\nimport numpy as np \nnp.random.seed(nr_seed)\nimport tensorflow as tf\ntf.random.set_seed(nr_seed)\nprint(tf.__version__)\n\n# Libraries\nimport json\nimport math\nimport os\n\n\nimport scipy as sp\nfrom functools import partial\nfrom collections import Counter\nimport json\n\nfrom PIL import Image\nimport numpy as np\nfrom keras import backend as K\nfrom keras import layers\nfrom tensorflow.keras.applications import EfficientNetV2B3\nfrom keras.callbacks import Callback, ModelCheckpoint\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential\nfrom keras.optimizers import Adam\nfrom tensorflow.keras.utils import plot_model\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport gc\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"# To have reproducible results and compare them\nnr_seed = 2023\nimport numpy as np \nnp.random.seed(nr_seed)\nimport tensorflow as tf\ntf.random.set_seed(nr_seed)\nprint(tf.__version__)\n\n# Libraries\nimport json\nimport math\nimport os\n\n\nimport scipy as sp\nfrom functools import partial\nfrom collections import Counter\nimport json\n\nfrom PIL import Image\nimport numpy as np\nfrom keras import backend as K\nfrom keras import layers\nfrom tensorflow.keras.applications import EfficientNetV2B3\nfrom keras.callbacks import Callback, ModelCheckpoint\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential\nfrom keras.optimizers import Adam\nfrom tensorflow.keras.utils import plot_model\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport gc\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:45:52.435966Z","iopub.execute_input":"2023-12-17T03:45:52.436198Z","iopub.status.idle":"2023-12-17T03:46:07.583019Z","shell.execute_reply.started":"2023-12-17T03:45:52.436173Z","shell.execute_reply":"2023-12-17T03:46:07.581861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install scikit-learn","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:46:07.584544Z","iopub.execute_input":"2023-12-17T03:46:07.585038Z","iopub.status.idle":"2023-12-17T03:46:11.477762Z","shell.execute_reply.started":"2023-12-17T03:46:07.585004Z","shell.execute_reply":"2023-12-17T03:46:11.476792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import cohen_kappa_score","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:46:11.479093Z","iopub.execute_input":"2023-12-17T03:46:11.479377Z","iopub.status.idle":"2023-12-17T03:46:12.062967Z","shell.execute_reply.started":"2023-12-17T03:46:11.479347Z","shell.execute_reply":"2023-12-17T03:46:12.062259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# detect and init the TPU\n#tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\n\n# instantiate a distribution strategy\n#tpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)\n\ntpu = tf.distribute.cluster_resolver.TPUClusterResolver()\ntf.config.experimental_connect_to_cluster(tpu)\ntf.tpu.experimental.initialize_tpu_system(tpu)\nstrategy = tf.distribute.experimental.TPUStrategy(tpu)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:46:12.064587Z","iopub.execute_input":"2023-12-17T03:46:12.065243Z","iopub.status.idle":"2023-12-17T03:46:20.476263Z","shell.execute_reply.started":"2023-12-17T03:46:12.065215Z","shell.execute_reply":"2023-12-17T03:46:20.475519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image size\nim_size = 224\n# Batch size\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:46:20.477133Z","iopub.execute_input":"2023-12-17T03:46:20.477350Z","iopub.status.idle":"2023-12-17T03:46:20.480759Z","shell.execute_reply.started":"2023-12-17T03:46:20.477326Z","shell.execute_reply":"2023-12-17T03:46:20.480076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:46:20.481568Z","iopub.execute_input":"2023-12-17T03:46:20.481792Z","iopub.status.idle":"2023-12-17T03:46:20.497174Z","shell.execute_reply.started":"2023-12-17T03:46:20.481769Z","shell.execute_reply":"2023-12-17T03:46:20.496547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/train-df/train_df_withoutINDEX.csv')\nval_df = pd.read_csv('../input/val-df/val_df_withoutINDEX.csv')\n\n\n\nprint(train_df.shape)\nprint(val_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:46:20.497969Z","iopub.execute_input":"2023-12-17T03:46:20.498183Z","iopub.status.idle":"2023-12-17T03:46:20.598626Z","shell.execute_reply.started":"2023-12-17T03:46:20.498161Z","shell.execute_reply":"2023-12-17T03:46:20.597921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!apt-get update && apt-get install -y python3-opencv\n!pip install opencv-python\n!apt-get update && apt-get install -y python3-tqdm\n!pip install tqdm","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:46:20.599645Z","iopub.execute_input":"2023-12-17T03:46:20.599949Z","iopub.status.idle":"2023-12-17T03:48:30.362209Z","shell.execute_reply.started":"2023-12-17T03:46:20.599918Z","shell.execute_reply":"2023-12-17T03:48:30.361095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport tqdm","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:48:30.363525Z","iopub.execute_input":"2023-12-17T03:48:30.363824Z","iopub.status.idle":"2023-12-17T03:48:31.163191Z","shell.execute_reply.started":"2023-12-17T03:48:30.363779Z","shell.execute_reply":"2023-12-17T03:48:31.162344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_samples(df, columns=4, rows=3):\n    fig=plt.figure(figsize=(5*columns, 4*rows))\n\n    for i in range(columns*rows):\n        image_path = df.loc[i,'id_code']\n        image_id = df.loc[i,'diagnosis']\n        img = cv2.imread(f'{image_path}')\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        #img = crop_image_from_gray(img)\n        img = cv2.resize(img, (im_size,im_size))\n        img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), im_size/40) ,-4 ,128)\n        \n        fig.add_subplot(rows, columns, i+1)\n        plt.title(image_id)\n        plt.imshow(img)\n    \n    plt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:48:31.166369Z","iopub.execute_input":"2023-12-17T03:48:31.166657Z","iopub.status.idle":"2023-12-17T03:48:31.173062Z","shell.execute_reply.started":"2023-12-17T03:48:31.166628Z","shell.execute_reply":"2023-12-17T03:48:31.172323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display train images\ndisplay_samples(train_df)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:48:31.173964Z","iopub.execute_input":"2023-12-17T03:48:31.174212Z","iopub.status.idle":"2023-12-17T03:48:34.477615Z","shell.execute_reply.started":"2023-12-17T03:48:31.174186Z","shell.execute_reply":"2023-12-17T03:48:34.476611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display train images\ndisplay_samples(val_df)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:48:34.478787Z","iopub.execute_input":"2023-12-17T03:48:34.479063Z","iopub.status.idle":"2023-12-17T03:48:39.268242Z","shell.execute_reply.started":"2023-12-17T03:48:34.479036Z","shell.execute_reply":"2023-12-17T03:48:39.267124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = BATCH_SIZE*2","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:48:39.269720Z","iopub.execute_input":"2023-12-17T03:48:39.270065Z","iopub.status.idle":"2023-12-17T03:48:39.274652Z","shell.execute_reply.started":"2023-12-17T03:48:39.270034Z","shell.execute_reply":"2023-12-17T03:48:39.273813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:48:39.276010Z","iopub.execute_input":"2023-12-17T03:48:39.276266Z","iopub.status.idle":"2023-12-17T03:48:39.288231Z","shell.execute_reply.started":"2023-12-17T03:48:39.276241Z","shell.execute_reply":"2023-12-17T03:48:39.287288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:48:39.289279Z","iopub.execute_input":"2023-12-17T03:48:39.289544Z","iopub.status.idle":"2023-12-17T03:48:39.298257Z","shell.execute_reply.started":"2023-12-17T03:48:39.289517Z","shell.execute_reply":"2023-12-17T03:48:39.297244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def crop_image_from_gray(img,tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n            img = np.stack([img1,img2,img3],axis=-1)\n        return img\n\n    \n    \ndef preprocess_image(image_path, desired_size=224):\n    img = cv2.imread(image_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = crop_image_from_gray(img)\n    img = cv2.resize(img, (desired_size,desired_size))\n    img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), desired_size/30) ,-4 ,128)\n    return img\n\n\ndef preprocess_image_old(image_path, desired_size=224):\n    img = cv2.imread(image_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    #img = crop_image_from_gray(img)\n    img = cv2.resize(img, (desired_size,desired_size))\n    img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), desired_size/40) ,-4 ,128)\n    return img","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:48:39.299229Z","iopub.execute_input":"2023-12-17T03:48:39.299494Z","iopub.status.idle":"2023-12-17T03:48:39.319143Z","shell.execute_reply.started":"2023-12-17T03:48:39.299467Z","shell.execute_reply":"2023-12-17T03:48:39.318298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# validation settebook\nN = val_df.shape[0]\nx_val = np.empty((N, im_size, im_size, 3), dtype=np.uint8)\n\nfor i, image_id in tqdm(enumerate(val_df['id_code']), total=N):\n    x_val[i, :, :, :] = preprocess_image(\n        f'{image_id}',\n        desired_size=im_size\n    )","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:48:39.320133Z","iopub.execute_input":"2023-12-17T03:48:39.320402Z","iopub.status.idle":"2023-12-17T03:59:49.164630Z","shell.execute_reply.started":"2023-12-17T03:48:39.320375Z","shell.execute_reply":"2023-12-17T03:59:49.163555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = train_df['diagnosis'].values\ny_val = val_df['diagnosis'].values\n\nprint(y_train.shape)\nprint(x_val.shape)\nprint(y_val.shape)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:59:49.165968Z","iopub.execute_input":"2023-12-17T03:59:49.166270Z","iopub.status.idle":"2023-12-17T03:59:49.171848Z","shell.execute_reply.started":"2023-12-17T03:59:49.166239Z","shell.execute_reply":"2023-12-17T03:59:49.170858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Metrics(Callback):\n    def __init__(self, validation_data=()):\n        super().__init__()\n        self.validation_data = validation_data\n\n    def on_epoch_end(self, epoch, logs=None):\n        x_val, y_val = self.validation_data[0], self.validation_data[1]\n        \n        y_pred = self.model.predict(x_val)\n        \n        coef = [0.5, 1.5, 2.5, 3.5]\n\n        for i, pred in enumerate(y_pred):\n            if pred < coef[0]:\n                y_pred[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                y_pred[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                y_pred[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                y_pred[i] = 3\n            else:\n                y_pred[i] = 4\n\n        _val_kappa = cohen_kappa_score(\n            y_val,\n            y_pred, \n            weights='quadratic')\n        self.val_kappas.append(_val_kappa)\n        print(f\"val_kappa: {_val_kappa:.4f}\")\n\n\n        if _val_kappa == max(self.val_kappas):\n            print(\"Validation Kappa has improved. Saving model.\")\n            self.model.save('test_44.h5')\n\n        return","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:59:49.172890Z","iopub.execute_input":"2023-12-17T03:59:49.173173Z","iopub.status.idle":"2023-12-17T03:59:49.183891Z","shell.execute_reply.started":"2023-12-17T03:59:49.173145Z","shell.execute_reply":"2023-12-17T03:59:49.182997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_datagen():\n    return ImageDataGenerator(\n        horizontal_flip = True,\n        vertical_flip = True,\n        rotation_range = 160,\n        zoom_range=0.35\n    )","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:59:49.184996Z","iopub.execute_input":"2023-12-17T03:59:49.185270Z","iopub.status.idle":"2023-12-17T03:59:49.195170Z","shell.execute_reply.started":"2023-12-17T03:59:49.185243Z","shell.execute_reply":"2023-12-17T03:59:49.194205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model():\n    effnet = EfficientNetV2B3(\n    input_shape=(im_size,im_size,3),\n    weights='imagenet',\n    include_top=False)\n    model = Sequential()\n    model.add(effnet)\n    model.add(layers.Dropout(0.25))\n    model.add(layers.Dense(2048))\n    model.add(layers.LeakyReLU())\n    model.add(layers.GlobalAveragePooling2D())\n    model.add(layers.Dropout(0.5))\n    model.add(layers.Dense(1, activation='linear'))\n    \n    \n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:59:49.196270Z","iopub.execute_input":"2023-12-17T03:59:49.196585Z","iopub.status.idle":"2023-12-17T03:59:49.204398Z","shell.execute_reply.started":"2023-12-17T03:59:49.196555Z","shell.execute_reply":"2023-12-17T03:59:49.203347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflow","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:59:49.205564Z","iopub.execute_input":"2023-12-17T03:59:49.205909Z","iopub.status.idle":"2023-12-17T03:59:52.660878Z","shell.execute_reply.started":"2023-12-17T03:59:49.205879Z","shell.execute_reply":"2023-12-17T03:59:52.659719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:59:52.662096Z","iopub.execute_input":"2023-12-17T03:59:52.662354Z","iopub.status.idle":"2023-12-17T03:59:52.666215Z","shell.execute_reply.started":"2023-12-17T03:59:52.662327Z","shell.execute_reply":"2023-12-17T03:59:52.665587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    model = build_model()\n    model.compile(\n        loss='mean_squared_error',\n        #optimizer=Adam(lr=0.001,decay=1e-6),\n        optimizer = tf.keras.optimizers.Adam(learning_rate=0.0001),\n        metrics=['mae']\n    )","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:59:52.667106Z","iopub.execute_input":"2023-12-17T03:59:52.667398Z","iopub.status.idle":"2023-12-17T04:00:28.003701Z","shell.execute_reply.started":"2023-12-17T03:59:52.667371Z","shell.execute_reply":"2023-12-17T04:00:28.002863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T04:00:28.004714Z","iopub.execute_input":"2023-12-17T04:00:28.004982Z","iopub.status.idle":"2023-12-17T04:00:28.051397Z","shell.execute_reply.started":"2023-12-17T04:00:28.004955Z","shell.execute_reply":"2023-12-17T04:00:28.050594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_bucket = 4\ndiv = round(train_df.shape[0]/num_bucket)\ndiv","metadata":{"execution":{"iopub.status.busy":"2023-12-17T04:00:28.052358Z","iopub.execute_input":"2023-12-17T04:00:28.052611Z","iopub.status.idle":"2023-12-17T04:00:28.057966Z","shell.execute_reply.started":"2023-12-17T04:00:28.052584Z","shell.execute_reply":"2023-12-17T04:00:28.057027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2023-12-17T04:00:28.058944Z","iopub.execute_input":"2023-12-17T04:00:28.059197Z","iopub.status.idle":"2023-12-17T04:00:28.070881Z","shell.execute_reply.started":"2023-12-17T04:00:28.059169Z","shell.execute_reply":"2023-12-17T04:00:28.069922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = pd.DataFrame({\n                        'val_loss': [0.0],\n                        'val_mean_absolute_error': [0.0],\n                        'loss': [0.0], \n                        'mean_absolute_error': [0.0],\n                        'bucket': [0.0]\n                        })","metadata":{"execution":{"iopub.status.busy":"2023-12-17T04:00:28.075137Z","iopub.execute_input":"2023-12-17T04:00:28.075400Z","iopub.status.idle":"2023-12-17T04:00:28.089087Z","shell.execute_reply.started":"2023-12-17T04:00:28.075374Z","shell.execute_reply":"2023-12-17T04:00:28.088228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Epochs\nepochs = [12,12,12,12]\nkappa_metrics = Metrics(validation_data=(x_val, y_val))\nkappa_metrics.val_kappas = []","metadata":{"execution":{"iopub.status.busy":"2023-12-17T04:00:28.090136Z","iopub.execute_input":"2023-12-17T04:00:28.090530Z","iopub.status.idle":"2023-12-17T04:00:28.101516Z","shell.execute_reply.started":"2023-12-17T04:00:28.090478Z","shell.execute_reply":"2023-12-17T04:00:28.100663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(epochs)  ","metadata":{"execution":{"iopub.status.busy":"2023-12-17T04:00:28.102363Z","iopub.execute_input":"2023-12-17T04:00:28.102607Z","iopub.status.idle":"2023-12-17T04:00:28.123712Z","shell.execute_reply.started":"2023-12-17T04:00:28.102580Z","shell.execute_reply":"2023-12-17T04:00:28.122989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(0,num_bucket):\n    if i != (num_bucket-1):\n        print(\"Bucket Nr: {}\".format(i))\n        \n        N = train_df.iloc[i*div:(1+i)*div].shape[0]\n        x_train = np.empty((N, im_size, im_size, 3), dtype=np.uint8)\n        for j, image_id in enumerate(tqdm(train_df.iloc[i*div:(1+i)*div,0])):\n            x_train[j, :, :, :] = preprocess_image_old(f'{image_id}', desired_size = im_size)\n\n        data_generator = create_datagen().flow(x_train, y_train[i*div:(1+i)*div], batch_size=BATCH_SIZE)\n        history = model.fit_generator(\n                        data_generator,\n                        steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                        epochs=epochs[i],\n                        validation_data=(x_val, y_val),\n                        callbacks=[kappa_metrics]\n                        )\n        \n        dic = history.history\n        df_model = pd.DataFrame(dic)\n        df_model['bucket'] = i\n    else:\n        print(\"Bucket Nr: {}\".format(i))\n        N = train_df.iloc[i*div:].shape[0]\n        x_train = np.empty((N, im_size, im_size, 3), dtype=np.uint8)\n        for j, image_id in enumerate(tqdm(train_df.iloc[i*div:,0])):\n            x_train[j, :, :, :] = preprocess_image_old(f'{image_id}', desired_size = im_size)\n            \n        data_generator = create_datagen().flow(x_train, y_train[i*div:], batch_size=BATCH_SIZE)\n        history = model.fit_generator(\n                        data_generator,\n                        steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n                        epochs=epochs[i],\n                        validation_data=(x_val, y_val),\n                        callbacks=[kappa_metrics]\n                        )\n        \n        dic = history.history\n        df_model = pd.DataFrame(dic)\n        df_model['bucket'] = i\n\n    results = pd.concat([results, df_model], ignore_index=True)\n    del data_generator\n    del x_train\n    gc.collect()\n    print('-'*40)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T04:00:28.124740Z","iopub.execute_input":"2023-12-17T04:00:28.124989Z","iopub.status.idle":"2023-12-17T05:44:31.151936Z","shell.execute_reply.started":"2023-12-17T04:00:28.124965Z","shell.execute_reply":"2023-12-17T05:44:31.150724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = results.iloc[1:]\nkappa_vals = kappa_metrics.val_kappas[:len(results)]\nresults['kappa'] = kappa_vals\nresults = results.reset_index(drop=True)\nresults = results.rename(index=str, columns={\"index\": \"epoch\"})\nprint(max(results.kappa))","metadata":{"execution":{"iopub.status.busy":"2023-12-17T05:44:31.162862Z","iopub.execute_input":"2023-12-17T05:44:31.163142Z","iopub.status.idle":"2023-12-17T05:44:31.176650Z","shell.execute_reply.started":"2023-12-17T05:44:31.163114Z","shell.execute_reply":"2023-12-17T05:44:31.175680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results[['loss', 'val_loss']].plot()\nresults[['mae', 'val_mae']].plot()\nresults[['kappa']].plot()\nresults.to_csv('model_results.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T05:44:31.177796Z","iopub.execute_input":"2023-12-17T05:44:31.178111Z","iopub.status.idle":"2023-12-17T05:44:31.752296Z","shell.execute_reply.started":"2023-12-17T05:44:31.178080Z","shell.execute_reply":"2023-12-17T05:44:31.750979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class OptimizedRounder(object):\n    def __init__(self):\n        self.coef_ = 0\n\n    def _kappa_loss(self, coef, X, y):\n        X_p = np.copy(X)\n        for i, pred in enumerate(X_p):\n            if pred < coef[0]:\n                X_p[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                X_p[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                X_p[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                X_p[i] = 3\n            else:\n                X_p[i] = 4\n\n        ll = cohen_kappa_score(y, X_p, weights='quadratic')\n        return -ll\n\n    def fit(self, X, y):\n        loss_partial = partial(self._kappa_loss, X=X, y=y)\n        initial_coef = [0.5, 1.5, 2.5, 3.5]\n        self.coef_ = sp.optimize.minimize(loss_partial, initial_coef, method='nelder-mead')\n        print(-loss_partial(self.coef_['x']))\n\n    def predict(self, X, coef):\n        X_p = np.copy(X)\n        for i, pred in enumerate(X_p):\n            if pred < coef[0]:\n                 X_p[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                X_p[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                X_p[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                X_p[i] = 3\n            else:\n                X_p[i] = 4\n        return X_p\n\n    def coefficients(self):\n        return self.coef_['x']","metadata":{"execution":{"iopub.status.busy":"2023-12-17T05:44:31.753497Z","iopub.execute_input":"2023-12-17T05:44:31.753807Z","iopub.status.idle":"2023-12-17T05:44:31.765132Z","shell.execute_reply.started":"2023-12-17T05:44:31.753776Z","shell.execute_reply":"2023-12-17T05:44:31.764245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_weights('test_44.h5')  \ny_val_pred = model.predict(x_val)\n\noptR = OptimizedRounder()\noptR.fit(y_val_pred, y_val)\ncoefficients = optR.coefficients()\nprint(f'Coefficients: {coefficients}')\ny_val_pred = optR.predict(y_val_pred, coefficients)\n\nscore = cohen_kappa_score(y_val_pred, y_val, weights='quadratic')\n\nprint('Optimized Validation QWK score: {}'.format(score))\nprint('Not Optimized Validation QWK score: {}'.format(max(results.kappa)))","metadata":{"execution":{"iopub.status.busy":"2023-12-17T05:44:31.766200Z","iopub.execute_input":"2023-12-17T05:44:31.766511Z","iopub.status.idle":"2023-12-17T05:44:49.194724Z","shell.execute_reply.started":"2023-12-17T05:44:31.766456Z","shell.execute_reply":"2023-12-17T05:44:49.193511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('EfficientNetV2B3_Blanced_test_44.h5')","metadata":{"execution":{"iopub.status.busy":"2023-12-17T05:44:49.196096Z","iopub.execute_input":"2023-12-17T05:44:49.196676Z","iopub.status.idle":"2023-12-17T05:44:52.252145Z","shell.execute_reply.started":"2023-12-17T05:44:49.196641Z","shell.execute_reply":"2023-12-17T05:44:52.250993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!apt-get update && apt-get install -y python3-seaborn\n!pip install seaborn\n!apt-get update && apt-get install -y python3-statsmodel\n!pip install statsmodel","metadata":{"execution":{"iopub.status.busy":"2023-12-17T05:44:52.254197Z","iopub.execute_input":"2023-12-17T05:44:52.254474Z","iopub.status.idle":"2023-12-17T05:46:59.417516Z","shell.execute_reply.started":"2023-12-17T05:44:52.254447Z","shell.execute_reply":"2023-12-17T05:46:59.416353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install statsmodels","metadata":{"execution":{"iopub.status.busy":"2023-12-17T05:46:59.419039Z","iopub.execute_input":"2023-12-17T05:46:59.419354Z","iopub.status.idle":"2023-12-17T05:47:07.647854Z","shell.execute_reply.started":"2023-12-17T05:46:59.419320Z","shell.execute_reply":"2023-12-17T05:47:07.646830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, roc_curve, auc\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport statsmodels.api as sm\n\n# Load the model and predict on the validation data\nmodel.load_weights('test_44.h5')\ny_val_pred = model.predict(x_val)\n\n# Instantiate an OptimizedRounder object and fit on the validation data\noptR = OptimizedRounder()\noptR.fit(y_val_pred, y_val)\ncoefficients = optR.coefficients()\nprint(f'Coefficients: {coefficients}')\n\n# Get the predicted labels using the optimized coefficients\ny_val_pred_rounded = optR.predict(y_val_pred, coefficients)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T05:47:07.649219Z","iopub.execute_input":"2023-12-17T05:47:07.649499Z","iopub.status.idle":"2023-12-17T05:47:23.657962Z","shell.execute_reply.started":"2023-12-17T05:47:07.649470Z","shell.execute_reply":"2023-12-17T05:47:23.656687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute confusion matrix\ncm = confusion_matrix(y_val, y_val_pred_rounded)\n\n# Plot confusion matrix\nfig, ax = plt.subplots(figsize=(8, 6))\nsns.heatmap(cm, annot=True, cmap='Blues', fmt='g', ax=ax)\n\n# Add labels, title, and axis ticks\nax.set_xlabel('Predicted labels')\nax.set_ylabel('True labels')\nax.set_title('Confusion Matrix')\nax.xaxis.set_ticklabels(['No-DR', 'Mild NPDR', 'Moderate NPDR', 'Severe NPDR', 'PDR'])\nax.yaxis.set_ticklabels(['No-DR', 'Mild NPDR', 'Moderate NPDR', 'Severe NPDR', 'PDR'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T05:47:23.659443Z","iopub.execute_input":"2023-12-17T05:47:23.660493Z","iopub.status.idle":"2023-12-17T05:47:24.007520Z","shell.execute_reply.started":"2023-12-17T05:47:23.660441Z","shell.execute_reply":"2023-12-17T05:47:24.006656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2023-12-17T05:47:24.008534Z","iopub.execute_input":"2023-12-17T05:47:24.008813Z","iopub.status.idle":"2023-12-17T05:47:24.012871Z","shell.execute_reply.started":"2023-12-17T05:47:24.008774Z","shell.execute_reply":"2023-12-17T05:47:24.012089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy = accuracy_score(y_val, y_val_pred_rounded)\n\nprint('Accuracy:', accuracy)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T05:47:24.013891Z","iopub.execute_input":"2023-12-17T05:47:24.014148Z","iopub.status.idle":"2023-12-17T05:47:24.025961Z","shell.execute_reply.started":"2023-12-17T05:47:24.014120Z","shell.execute_reply":"2023-12-17T05:47:24.025175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report\n\n# assuming y_val and y_val_pred_rounded are the true and predicted labels respectively\ncm = confusion_matrix(y_val, y_val_pred_rounded)\n\n# Calculate precision, recall, and accuracy\nreport = classification_report(y_val, y_val_pred_rounded, target_names=['No-DR', 'Mild NPDR', 'Moderate NPDR', 'Severe NPDR', 'PDR'])\n\n# Print confusion matrix, precision, recall, and accuracy\nprint('Confusion Matrix:\\n', cm)\nprint('Classification Report:\\n', report)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T05:47:24.027053Z","iopub.execute_input":"2023-12-17T05:47:24.027335Z","iopub.status.idle":"2023-12-17T05:47:24.060731Z","shell.execute_reply.started":"2023-12-17T05:47:24.027304Z","shell.execute_reply":"2023-12-17T05:47:24.059966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Map 0 to 'No DR' and 1,2,3,4 to 'DR' in y_true\ny_true_binary = np.where(y_val == 0, 0, 1)\ny_val_pred_rounded_binary = np.where(y_val_pred_rounded == 0, 0, 1)\n\n# Compute ROC curve and AUC\nfpr, tpr, thresholds = roc_curve(y_true_binary, y_val_pred_rounded_binary, pos_label=1)\nroc_auc = auc(fpr, tpr)\n\nplt.plot(fpr, tpr, color='darkorange', lw=2, label='ROC curve (area = %0.2f)' % roc_auc)\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver operating characteristic curve')\nplt.legend(loc=\"lower right\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T05:47:24.061864Z","iopub.execute_input":"2023-12-17T05:47:24.062158Z","iopub.status.idle":"2023-12-17T05:47:24.228935Z","shell.execute_reply.started":"2023-12-17T05:47:24.062128Z","shell.execute_reply":"2023-12-17T05:47:24.228152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc = accuracy_score(y_true_binary, y_val_pred_rounded_binary)\n\nprint('Accuracy for binary classification:', acc)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T05:47:24.229926Z","iopub.execute_input":"2023-12-17T05:47:24.230198Z","iopub.status.idle":"2023-12-17T05:47:24.235941Z","shell.execute_reply.started":"2023-12-17T05:47:24.230169Z","shell.execute_reply":"2023-12-17T05:47:24.235211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute confusion matrix\ncm = confusion_matrix(y_true_binary, y_val_pred_rounded_binary)\n\n# Calculate specificity and sensitivity\ntn, fp, fn, tp = cm.ravel()\nspecificity = tn / (tn + fp)\nsensitivity = tp / (tp + fn)\n\n# Display confusion matrix with percentages\nfig, ax = plt.subplots(figsize=(8, 6))\nsns.heatmap(cm, annot=True, cmap='Blues', fmt='g', ax=ax)\n\n# Add labels, title, and axis ticks\nax.set_xlabel('Predicted labels')\nax.set_ylabel('True labels')\nax.set_title('Confusion Matrix (Specificity={:.2f}, Sensitivity={:.2f})'.format(specificity, sensitivity))\nax.xaxis.set_ticklabels(['No-DR', 'DR'])\nax.yaxis.set_ticklabels(['No-DR', 'DR'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T05:47:24.236951Z","iopub.execute_input":"2023-12-17T05:47:24.237255Z","iopub.status.idle":"2023-12-17T05:47:24.436401Z","shell.execute_reply.started":"2023-12-17T05:47:24.237226Z","shell.execute_reply":"2023-12-17T05:47:24.435279Z"},"trusted":true},"execution_count":null,"outputs":[]}]}