{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"},{"sourceId":418031,"sourceType":"datasetVersion","datasetId":131128},{"sourceId":2812287,"sourceType":"datasetVersion","datasetId":1719146},{"sourceId":2819730,"sourceType":"datasetVersion","datasetId":1723812},{"sourceId":2823796,"sourceType":"datasetVersion","datasetId":1727005},{"sourceId":5579822,"sourceType":"datasetVersion","datasetId":3211385},{"sourceId":5579826,"sourceType":"datasetVersion","datasetId":3211388}],"dockerImageVersionId":30628,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# To have reproducible results and compare them\nnr_seed = 2023\nimport numpy as np \nnp.random.seed(nr_seed)\nimport tensorflow as tf\nfrom tensorflow import keras\ntf.random.set_seed(nr_seed)\nprint(tf.__version__)\n\n# Libraries\nimport json\nimport math\nimport os\nimport cv2\nimport PIL\nimport pathlib\n\n\nimport scipy as sp\nfrom functools import partial\nfrom collections import Counter\nimport json\n\nfrom PIL import Image\nimport numpy as np\nfrom keras import backend as K\nfrom keras import layers\nfrom tensorflow.keras.applications import EfficientNetV2B3\nfrom keras.callbacks import Callback, ModelCheckpoint\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential\nfrom keras.optimizers import Adam\nfrom tensorflow.keras.utils import plot_model\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport gc\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n%matplotlib inline\n!pip install --upgrade tensorflow\n\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Flatten, Dense\nfrom tensorflow.keras.preprocessing import image_dataset_from_directory\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n# from tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\n\n\nfrom IPython.display import clear_output\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-02T16:29:38.761282Z","iopub.execute_input":"2024-01-02T16:29:38.761885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install scikit-learn","metadata":{"execution":{"iopub.status.busy":"2024-01-02T15:48:34.994117Z","iopub.execute_input":"2024-01-02T15:48:34.994991Z","iopub.status.idle":"2024-01-02T15:48:38.190413Z","shell.execute_reply.started":"2024-01-02T15:48:34.994955Z","shell.execute_reply":"2024-01-02T15:48:38.189311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import cohen_kappa_score","metadata":{"execution":{"iopub.status.busy":"2024-01-02T15:48:38.191629Z","iopub.execute_input":"2024-01-02T15:48:38.191892Z","iopub.status.idle":"2024-01-02T15:48:38.65484Z","shell.execute_reply.started":"2024-01-02T15:48:38.191866Z","shell.execute_reply":"2024-01-02T15:48:38.654137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/train-df/train_df_withoutINDEX.csv')\nval_df = pd.read_csv('../input/val-df/val_df_withoutINDEX.csv')\n\n\n\nprint(train_df.shape)\nprint(val_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-01-02T15:48:38.656505Z","iopub.execute_input":"2024-01-02T15:48:38.657174Z","iopub.status.idle":"2024-01-02T15:48:38.857396Z","shell.execute_reply.started":"2024-01-02T15:48:38.657146Z","shell.execute_reply":"2024-01-02T15:48:38.85674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# detect and init the TPU\n#tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\n\n# instantiate a distribution strategy\n#tpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)\n\ntpu = tf.distribute.cluster_resolver.TPUClusterResolver()\ntf.config.experimental_connect_to_cluster(tpu)\ntf.tpu.experimental.initialize_tpu_system(tpu)\ntf.config.set_soft_device_placement(True)\nstrategy = tf.distribute.experimental.TPUStrategy(tpu)\n","metadata":{"execution":{"iopub.status.busy":"2024-01-02T15:48:38.85838Z","iopub.execute_input":"2024-01-02T15:48:38.858694Z","iopub.status.idle":"2024-01-02T15:48:47.027006Z","shell.execute_reply.started":"2024-01-02T15:48:38.858661Z","shell.execute_reply":"2024-01-02T15:48:47.026302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image size\nim_size = 224\n# Batch size\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync","metadata":{"execution":{"iopub.status.busy":"2024-01-02T15:48:47.027958Z","iopub.execute_input":"2024-01-02T15:48:47.028194Z","iopub.status.idle":"2024-01-02T15:48:47.031611Z","shell.execute_reply.started":"2024-01-02T15:48:47.028168Z","shell.execute_reply":"2024-01-02T15:48:47.030957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2024-01-02T15:48:47.032493Z","iopub.execute_input":"2024-01-02T15:48:47.032749Z","iopub.status.idle":"2024-01-02T15:48:47.047849Z","shell.execute_reply.started":"2024-01-02T15:48:47.032723Z","shell.execute_reply":"2024-01-02T15:48:47.047179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/train-df/train_df_withoutINDEX.csv')\nval_df = pd.read_csv('../input/val-df/val_df_withoutINDEX.csv')\n\n\n\nprint(train_df.shape)\nprint(val_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-01-02T15:48:47.04885Z","iopub.execute_input":"2024-01-02T15:48:47.049104Z","iopub.status.idle":"2024-01-02T15:48:47.119902Z","shell.execute_reply.started":"2024-01-02T15:48:47.049078Z","shell.execute_reply":"2024-01-02T15:48:47.119064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!apt-get update && apt-get install -y python3-opencv\n!pip install opencv-python\n!apt-get update && apt-get install -y python3-tqdm\n!pip install tqdm","metadata":{"execution":{"iopub.status.busy":"2024-01-02T15:48:47.120938Z","iopub.execute_input":"2024-01-02T15:48:47.121206Z","iopub.status.idle":"2024-01-02T15:50:19.96164Z","shell.execute_reply.started":"2024-01-02T15:48:47.121179Z","shell.execute_reply":"2024-01-02T15:50:19.960313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport tqdm","metadata":{"execution":{"iopub.status.busy":"2024-01-02T15:50:19.965661Z","iopub.execute_input":"2024-01-02T15:50:19.965956Z","iopub.status.idle":"2024-01-02T15:50:19.984149Z","shell.execute_reply.started":"2024-01-02T15:50:19.965927Z","shell.execute_reply":"2024-01-02T15:50:19.983355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_samples(df, columns=4, rows=3):\n    fig=plt.figure(figsize=(5*columns, 4*rows))\n\n    for i in range(columns*rows):\n        image_path = df.loc[i,'id_code']\n        image_id = df.loc[i,'diagnosis']\n        img = cv2.imread(f'{image_path}')\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        #img = crop_image_from_gray(img)\n        img = cv2.resize(img, (im_size,im_size))\n        img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), im_size/40) ,-4 ,128)\n        \n        fig.add_subplot(rows, columns, i+1)\n        plt.title(image_id)\n        plt.imshow(img)\n    \n    plt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2024-01-02T15:50:19.985038Z","iopub.execute_input":"2024-01-02T15:50:19.985292Z","iopub.status.idle":"2024-01-02T15:50:19.991867Z","shell.execute_reply.started":"2024-01-02T15:50:19.985266Z","shell.execute_reply":"2024-01-02T15:50:19.991148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display train images\ndisplay_samples(train_df)","metadata":{"execution":{"iopub.status.busy":"2024-01-02T15:50:19.992856Z","iopub.execute_input":"2024-01-02T15:50:19.993115Z","iopub.status.idle":"2024-01-02T15:50:23.162057Z","shell.execute_reply.started":"2024-01-02T15:50:19.993088Z","shell.execute_reply":"2024-01-02T15:50:23.160949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display train images\ndisplay_samples(val_df)","metadata":{"execution":{"iopub.status.busy":"2024-01-02T15:50:23.164077Z","iopub.execute_input":"2024-01-02T15:50:23.164402Z","iopub.status.idle":"2024-01-02T15:50:27.964146Z","shell.execute_reply.started":"2024-01-02T15:50:23.164369Z","shell.execute_reply":"2024-01-02T15:50:27.962978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = BATCH_SIZE*2","metadata":{"execution":{"iopub.status.busy":"2024-01-02T15:50:27.965575Z","iopub.execute_input":"2024-01-02T15:50:27.965869Z","iopub.status.idle":"2024-01-02T15:50:27.970516Z","shell.execute_reply.started":"2024-01-02T15:50:27.96584Z","shell.execute_reply":"2024-01-02T15:50:27.969637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2024-01-02T15:50:27.971506Z","iopub.execute_input":"2024-01-02T15:50:27.971759Z","iopub.status.idle":"2024-01-02T15:50:27.99568Z","shell.execute_reply.started":"2024-01-02T15:50:27.971732Z","shell.execute_reply":"2024-01-02T15:50:27.994791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2024-01-02T15:50:27.996806Z","iopub.execute_input":"2024-01-02T15:50:27.997075Z","iopub.status.idle":"2024-01-02T15:50:28.010927Z","shell.execute_reply.started":"2024-01-02T15:50:27.997049Z","shell.execute_reply":"2024-01-02T15:50:28.010031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def crop_image_from_gray(img,tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n            img = np.stack([img1,img2,img3],axis=-1)\n        return img\n\ndef preprocess_image(image_path, desired_size=224):\n    img = cv2.imread(image_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = crop_image_from_gray(img)\n    img = cv2.resize(img, (desired_size,desired_size))\n    img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), desired_size/30) ,-4 ,128)\n    return img\n\ndef preprocess_image_old(image_path, desired_size=224):\n    img = cv2.imread(image_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    #img = crop_image_from_gray(img)\n    img = cv2.resize(img, (desired_size,desired_size))\n    img = cv2.addWeighted(img,4,cv2.GaussianBlur(img, (0,0), desired_size/40) ,-4 ,128)\n    return img","metadata":{"execution":{"iopub.status.busy":"2024-01-02T15:50:28.012019Z","iopub.execute_input":"2024-01-02T15:50:28.012268Z","iopub.status.idle":"2024-01-02T15:50:28.029674Z","shell.execute_reply.started":"2024-01-02T15:50:28.012243Z","shell.execute_reply":"2024-01-02T15:50:28.028731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# validation settebook\nN = val_df.shape[0]\nx_val = np.empty((N, im_size, im_size, 3), dtype=np.uint8)\n\nfor i, image_id in tqdm(enumerate(val_df['id_code']), total=N):\n    x_val[i, :, :, :] = preprocess_image(\n        f'{image_id}',\n        desired_size=im_size\n    )","metadata":{"execution":{"iopub.status.busy":"2024-01-02T15:50:28.030904Z","iopub.execute_input":"2024-01-02T15:50:28.031561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.applications.resnet50 import preprocess_input","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from tensorflow import keras\n\n#callbacks = [\n #   keras.callbacks.EarlyStopping(\n #       # Stop training when `val_loss` is no longer improving\n #       monitor='val_loss',\n #       # \"no longer improving\" being defined as \"no better than 1e-2 less\"\n #       min_delta=1e-3,\n #       # \"no longer improving\" being further defined as \"for at least 2 epochs\"\n #       patience=25,\n #       verbose=1)\n#]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = train_df['diagnosis'].values\ny_val = val_df['diagnosis'].values\n\nprint(y_train.shape)\nprint(x_val.shape)\nprint(y_val.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Metrics(Callback):\n    def __init__(self, validation_data=()):\n        super().__init__()\n        self.validation_data = validation_data\n\n    def on_epoch_end(self, epoch, logs=None):\n        x_val, y_val = self.validation_data[0], self.validation_data[1]\n        \n        y_pred = self.model.predict(x_val)\n        \n        coef = [0.5, 1.5, 2.5, 3.5]\n\n        for i, pred in enumerate(y_pred):\n            if pred < coef[0]:\n                y_pred[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                y_pred[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                y_pred[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                y_pred[i] = 3\n            else:\n                y_pred[i] = 4\n\n        _val_kappa = cohen_kappa_score(\n            y_val,\n            y_pred, \n            weights='quadratic')\n        self.val_kappas.append(_val_kappa)\n        print(f\"val_kappa: {_val_kappa:.4f}\")\n\n\n        if _val_kappa == max(self.val_kappas):\n            print(\"Validation Kappa has improved. Saving model.\")\n            self.model.save('testtest44.h5')\n\n        return","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_datagen():\n    return ImageDataGenerator(\n        horizontal_flip = True,\n        vertical_flip = True,\n        rotation_range = 160,\n        zoom_range=0.35\n    )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --upgrade tensorflow","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import tensorflow as tf\n#tf.compat.v1.disable_eager_execution()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#data_augmentation = keras.Sequential(\n#    [\n#        layers.experimental.preprocessing.Resizing(height=256, width = 256),\n#        layers.experimental.preprocessing.RandomFlip(mode=\"horizontal\"),\n#        layers.experimental.preprocessing.RandomContrast(factor=0.1)\n#    ])\n#data_augmentation = keras.Sequential(\n#    [\n#        layers.experimental.preprocessing.Resizing(height=im_size, width=im_size),\n#        layers.experimental.preprocessing.Rescaling(1./255),  # Assuming pixel values are in the range [0, 255]\n#        layers.experimental.preprocessing.RandomFlip(mode=\"horizontal\"),\n#        layers.experimental.preprocessing.RandomBrightnessContrast(factor=0.1)\n#    ])\n\n#import tensorflow as tf\n\n#import pandas as pd\n\n# Assuming 'train_df' is your DataFrame\n#train_df = pd.read_csv('../input/train-df/train_df_withoutINDEX.csv')\n\n# Define your augment function\n#def augment(image,label):\n #   new_height = new_width = 256\n  #  image = tf.image.resize(image, (new_height, new_width))\n    \n   # if tf.random.uniform((), minval = 0, maxval=1)<0.1:\n    #    image = tf.tile(tf.image.rgb_to_grayscale(image),[1,1,3])\n        \n    #image = tf.image.random_brightness(image,max_delta=0.1)\n    #image = tf.image.random_contrast(image, lower=0.1,upper=0.2)\n    \n    #image = tf.image.random_flip_left_right(image)\n    \n    #return image,label\n# Apply the augment function to each row\n#train_df = train_df.apply(augment, axis=1)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#def build_model():\n #   effnet = EfficientNetV2B3(\n  #  input_shape=(im_size,im_size,3),\n  #  weights='imagenet',\n  #  include_top=False)\n  #  model = Sequential()\n  #  model.add(effnet)\n  #  model.add(layers.Dropout(0.25))\n  #  model.add(layers.Dense(2048))\n  #  model.add(layers.LeakyReLU())\n  #  model.add(layers.GlobalAveragePooling2D())\n  # model.add(layers.Dropout(0.5))\n  #  model.add(layers.Dense(1, activation='linear'))\n  #  model.compile(loss='categorical_crossentropy', optimizer='adam', metrics=['accuracy'])\n  #  model.add(data_augmentation)\n    \n  #  return model\n\n#def build_model():\n#    effnet = EfficientNetV2B3(\n#        input_shape=(im_size, im_size, 3),  # Assuming RGB images\n#        weights='imagenet',\n#        include_top=False)\n    \n#    model = keras.Sequential(effnet)\n#    #model.add(effnet)\n#    model.add(data_augmentation)  # Add data augmentation as the first layer\n#    model.add(layers.Dropout(0.25))\n#    model.add(layers.Dense(2048))\n#    model.add(layers.LeakyReLU())\n#    model.add(layers.GlobalAveragePooling2D())\n#    model.add(layers.Dropout(0.5))\n#   model.add(layers.Dense(1, activation='linear'))\n#    model.compile(loss='categorical_crossentropy', optimizer='adam', metrics=['accuracy'])\n    \n#   return model","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflow","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_augmentation = keras.Sequential([\n    layers.experimental.preprocessing.Resizing(height=im_size, width=im_size),\n    layers.experimental.preprocessing.Rescaling(1./255),\n    layers.experimental.preprocessing.RandomFlip(mode=\"horizontal\"),\n    layers.experimental.preprocessing.RandomContrast(factor=0.1),\n    layers.experimental.preprocessing.RandomZoom(0.3)\n    \n])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#num_classes = 5\n\n#model = Sequential([\n#    data_augmentation,\n#    layers.Conv2D(16,3,padding=\"same\",activation=\"relu\"),\n#    layers.GlobalMaxPool2D(),\n#    layers.Conv2D(32,3,padding=\"same\",activation=\"relu\"),\n#    layers.GlobalMaxPool2D(),\n#    layers.Conv2D(64,3,padding=\"same\",activation=\"relu\"),\n#    layers.GlobalMaxPool2D(),\n#    layers.Dropout(0.2),\n#    layers.Flatten(),\n#    layers.Dense(128, activation=\"relu\"),\n#    layers.Dense(num_classes)\n#])\n\n#model.compile(\n#        loss='mean_squared_error',\n#        optimizer=tf.keras.optimizers.Adam(learning_rate=0.0001),\n#        metrics=['mae']\n#    )\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#with strategy.scope():\n#    model = build_model()\n#    model.compile(\n#        loss='mean_squared_error',\n#        #optimizer=Adam(lr=0.001,decay=1e-6),\n#        optimizer = tf.keras.optimizers.Adam(learning_rate=0.0001),\n#        metrics=['mae']\n#    )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras import layers\nfrom tensorflow.keras.applications import EfficientNetV2B3\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.metrics import MeanSquaredError, MeanAbsoluteError\nfrom tensorflow.keras.models import Model\nimport tensorflow as tf\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nfrom tensorflow.keras.applications.efficientnet_v2 import preprocess_input\nimport gc\n\n# Constants\n#im_size = 224\n#BATCH_SIZE = 256\n\n# Data Augmentation Layer\n#data_augmentation = keras.Sequential(\n#    [\n#        layers.experimental.preprocessing.Resizing(height=im_size, width=im_size),  # Update height and width to 256\n#        layers.experimental.preprocessing.Rescaling(1./255),\n#        layers.experimental.preprocessing.RandomFlip(mode=\"horizontal\"),\n#        layers.experimental.preprocessing.RandomBrightnessContrast(factor=0.1)\n#    ]\n#)\n\n# Function to build the model\n#def build_model():\n#    effnet = EfficientNetV2B3(\n#        input_shape=(224, 224, 3),\n#        weights='imagenet',\n#        include_top=False\n#    )\n\n    # Add data augmentation as the first layer\n#    inputs = tf.keras.Input(shape=(im_size, im_size, 3))\n#    x = data_augmentation(inputs)\n#    x = effnet(x, training=False)\n#    x = layers.Dropout(0.25)(x)\n#    x = layers.Dense(2048)(x)\n#    x = layers.LeakyReLU()(x)\n#    x = layers.GlobalAveragePooling2D()(x)\n#    x = layers.Dropout(0.5)(x)\n#    outputs = layers.Dense(1, activation='linear')(x)\n\n#    model = Model(inputs, outputs)\n\n#    return model\n\n# Compile the model\n#with tf.device('/device:GPU:0'):  # Change GPU:0 to GPU:1 if you have multiple GPUs\n#    model = build_model()\n#    model.compile(\n#        loss='mean_squared_error',\n#        optimizer=Adam(learning_rate=0.0001),\n#        metrics=[MeanAbsoluteError()]\n#    )\n\n# Model summary\n#model.summary()\n\ndef build_model(with_augmentation=True):\n    effnet = EfficientNetV2B3(\n        input_shape=(im_size, im_size, 3),\n        weights='imagenet',\n        include_top=False\n    )\n    \n    # Create a model for training with data augmentation\n    if with_augmentation:\n        data_augmentation = keras.Sequential([\n            layers.experimental.preprocessing.Resizing(height=im_size, width=im_size),\n            layers.experimental.preprocessing.Rescaling(1./255),\n            layers.experimental.preprocessing.RandomFlip(mode=\"horizontal\"),\n            layers.experimental.preprocessing.RandomContrast(factor=0.1),\n            layers.experimental.preprocessing.RandomZoom(0.3)\n        ])\n        model = Sequential([data_augmentation])\n        model.add(layers.Dropout(0.25))\n        model.add(layers.Dense(2048))\n        model.add(layers.LeakyReLU())\n        model.add(layers.GlobalAveragePooling2D())\n        model.add(layers.Dropout(0.5))\n        model.add(layers.Dense(1, activation='linear'))\n    else:\n    # Create a model for inference without data augmentation\n        model = Sequential()\n        model.add(layers.Dropout(0.25))\n        model.add(layers.Dense(2048))\n        model.add(layers.LeakyReLU())\n        model.add(layers.GlobalAveragePooling2D())\n        model.add(layers.Dropout(0.5))\n        model.add(layers.Dense(1, activation='linear'))\n    \n# Compile the model\n    model.compile(\n        loss='mean_squared_error',\n        optimizer=tf.keras.optimizers.Adam(learning_rate=0.0001),\n        metrics=['mae']\n    )\n    \n    return model\n\n# Build the model for training with data augmentation\nwith strategy.scope():\n    model_with_augmentation = build_model(with_augmentation=True)\n    \n    # Create a dummy input tensor to build the model\n    im_size = 224  # Set the image size\n    dummy_input = tf.keras.layers.Input(shape=(im_size, im_size, 3))\n    _ = model_with_augmentation(dummy_input)\n    \n# Build the model for inference without data augmentation\nmodel_without_augmentation = build_model(with_augmentation=False)\n\n# Display the summary of the model with data augmentation\nmodel_with_augmentation.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import matplotlib.pyplot as plt\n#import numpy as np\n#from tensorflow.keras.preprocessing import image\n\n# Sample image for visualization\n#sample_image_path = '../input/diabetic-retinopathy-resized/resized_train/resized_train/10_left.jpeg'  # Replace with the path to your sample image\n#img = image.load_img(sample_image_path, target_size=(256, 256))\n#original_image = image.img_to_array(img)\n\n# Function to apply data augmentation and plot images\n#def visualize_augmentation(data_augmentation, original_image):\n #   augmented_images = [data_augmentation(image=np.expand_dims(original_image, 0))['image'][0] for _ in range(5)]\n\n #   plt.figure(figsize=(12, 4))\n #   plt.subplot(1, 6, 1)\n #   plt.imshow(original_image.astype('uint8'))\n #   plt.title('Original')\n\n #   for i, augmented_image in enumerate(augmented_images):\n #       plt.subplot(1, 6, i + 2)\n #       plt.imshow(augmented_image.astype('uint8'))\n #       plt.title(f'Augmented {i+1}')\n\n #   plt.show()\n\n# Visualize data augmentation\n#visualize_augmentation(data_augmentation, original_image)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_bucket = 4\ndiv = round(train_df.shape[0]/num_bucket)\ndiv","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = pd.DataFrame({\n    'val_loss': [0.0],\n    'val_mean_absolute_error': [0.0],\n    'loss': [0.0], \n    'mean_absolute_error': [0.0],\n    'bucket': [0.0]\n})","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Epochs\nepochs = [4, 4, 4, 4]\nkappa_metrics = Metrics(validation_data=(x_val, y_val))\nkappa_metrics.val_kappas = []\n\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print(type(data_generator))\n#sample_batch = next(data_generator)\n#print(type(sample_batch))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(0, num_bucket):\n    if i != (num_bucket-1):\n        print(\"Bucket Nr: {}\".format(i))\n\n        N = train_df.iloc[i*div:(1+i)*div].shape[0]\n        x_train = np.empty((N, im_size, im_size, 3), dtype=np.uint8)\n        for j, image_id in enumerate(tqdm(train_df.iloc[i*div:(1+i)*div, 0])):\n            x_train[j, :, :, :] = preprocess_image(f'{image_id}', desired_size=im_size)\n        \n        with tf.device('/device:CPU:0'):\n            augmented_images = data_augmentation(x_train)\n\n        # Use the model with data augmentation for training\n        history = model_with_augmentation.fit(\n            augmented_images,\n            y_train[i*div:(1+i)*div],\n            batch_size=BATCH_SIZE,\n            epochs=epochs[i],\n            validation_data=(x_val, y_val),\n            callbacks=[kappa_metrics]\n        )\n    else:\n        print(\"Bucket Nr: {}\".format(i))\n        N = train_df.iloc[i*div:].shape[0]\n        x_train = np.empty((N, im_size, im_size, 3), dtype=np.uint8)\n        for j, image_id in enumerate(tqdm(train_df.iloc[i*div:, 0])):\n            x_train[j, :, :, :] = preprocess_image(f'{image_id}', desired_size=im_size)\n        \n        # Use the model with data augmentation for training\n        history = model_with_augmentation.fit(\n            x_train,\n            y_train[i*div:],\n            batch_size=BATCH_SIZE,\n            epochs=epochs[i],\n            validation_data=(x_val, y_val),\n            callbacks=[kappa_metrics]\n        )\n\n    dic = history.history\n    df_model = pd.DataFrame(dic)\n    df_model['bucket'] = i\n\n    results = pd.concat([results, df_model], ignore_index=True)\n    del x_train\n    gc.collect()\n    print('-'*40)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = results.iloc[1:]\nkappa_vals = kappa_metrics.val_kappas[:len(results)]\nresults['kappa'] = kappa_vals\nresults = results.reset_index(drop=True)\nresults = results.rename(index=str, columns={\"index\": \"epoch\"})\nprint(max(results.kappa))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results[['loss', 'val_loss']].plot()\nresults[['mae', 'val_mae']].plot()\nresults[['kappa']].plot()\nresults.to_csv('model_results.csv',index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class OptimizedRounder(object):\n    def __init__(self):\n        self.coef_ = 0\n\n    def _kappa_loss(self, coef, X, y):\n        X_p = np.copy(X)\n        for i, pred in enumerate(X_p):\n            if pred < coef[0]:\n                X_p[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                X_p[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                X_p[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                X_p[i] = 3\n            else:\n                X_p[i] = 4\n\n        ll = cohen_kappa_score(y, X_p, weights='quadratic')\n        return -ll\n\n    def fit(self, X, y):\n        loss_partial = partial(self._kappa_loss, X=X, y=y)\n        initial_coef = [0.5, 1.5, 2.5, 3.5]\n        self.coef_ = sp.optimize.minimize(loss_partial, initial_coef, method='nelder-mead')\n        print(-loss_partial(self.coef_['x']))\n\n    def predict(self, X, coef):\n        X_p = np.copy(X)\n        for i, pred in enumerate(X_p):\n            if pred < coef[0]:\n                 X_p[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                X_p[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                X_p[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                X_p[i] = 3\n            else:\n                X_p[i] = 4\n        return X_p\n\n    def coefficients(self):\n        return self.coef_['x']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_with_augmentation.load_weights('testtest44.h5')  \ny_val_pred = model_with_augmentation.predict(x_val)\n\noptR = OptimizedRounder()\noptR.fit(y_val_pred, y_val)\ncoefficients = optR.coefficients()\nprint(f'Coefficients: {coefficients}')\ny_val_pred = optR.predict(y_val_pred, coefficients)\n\nscore = cohen_kappa_score(y_val_pred, y_val, weights='quadratic')\n\nprint('Optimized Validation QWK score: {}'.format(score))\nprint('Not Optimized Validation QWK score: {}'.format(max(results.kappa)))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('EfficientNetV2B3_Blanced_testtest44.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!apt-get update && apt-get install -y python3-seaborn\n!pip install seaborn\n!apt-get update && apt-get install -y python3-statsmodel\n!pip install statsmodel","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install statsmodels","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, roc_curve, auc\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport statsmodels.api as sm\n\n# Load the model and predict on the validation data\nmodel_with_augmentation.load_weights('testnew204.h5')\ny_val_pred = model_with_augmentation.predict(x_val)\n\n# Instantiate an OptimizedRounder object and fit on the validation data\noptR = OptimizedRounder()\noptR.fit(y_val_pred, y_val)\ncoefficients = optR.coefficients()\nprint(f'Coefficients: {coefficients}')\n\n# Get the predicted labels using the optimized coefficients\ny_val_pred_rounded = optR.predict(y_val_pred, coefficients)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute confusion matrix\ncm = confusion_matrix(y_val, y_val_pred_rounded)\n\n# Plot confusion matrix\nfig, ax = plt.subplots(figsize=(8, 6))\nsns.heatmap(cm, annot=True, cmap='Blues', fmt='g', ax=ax)\n\n# Add labels, title, and axis ticks\nax.set_xlabel('Predicted labels')\nax.set_ylabel('True labels')\nax.set_title('Confusion Matrix')\nax.xaxis.set_ticklabels(['No-DR', 'Mild NPDR', 'Moderate NPDR', 'Severe NPDR', 'PDR'])\nax.yaxis.set_ticklabels(['No-DR', 'Mild NPDR', 'Moderate NPDR', 'Severe NPDR', 'PDR'])\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy = accuracy_score(y_val, y_val_pred_rounded)\n\nprint('Accuracy:', accuracy)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_with_augmentation.diagnosis.value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}