{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"},{"sourceId":9849825,"sourceType":"datasetVersion","datasetId":6043756},{"sourceId":144039,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":122067,"modelId":145167},{"sourceId":145019,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":122924,"modelId":145987}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import Libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport shap\nimport random\nimport warnings\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.utils import class_weight\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, cohen_kappa_score\nfrom keras.models import Model\nfrom keras import optimizers, applications\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom keras.callbacks import EarlyStopping, ReduceLROnPlateau\nfrom keras.layers import Dense, Dropout, GlobalAveragePooling2D, Input\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras import layers, models, optimizers, regularizers\n\n%matplotlib inline\nsns.set(style=\"whitegrid\")\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-17T04:26:30.517479Z","iopub.execute_input":"2024-11-17T04:26:30.517855Z","iopub.status.idle":"2024-11-17T04:26:30.529498Z","shell.execute_reply.started":"2024-11-17T04:26:30.517820Z","shell.execute_reply":"2024-11-17T04:26:30.528522Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"shape_of_images = pd.read_csv('/kaggle/input/shape-and-size-of-the-data/image_shapes.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T04:26:44.183774Z","iopub.execute_input":"2024-11-17T04:26:44.184221Z","iopub.status.idle":"2024-11-17T04:26:44.195436Z","shell.execute_reply.started":"2024-11-17T04:26:44.184149Z","shell.execute_reply":"2024-11-17T04:26:44.194142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"shape_of_images#check the size of images dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T04:26:48.461053Z","iopub.execute_input":"2024-11-17T04:26:48.461896Z","iopub.status.idle":"2024-11-17T04:26:48.473382Z","shell.execute_reply.started":"2024-11-17T04:26:48.461856Z","shell.execute_reply":"2024-11-17T04:26:48.472518Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load the data","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/train.csv')\ntest = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/test.csv')\nprint('Number of train samples: ', train.shape[0])\nprint('Number of test samples: ', test.shape[0])\n\n# Preprocecss data\ntrain[\"id_code\"] = train[\"id_code\"].apply(lambda x: x + \".png\")\ntest[\"id_code\"] = test[\"id_code\"].apply(lambda x: x + \".png\")\ntrain['diagnosis'] = train['diagnosis'].astype('str')\ndisplay(train.head())","metadata":{"execution":{"iopub.status.busy":"2024-11-17T04:27:14.724397Z","iopub.execute_input":"2024-11-17T04:27:14.725042Z","iopub.status.idle":"2024-11-17T04:27:14.751380Z","shell.execute_reply.started":"2024-11-17T04:27:14.725004Z","shell.execute_reply":"2024-11-17T04:27:14.750373Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Parameters","metadata":{}},{"cell_type":"code","source":"# Model parameters\nBATCH_SIZE = 8\nEPOCHS = 40\nWARMUP_EPOCHS = 10\nLEARNING_RATE = 1e-4\nWARMUP_LEARNING_RATE = 1e-3\nHEIGHT = 400\nWIDTH = 400\nCANAL = 3\nN_CLASSES = train['diagnosis'].nunique()\nES_PATIENCE = 5\nRLROP_PATIENCE = 3\nDECAY_DROP = 0.5","metadata":{"execution":{"iopub.status.busy":"2024-11-17T04:27:30.398828Z","iopub.execute_input":"2024-11-17T04:27:30.399229Z","iopub.status.idle":"2024-11-17T04:27:30.405547Z","shell.execute_reply.started":"2024-11-17T04:27:30.399192Z","shell.execute_reply":"2024-11-17T04:27:30.404557Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Image Preprocessing","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport cv2\n\ndef preprocess_image(image):\n    #Custom preprocessing function for image generators.\n    #Applies Gaussian blur and resizes the image to 400x400 pixels.\n    # Convert image to numpy array (if it's a TensorFlow tensor)\n    if isinstance(image, tf.Tensor):\n        image = image.numpy()\n\n    # Apply Gaussian blur (kernel size of 5x5)\n    image = cv2.GaussianBlur(image, (5, 5), 0)\n\n    # Resize the image to 400x400\n    image = cv2.resize(image, (400, 400))\n\n    # Normalize the image to the range [0, 1]\n    image = image / 255.0\n    \n    return image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T04:28:03.086291Z","iopub.execute_input":"2024-11-17T04:28:03.086670Z","iopub.status.idle":"2024-11-17T04:28:03.092946Z","shell.execute_reply.started":"2024-11-17T04:28:03.086636Z","shell.execute_reply":"2024-11-17T04:28:03.091904Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train test split","metadata":{}},{"cell_type":"code","source":"#X_train, X_val = train_test_split(train, test_size=0.2, random_state=123)\nX_train, X_test = train_test_split(train, test_size=0.2, random_state=123)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T04:28:06.907969Z","iopub.execute_input":"2024-11-17T04:28:06.908718Z","iopub.status.idle":"2024-11-17T04:28:06.916140Z","shell.execute_reply.started":"2024-11-17T04:28:06.908679Z","shell.execute_reply":"2024-11-17T04:28:06.915337Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Generator","metadata":{}},{"cell_type":"code","source":"train_datagen=ImageDataGenerator( \n                                 rotation_range=360,\n                                 horizontal_flip=True,\n                                 vertical_flip=True,\n                                 validation_split=0.1,\n                                 preprocessing_function=preprocess_image)\n\ntrain_generator=train_datagen.flow_from_dataframe(\n    dataframe=X_train,\n    directory=\"/kaggle/input/aptos2019-blindness-detection/train_images\",\n    x_col=\"id_code\",\n    y_col=\"diagnosis\",\n    class_mode=\"categorical\",\n    batch_size=BATCH_SIZE,\n    target_size=(HEIGHT,WIDTH),\n    seed=0,\n    subset='training')\n\nvalidation_datagen = ImageDataGenerator(preprocessing_function=preprocess_image)\n\nvalid_generator=train_datagen.flow_from_dataframe(\n    dataframe=X_train,\n    directory=\"/kaggle/input/aptos2019-blindness-detection/train_images\",\n    x_col=\"id_code\",\n    y_col=\"diagnosis\",\n    class_mode=\"categorical\", \n    batch_size=BATCH_SIZE,   \n    target_size=(HEIGHT,WIDTH),\n    seed=0,\n    subset= 'validation')\n\ntest_datagen = ImageDataGenerator(preprocessing_function=preprocess_image)\n\ntest_generator = test_datagen.flow_from_dataframe(  \n        dataframe=X_test,\n        directory = \"/kaggle/input/aptos2019-blindness-detection/train_images\",\n        x_col=\"id_code\",\n        y_col=\"diagnosis\",\n        batch_size=1,\n        class_mode=\"categorical\",\n        shuffle=False,\n        target_size=(HEIGHT, WIDTH),\n        seed=0)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T04:28:15.845369Z","iopub.execute_input":"2024-11-17T04:28:15.845732Z","iopub.status.idle":"2024-11-17T04:28:17.506668Z","shell.execute_reply.started":"2024-11-17T04:28:15.845700Z","shell.execute_reply":"2024-11-17T04:28:17.505730Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get a batch of images from the training generator\nimages, labels = next(train_generator)\n\n# Plot the images\nplt.figure(figsize=(12, 12))\nfor i in range(len(images)):\n    plt.subplot(4, 4, i + 1)\n    plt.imshow(images[i])\n    plt.title(f\"Label: {labels[i]}\")\n    plt.axis('off')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-17T04:28:23.395833Z","iopub.execute_input":"2024-11-17T04:28:23.396687Z","iopub.status.idle":"2024-11-17T04:28:26.061625Z","shell.execute_reply.started":"2024-11-17T04:28:23.396646Z","shell.execute_reply":"2024-11-17T04:28:26.060653Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get a batch of images and labels from the train generator\nimages, labels = next(train_generator)\n\n# Print the shape of the images and labels batch\nprint(\"Shape of images:\", images.shape)\nprint(\"Shape of labels:\", labels.shape)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-17T04:28:32.202901Z","iopub.execute_input":"2024-11-17T04:28:32.203675Z","iopub.status.idle":"2024-11-17T04:28:33.331023Z","shell.execute_reply.started":"2024-11-17T04:28:32.203640Z","shell.execute_reply":"2024-11-17T04:28:33.330047Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"def create_model(input_shape, n_out):\n    input_tensor = Input(shape=input_shape)\n    \n    # Use EfficientNetB0 as the base model instead of ResNet50\n    base_model = EfficientNetB0(weights='imagenet', \n                                include_top=False,\n                                input_tensor=input_tensor)\n    \n    # Add global average pooling and dense layers\n    x = GlobalAveragePooling2D()(base_model.output)\n    x = Dropout(0.7)(x)\n    x = Dense(2048, activation='relu',kernel_regularizer=regularizers.l2(0.01))(x)  # Fully connected layer\n    x = Dropout(0.5)(x)\n    final_output = Dense(n_out, activation='softmax', name='final_output')(x)  # Output layer for classification\n    \n    # Create the model\n    model = Model(inputs=input_tensor, outputs=final_output)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2024-11-17T04:28:42.805207Z","iopub.execute_input":"2024-11-17T04:28:42.805944Z","iopub.status.idle":"2024-11-17T04:28:42.812664Z","shell.execute_reply.started":"2024-11-17T04:28:42.805904Z","shell.execute_reply":"2024-11-17T04:28:42.811804Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the model\nmodel = create_model(input_shape=(HEIGHT, WIDTH, CANAL), n_out=N_CLASSES)\n\n# Freeze all layers initially\nfor layer in model.layers:\n    layer.trainable = False\n\n# Unfreeze the last 5 layers for fine-tuning\nfor i in range(-5, 0):\n    model.layers[i].trainable = True\n\n# Compute class weights to handle imbalanced data\nclass_weights = class_weight.compute_class_weight(\n    class_weight='balanced', \n    classes=np.unique(train['diagnosis'].astype('int').values), \n    y=train['diagnosis'].astype('int').values\n)\n\n# Compile the model with Adam optimizer and categorical cross-entropy loss\nmetric_list = [\"accuracy\"]\noptimizer = optimizers.Adam(learning_rate=WARMUP_LEARNING_RATE)  # Ensure the correct learning rate is used\nmodel.compile(optimizer=optimizer, loss='categorical_crossentropy', metrics=metric_list)\n\n# Print the model summary\nmodel.summary()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-11-17T04:28:46.622721Z","iopub.execute_input":"2024-11-17T04:28:46.623162Z","iopub.status.idle":"2024-11-17T04:28:48.117287Z","shell.execute_reply.started":"2024-11-17T04:28:46.623109Z","shell.execute_reply":"2024-11-17T04:28:48.116329Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define steps per epoch\nSTEP_SIZE_TRAIN = train_generator.n // train_generator.batch_size\nSTEP_SIZE_VALID = valid_generator.n // valid_generator.batch_size","metadata":{"execution":{"iopub.status.busy":"2024-11-17T04:28:53.392475Z","iopub.execute_input":"2024-11-17T04:28:53.392860Z","iopub.status.idle":"2024-11-17T04:28:53.397606Z","shell.execute_reply.started":"2024-11-17T04:28:53.392824Z","shell.execute_reply":"2024-11-17T04:28:53.396544Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.utils import class_weight\n# Compute class weights\nclass_weights_array = class_weight.compute_class_weight(\n    class_weight='balanced',\n    classes=np.unique(train['diagnosis'].astype('int').values),\n    y=train['diagnosis'].astype('int').values\n)\n\n# Create a class weights dictionary\nclass_weights_dict = {i: class_weights_array[i] for i in range(len(class_weights_array))}\n\n# Define steps per epoch\nSTEP_SIZE_TRAIN = train_generator.n // train_generator.batch_size\nSTEP_SIZE_VALID = valid_generator.n // valid_generator.batch_size\n\n# Fit the model\nhistory_warmup = model.fit(\n    train_generator,\n    steps_per_epoch=STEP_SIZE_TRAIN,\n    validation_data=valid_generator,\n    validation_steps=STEP_SIZE_VALID,\n    epochs=WARMUP_EPOCHS,\n    class_weight=class_weights_dict,  # Use the dictionary here\n    verbose=1\n).history\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T04:29:03.931706Z","iopub.execute_input":"2024-11-17T04:29:03.932383Z","iopub.status.idle":"2024-11-17T05:03:06.884406Z","shell.execute_reply.started":"2024-11-17T04:29:03.932344Z","shell.execute_reply":"2024-11-17T05:03:06.883532Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Fine tune the complete model\n","metadata":{}},{"cell_type":"code","source":"for layer in model.layers:\n    layer.trainable = True\n\nes = EarlyStopping(monitor='val_loss', mode='min', patience=ES_PATIENCE, restore_best_weights=True, verbose=1)\nrlrop = ReduceLROnPlateau(monitor='val_loss', mode='min', patience=RLROP_PATIENCE, factor=DECAY_DROP, min_lr=1e-6, verbose=1)\n\ncallback_list = [es, rlrop]\noptimizer = optimizers.Adam(learning_rate=LEARNING_RATE)\nmodel.compile(optimizer=optimizer, loss='categorical_crossentropy',  metrics=metric_list)\nmodel.summary()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-11-17T05:04:42.159927Z","iopub.execute_input":"2024-11-17T05:04:42.160698Z","iopub.status.idle":"2024-11-17T05:04:42.483572Z","shell.execute_reply.started":"2024-11-17T05:04:42.160655Z","shell.execute_reply":"2024-11-17T05:04:42.482641Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history_finetunning = model.fit(train_generator,\n                                          steps_per_epoch=STEP_SIZE_TRAIN,\n                                          validation_data=valid_generator,\n                                          validation_steps=STEP_SIZE_VALID,\n                                          epochs=EPOCHS,\n                                          callbacks=callback_list,\n                                          class_weight=class_weights_dict,\n                                          verbose=1).history","metadata":{"execution":{"iopub.status.busy":"2024-11-17T06:00:03.338310Z","iopub.execute_input":"2024-11-17T06:00:03.338987Z","iopub.status.idle":"2024-11-17T06:32:14.474558Z","shell.execute_reply.started":"2024-11-17T06:00:03.338946Z","shell.execute_reply":"2024-11-17T06:32:14.473642Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model loss Graph","metadata":{}},{"cell_type":"code","source":"history = {'loss': history_warmup['loss'] + history_finetunning['loss'], \n           'val_loss': history_warmup['val_loss'] + history_finetunning['val_loss'], \n           'accuracy': history_warmup['accuracy'] + history_finetunning['accuracy'], \n           'val_accuracy': history_warmup['val_accuracy'] + history_finetunning['val_accuracy']}\n\nsns.set_style(\"whitegrid\")\nfig, (ax1, ax2) = plt.subplots(2, 1, sharex='col', figsize=(20, 14))\n\nax1.plot(history['loss'], label='Train loss')\nax1.plot(history['val_loss'], label='Validation loss')\nax1.legend(loc='best')\nax1.set_title('Loss')\n\nax2.plot(history['accuracy'], label='Train accuracy')\nax2.plot(history['val_accuracy'], label='Validation accuracy')\nax2.legend(loc='best')\nax2.set_title('Accuracy')\n\nplt.xlabel('Epochs')\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-17T06:33:04.129058Z","iopub.execute_input":"2024-11-17T06:33:04.129963Z","iopub.status.idle":"2024-11-17T06:33:04.821861Z","shell.execute_reply.started":"2024-11-17T06:33:04.129920Z","shell.execute_reply":"2024-11-17T06:33:04.820943Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Evaluation ","metadata":{}},{"cell_type":"code","source":"# Create empty arays to keep the predictions and labels\nlastFullTrainPred = np.empty((0, N_CLASSES))\nlastFullTrainLabels = np.empty((0, N_CLASSES))\nlastFullValPred = np.empty((0, N_CLASSES))\nlastFullValLabels = np.empty((0, N_CLASSES))\n\n# Add train predictions and labels\nfor i in range(STEP_SIZE_TRAIN+1):\n    im, lbl = next(train_generator)\n    scores = model.predict(im, batch_size=train_generator.batch_size)\n    lastFullTrainPred = np.append(lastFullTrainPred, scores, axis=0)\n    lastFullTrainLabels = np.append(lastFullTrainLabels, lbl, axis=0)\n\n# Add validation predictions and labels\nfor i in range(STEP_SIZE_VALID+1):\n    im, lbl = next(valid_generator)\n    scores = model.predict(im, batch_size=valid_generator.batch_size)\n    lastFullValPred = np.append(lastFullValPred, scores, axis=0)\n    lastFullValLabels = np.append(lastFullValLabels, lbl, axis=0)\n    \n    \nlastFullComPred = np.concatenate((lastFullTrainPred, lastFullValPred))\nlastFullComLabels = np.concatenate((lastFullTrainLabels, lastFullValLabels))\ncomplete_labels = [np.argmax(label) for label in lastFullComLabels]\n\ntrain_preds = [np.argmax(pred) for pred in lastFullTrainPred]\ntrain_labels = [np.argmax(label) for label in lastFullTrainLabels]\nvalidation_preds = [np.argmax(pred) for pred in lastFullValPred]\nvalidation_labels = [np.argmax(label) for label in lastFullValLabels]","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-11-17T06:33:14.952465Z","iopub.execute_input":"2024-11-17T06:33:14.952857Z","iopub.status.idle":"2024-11-17T06:40:16.120452Z","shell.execute_reply.started":"2024-11-17T06:33:14.952820Z","shell.execute_reply":"2024-11-17T06:40:16.119435Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(1, 2, sharex='col', figsize=(24, 7))\nlabels = ['0 - No DR', '1 - Mild', '2 - Moderate', '3 - Severe', '4 - Proliferative DR']\ntrain_cnf_matrix = confusion_matrix(train_labels, train_preds)\nvalidation_cnf_matrix = confusion_matrix(validation_labels, validation_preds)\n\ntrain_cnf_matrix_norm = train_cnf_matrix.astype('float') / train_cnf_matrix.sum(axis=1)[:, np.newaxis]\nvalidation_cnf_matrix_norm = validation_cnf_matrix.astype('float') / validation_cnf_matrix.sum(axis=1)[:, np.newaxis]\n\ntrain_df_cm = pd.DataFrame(train_cnf_matrix_norm, index=labels, columns=labels)\nvalidation_df_cm = pd.DataFrame(validation_cnf_matrix_norm, index=labels, columns=labels)\n\nsns.heatmap(train_df_cm, annot=True, fmt='.2f', cmap=\"Blues\", ax=ax1).set_title('Train')\nsns.heatmap(validation_df_cm, annot=True, fmt='.2f', cmap=sns.cubehelix_palette(8), ax=ax2).set_title('Validation')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-17T06:41:30.922698Z","iopub.execute_input":"2024-11-17T06:41:30.923112Z","iopub.status.idle":"2024-11-17T06:41:31.813076Z","shell.execute_reply.started":"2024-11-17T06:41:30.923072Z","shell.execute_reply":"2024-11-17T06:41:31.812094Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Quadratic Weighted Kappa","metadata":{}},{"cell_type":"code","source":"print(\"Train Cohen Kappa score: %.3f\" % cohen_kappa_score(train_preds,train_labels, weights='quadratic'))\nprint(\"Validation Cohen Kappa score: %.3f\" % cohen_kappa_score(validation_preds, validation_labels, weights='quadratic'))\nprint(\"Complete set Cohen Kappa score: %.3f\" % cohen_kappa_score(train_preds+validation_preds, train_labels+validation_labels, weights='quadratic'))","metadata":{"execution":{"iopub.status.busy":"2024-11-17T06:41:54.767633Z","iopub.execute_input":"2024-11-17T06:41:54.768017Z","iopub.status.idle":"2024-11-17T06:41:54.781061Z","shell.execute_reply.started":"2024-11-17T06:41:54.767982Z","shell.execute_reply":"2024-11-17T06:41:54.780095Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.models import load_model\n\n# Load the model from the output directory\nmodel_loaded = load_model('/kaggle/input/efficient-model-400x400-top-model/tensorflow2/default/1/top efficient B0 multi.h5')  # Adjust path if necessary\n\n# Verify that the model is loaded\nmodel_loaded.summary()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-11-16T07:17:56.717350Z","iopub.execute_input":"2024-11-16T07:17:56.718152Z","iopub.status.idle":"2024-11-16T07:17:59.051760Z","shell.execute_reply.started":"2024-11-16T07:17:56.718112Z","shell.execute_reply":"2024-11-16T07:17:59.050796Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Evalution On Test Data","metadata":{}},{"cell_type":"code","source":"# Example: Use model to make predictions on test data\ntest_loss, test_acc = model.evaluate(test_generator)\nprint(\"Test accuracy:\", test_acc)\n\n# Make predictions\npredictions = model.predict(test_generator)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-17T06:42:16.834416Z","iopub.execute_input":"2024-11-17T06:42:16.835133Z","iopub.status.idle":"2024-11-17T06:45:15.650048Z","shell.execute_reply.started":"2024-11-17T06:42:16.835096Z","shell.execute_reply":"2024-11-17T06:45:15.649212Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"STEP_SIZE_TEST = test_generator.n // test_generator.batch_size","metadata":{"execution":{"iopub.status.busy":"2024-11-17T06:45:24.115862Z","iopub.execute_input":"2024-11-17T06:45:24.116260Z","iopub.status.idle":"2024-11-17T06:45:24.120849Z","shell.execute_reply.started":"2024-11-17T06:45:24.116223Z","shell.execute_reply":"2024-11-17T06:45:24.119891Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom sklearn.metrics import accuracy_score, roc_auc_score, recall_score, confusion_matrix\nimport numpy as np\nimport tensorflow as tf\n\n# Evaluate on validation data\nloss, acc = model.evaluate(test_generator, verbose=1)\n\n# Get predictions on validation data\ntest_predictions = model.predict(test_generator)\n\n# Convert predictions to class labels\ntest_predicted_classes = np.argmax(test_predictions, axis=1)\n\n# Get the true labels from the validation generator\ntest_true_labels = test_generator.classes\n\n# Accuracy\naccuracy = accuracy_score(test_true_labels, test_predicted_classes)\n\n# AUROC (for multi-class, one-vs-rest approach)\nauroc = roc_auc_score(tf.keras.utils.to_categorical(test_true_labels, num_classes=5), test_predictions, multi_class='ovr')\n\n# Sensitivity (recall)\nsensitivity = recall_score(test_true_labels, test_predicted_classes, average=None)\n\n# Specificity from confusion matrix\ncm = confusion_matrix(test_true_labels, test_predicted_classes)\nspecificity = []\nfor i in range(len(cm)):\n    tn = sum(sum(cm)) - (cm[i, :].sum() + cm[:, i].sum() - cm[i, i])\n    fp = cm[:, i].sum() - cm[i, i]\n    specificity.append(tn / (tn + fp))\n\n\n# Overall Sensitivity (macro-average recall)\noverall_sensitivity = np.mean(sensitivity)\n\n# Overall Specificity\noverall_specificity = np.mean(specificity)\n\n# Plotting Accuracy, AUROC, Sensitivity, and Specificity\n\n# Create a figure with subplots\nfig, axs = plt.subplots(2, 2, figsize=(12, 10))\n\n# Accuracy plot\naxs[0, 0].bar([\"Accuracy\"], [accuracy])\naxs[0, 0].set_ylim([0, 1])\naxs[0, 0].set_title(\"Test Accuracy\")\naxs[0, 0].set_ylabel(\"Score\")\n\n# AUROC plot\naxs[0, 1].bar([\"AUROC\"], [auroc])\naxs[0, 1].set_ylim([0, 1])\naxs[0, 1].set_title(\"Test AUROC\")\naxs[0, 1].set_ylabel(\"Score\")\n\n# Sensitivity plot (per class)\naxs[1, 0].bar(np.arange(len(sensitivity)), sensitivity)\naxs[1, 0].set_ylim([0, 1])\naxs[1, 0].set_title(\"Test Sensitivity per Class\")\naxs[1, 0].set_ylabel(\"Score\")\naxs[1, 0].set_xlabel(\"Class\")\n\n# Specificity plot (per class)\naxs[1, 1].bar(np.arange(len(specificity)), specificity)\naxs[1, 1].set_ylim([0, 1])\naxs[1, 1].set_title(\" Test Specificity per Class\")\naxs[1, 1].set_ylabel(\"Score\")\naxs[1, 1].set_xlabel(\"Class\")\n\n# Adjust layout\nplt.tight_layout()\n\n# Show the plot\nplt.show()\n\n\n# Print overall sensitivity and specificity\nprint(f\"Overall Sensitivity: {overall_sensitivity:.2f}\")\nprint(f\"Overall Specificity: {overall_specificity:.2f}\")","metadata":{"execution":{"iopub.status.busy":"2024-11-17T06:45:32.656805Z","iopub.execute_input":"2024-11-17T06:45:32.657291Z","iopub.status.idle":"2024-11-17T06:48:04.986887Z","shell.execute_reply.started":"2024-11-17T06:45:32.657235Z","shell.execute_reply":"2024-11-17T06:48:04.985970Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('auroc score for test data:',auroc)\nprint('sensitivity score for test data:',sensitivity)\nprint('specificity score for test data:',specificity)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T06:48:18.362342Z","iopub.execute_input":"2024-11-17T06:48:18.362724Z","iopub.status.idle":"2024-11-17T06:48:18.368709Z","shell.execute_reply.started":"2024-11-17T06:48:18.362688Z","shell.execute_reply":"2024-11-17T06:48:18.367677Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ROC Curve","metadata":{}},{"cell_type":"code","source":"import numpy as np\n# Get the true labels and predicted probabilities from the test generator\ntest_labels = []\ntest_preds = []\n\n# Loop through the test generator to get predictions and true labels\nfor i in range(STEP_SIZE_TEST):  # Adjust to match the number of test batches\n    im, lbl = next(test_generator)  # Get a batch of images and labels\n    scores = model.predict(im, batch_size=test_generator.batch_size)  # Predict probabilities\n    test_preds.append(scores)  # Append predicted probabilities\n    test_labels.append(lbl)  # Append true labels\n\n# Convert lists to numpy arrays\ntest_preds = np.vstack(test_preds)  # Stack all predictions\ntest_labels = np.vstack(test_labels)  # Stack all true labels\n","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-11-17T06:48:36.982473Z","iopub.execute_input":"2024-11-17T06:48:36.983078Z","iopub.status.idle":"2024-11-17T06:50:38.563065Z","shell.execute_reply.started":"2024-11-17T06:48:36.983039Z","shell.execute_reply":"2024-11-17T06:50:38.562176Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc\nimport matplotlib.pyplot as plt\nfrom itertools import cycle\nimport numpy as np\n\n# Assuming you have 5 classes\nn_classes = 5\nlabels = ['0 - No DR', '1 - Mild', '2 - Moderate', '3 - Severe', '4 - Proliferative DR']\n\n# Initialize dictionaries for False Positive Rate (fpr), True Positive Rate (tpr), and roc_auc\nfpr = dict()\ntpr = dict()\nroc_auc = dict()\n\n# Calculate ROC curve and AUC for each class\nfor i in range(n_classes):\n    fpr[i], tpr[i], _ = roc_curve(test_labels[:, i], test_preds[:, i])\n    roc_auc[i] = auc(fpr[i], tpr[i])\n\n# Plot ROC curve for each class\nplt.figure(figsize=(5, 5))\ncolors = cycle(['blue', 'red', 'green', 'orange', 'purple'])\n\nfor i, color in zip(range(n_classes), colors):\n    plt.plot(fpr[i], tpr[i], color=color, lw=2,\n             label='ROC curve of class {0} (area = {1:0.2f})'.format(labels[i], roc_auc[i]))\n\nplt.plot([0, 1], [0, 1], 'k--', lw=2)  # Diagonal line for a random classifier\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title(' Eff B0 ROC Curve for Multi-Class Classification')\nplt.legend(loc=\"lower right\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-17T06:51:01.702119Z","iopub.execute_input":"2024-11-17T06:51:01.702538Z","iopub.status.idle":"2024-11-17T06:51:02.029725Z","shell.execute_reply.started":"2024-11-17T06:51:01.702501Z","shell.execute_reply":"2024-11-17T06:51:02.028838Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.metrics import roc_curve, auc\nimport matplotlib.pyplot as plt\n\n# Get the true labels and predicted probabilities from the test generator\ntest_labels = []\ntest_preds = []\n\n# Loop through the test generator to get predictions and true labels\nfor i in range(STEP_SIZE_TEST):  # Adjust to match the number of test batches\n    im, lbl = next(test_generator)  # Get a batch of images and labels\n    scores = model.predict(im, batch_size=test_generator.batch_size)  # Predict probabilities\n    test_preds.append(scores)  # Append predicted probabilities\n    test_labels.append(lbl)  # Append true labels\n\n# Convert lists to numpy arrays\ntest_preds = np.vstack(test_preds)  # Stack all predictions\ntest_labels = np.vstack(test_labels)  # Stack all true labels\n\n# Assuming you have 5 classes\nn_classes = 5\n\n# Convert labels to binary format\n# Create binary labels for the true labels\nbinary_labels = np.zeros((test_labels.shape[0], n_classes))\nfor i in range(test_labels.shape[0]):\n    binary_labels[i, int(np.argmax(test_labels[i]))] = 1\n\n# Calculate ROC curve and AUC for the micro-average\nfpr, tpr, _ = roc_curve(binary_labels.ravel(), test_preds.ravel())\nroc_auc = auc(fpr, tpr)\n\n# Plot the micro-average ROC curve\nplt.figure(figsize=(5, 6))\nplt.plot(fpr, tpr, color='blue', lw=2, label='Micro-average ROC curve (area = {0:0.2f})'.format(roc_auc))\nplt.plot([0, 1], [0, 1], 'k--', lw=2)  # Diagonal line for a random classifier\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Micro-average ROC Curve for Multi-Class Classification')\nplt.legend(loc=\"lower right\")\nplt.grid()  # Optional: add a grid for better readability\nplt.show()\n\n# Print the micro-average AUC\nprint(f'Micro-average AUC: {roc_auc:.2f}')\n","metadata":{"execution":{"iopub.status.busy":"2024-11-17T06:52:18.461206Z","iopub.execute_input":"2024-11-17T06:52:18.461591Z","iopub.status.idle":"2024-11-17T06:54:22.279692Z","shell.execute_reply.started":"2024-11-17T06:52:18.461556Z","shell.execute_reply":"2024-11-17T06:54:22.278789Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Save model","metadata":{}},{"cell_type":"code","source":"model.save(\"top efficient B0 multi.h5\")","metadata":{"execution":{"iopub.status.busy":"2024-11-17T06:55:33.610371Z","iopub.execute_input":"2024-11-17T06:55:33.610736Z","iopub.status.idle":"2024-11-17T06:55:34.369630Z","shell.execute_reply.started":"2024-11-17T06:55:33.610705Z","shell.execute_reply":"2024-11-17T06:55:34.368787Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Multi Level Efficient ","metadata":{}},{"cell_type":"code","source":"\nfrom tensorflow.keras import layers, Model","metadata":{"execution":{"iopub.status.busy":"2024-11-17T06:55:37.770330Z","iopub.execute_input":"2024-11-17T06:55:37.770986Z","iopub.status.idle":"2024-11-17T06:55:37.775391Z","shell.execute_reply.started":"2024-11-17T06:55:37.770949Z","shell.execute_reply":"2024-11-17T06:55:37.774209Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the EfficientNet-B0 base model without the top (fully connected) layers\nbase_model = EfficientNetB0(weights='imagenet', include_top=False, input_shape=(400, 400, 3))\n\n# Extracting features at different levels\nblock3_output = base_model.get_layer('block3a_expand_activation').output  # (32, 100, 100)\nblock5_output = base_model.get_layer('block5a_expand_activation').output  # (24, 50, 50)\nblock7_output = base_model.get_layer('block7a_expand_activation').output  # (40, 13, 13)\n\n# Bilinear Interpolation to resize feature maps to (25x25)\nblock3_resized = layers.Lambda(lambda x: tf.image.resize(x, (25, 25), method='bilinear'))(block3_output)\nblock5_resized = layers.Lambda(lambda x: tf.image.resize(x, (25, 25), method='bilinear'))(block5_output)\nblock7_resized = layers.Lambda(lambda x: tf.image.resize(x, (25, 25), method='bilinear'))(block7_output)\n\n# Apply 1x1 convolution to unify the channel depth across feature maps\nblock3_depth_adjusted = layers.Conv2D(320, (1, 1), padding='same', activation='relu')(block3_resized)\nblock5_depth_adjusted = layers.Conv2D(320, (1, 1), padding='same', activation='relu')(block5_resized)\nblock7_depth_adjusted = layers.Conv2D(320, (1, 1), padding='same', activation='relu')(block7_resized)\n\n# Concatenate the resized and depth-adjusted feature maps\nconcatenated_features = layers.Concatenate()([block3_depth_adjusted, block5_depth_adjusted, block7_depth_adjusted])\n\n# Apply a 1x1 convolution to reduce the depth of concatenated features\nconv1x1 = layers.Conv2D(320, (1, 1), padding='same', activation='relu')(concatenated_features)\n\n# Adaptive Average Pooling to reduce to (1, 1, 320)\nadaptive_avg_pool = layers.GlobalAveragePooling2D()(conv1x1)\n\n# Final fully connected layer for classification (5 classes as per your use case)\noutput = layers.Dense(5, activation='softmax')(adaptive_avg_pool)\n\n# Create the final model\nmulti_level_efficientnet_b0 = Model(inputs=base_model.input, outputs=output)\n\n# Compile the model\nmulti_level_efficientnet_b0.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Print the model summary to see the architecture\nmulti_level_efficientnet_b0.summary()\n","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-11-17T06:55:41.959230Z","iopub.execute_input":"2024-11-17T06:55:41.960018Z","iopub.status.idle":"2024-11-17T06:55:43.426433Z","shell.execute_reply.started":"2024-11-17T06:55:41.959976Z","shell.execute_reply":"2024-11-17T06:55:43.425519Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history_multi_level_efficientB0 = multi_level_efficientnet_b0.fit(train_generator,\n                                          steps_per_epoch=STEP_SIZE_TRAIN,\n                                          validation_data=valid_generator,\n                                          validation_steps=STEP_SIZE_VALID,\n                                          epochs=EPOCHS,\n                                          callbacks=callback_list,\n                                          class_weight=class_weights_dict,\n                                          verbose=1).history","metadata":{"execution":{"iopub.status.busy":"2024-11-17T06:55:47.247084Z","iopub.execute_input":"2024-11-17T06:55:47.247497Z","iopub.status.idle":"2024-11-17T07:16:49.944299Z","shell.execute_reply.started":"2024-11-17T06:55:47.247460Z","shell.execute_reply":"2024-11-17T07:16:49.943307Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for layer in multi_level_efficientnet_b0.layers:\n    layer.trainable = True\n\nes = EarlyStopping(monitor='val_loss', mode='min', patience=ES_PATIENCE, restore_best_weights=True, verbose=1)\nrlrop = ReduceLROnPlateau(monitor='val_loss', mode='min', patience=RLROP_PATIENCE, factor=DECAY_DROP, min_lr=1e-6, verbose=1)\n\ncallback_list = [es, rlrop]\noptimizer = optimizers.Adam(learning_rate=LEARNING_RATE)\nmulti_level_efficientnet_b0.compile(optimizer=optimizer, loss='categorical_crossentropy',  metrics=metric_list)\nmulti_level_efficientnet_b0.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T07:17:52.525002Z","iopub.execute_input":"2024-11-17T07:17:52.525947Z","iopub.status.idle":"2024-11-17T07:17:52.853591Z","shell.execute_reply.started":"2024-11-17T07:17:52.525903Z","shell.execute_reply":"2024-11-17T07:17:52.852426Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#fine tune model multi level efficient net B0\nhistory_multi_level_efficientB0_finetune = multi_level_efficientnet_b0.fit(train_generator,\n                                          steps_per_epoch=STEP_SIZE_TRAIN,\n                                          validation_data=valid_generator,\n                                          validation_steps=STEP_SIZE_VALID,\n                                          epochs=EPOCHS,\n                                          callbacks=callback_list,\n                                          class_weight=class_weights_dict,\n                                          verbose=1).history","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T07:17:58.238458Z","iopub.execute_input":"2024-11-17T07:17:58.238842Z","iopub.status.idle":"2024-11-17T07:57:54.843720Z","shell.execute_reply.started":"2024-11-17T07:17:58.238807Z","shell.execute_reply":"2024-11-17T07:57:54.842870Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming 'history_multi_level_efficientB0' is a dictionary\nepochs = range(1, len(history_multi_level_efficientB0_finetune[\"accuracy\"]) + 1)\ntrain_acc = history_multi_level_efficientB0_finetune[\"accuracy\"]\nval_acc = history_multi_level_efficientB0_finetune[\"val_accuracy\"]\n\n# Plotting the training and validation accuracy\nimport matplotlib.pyplot as plt\n\nplt.figure()\nplt.plot(epochs, train_acc, \"r\", label=\"Training accuracy\")\nplt.plot(epochs, val_acc, \"b\", label=\"Validation accuracy\")\nplt.title(\"Training and validation accuracy for Multi-Level EfficientB0\")\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.legend()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-17T08:01:47.024236Z","iopub.execute_input":"2024-11-17T08:01:47.024673Z","iopub.status.idle":"2024-11-17T08:01:47.383004Z","shell.execute_reply.started":"2024-11-17T08:01:47.024633Z","shell.execute_reply":"2024-11-17T08:01:47.381814Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming 'history_multi_level_efficientB0' is a dictionary\nepochs = range(1, len(history_multi_level_efficientB0_finetune[\"loss\"]) + 1)\ntrain_acc = history_multi_level_efficientB0_finetune[\"loss\"]\nval_acc = history_multi_level_efficientB0_finetune[\"val_loss\"]\n\n# Plotting the training and validation accuracy\nimport matplotlib.pyplot as plt\n\nplt.figure()\nplt.plot(epochs, train_acc, \"r\", label=\"Training loss\")\nplt.plot(epochs, val_acc, \"b\", label=\"Validation loss\")\nplt.title(\"Training and validation loss for Multi-Level EfficientB0\")\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.legend()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-17T08:01:52.320627Z","iopub.execute_input":"2024-11-17T08:01:52.321533Z","iopub.status.idle":"2024-11-17T08:01:52.663499Z","shell.execute_reply.started":"2024-11-17T08:01:52.321480Z","shell.execute_reply":"2024-11-17T08:01:52.662569Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":" multi_level_efficientnet_b0.save(\"top multi efficient B0 multi finetune.h5\")","metadata":{"execution":{"iopub.status.busy":"2024-11-16T10:10:51.794419Z","iopub.execute_input":"2024-11-16T10:10:51.794828Z","iopub.status.idle":"2024-11-16T10:10:52.429235Z","shell.execute_reply.started":"2024-11-16T10:10:51.794789Z","shell.execute_reply":"2024-11-16T10:10:52.428304Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Evaluate on test data","metadata":{}},{"cell_type":"code","source":"# Example: Use model to make predictions on test data\ntest_loss, test_acc = multi_level_efficientnet_b0.evaluate(test_generator)\nprint(\"Test accuracy:\", test_acc)\n\n# Make predictions\npredictions_Ml = multi_level_efficientnet_b0.predict(test_generator)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-17T08:02:04.462126Z","iopub.execute_input":"2024-11-17T08:02:04.462495Z","iopub.status.idle":"2024-11-17T08:04:45.092487Z","shell.execute_reply.started":"2024-11-17T08:02:04.462464Z","shell.execute_reply":"2024-11-17T08:04:45.091624Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#predictions using argmax\npredicted_classes = np.argmax(predictions,axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T08:04:52.775133Z","iopub.execute_input":"2024-11-17T08:04:52.775779Z","iopub.status.idle":"2024-11-17T08:04:52.780037Z","shell.execute_reply.started":"2024-11-17T08:04:52.775740Z","shell.execute_reply":"2024-11-17T08:04:52.779041Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"true_labels = test_generator.classes","metadata":{"execution":{"iopub.status.busy":"2024-11-17T08:04:56.780662Z","iopub.execute_input":"2024-11-17T08:04:56.781042Z","iopub.status.idle":"2024-11-17T08:04:56.785463Z","shell.execute_reply.started":"2024-11-17T08:04:56.781004Z","shell.execute_reply":"2024-11-17T08:04:56.784523Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"STEP_SIZE_TEST = test_generator.n // test_generator.batch_size","metadata":{"execution":{"iopub.status.busy":"2024-11-17T08:05:00.547639Z","iopub.execute_input":"2024-11-17T08:05:00.548015Z","iopub.status.idle":"2024-11-17T08:05:00.552516Z","shell.execute_reply.started":"2024-11-17T08:05:00.547980Z","shell.execute_reply":"2024-11-17T08:05:00.551507Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# Get the true labels and predicted probabilities from the test generator\ntest_labels = []\ntest_preds = []\n\n# Loop through the test generator to get predictions and true labels\nfor i in range(STEP_SIZE_TEST):  # Adjust to match the number of test batches\n    im, lbl = next(test_generator)  # Get a batch of images and labels\n    scores = multi_level_efficientnet_b0.predict(im, batch_size=test_generator.batch_size)  # Predict probabilities\n    test_preds.append(scores)  # Append predicted probabilities\n    test_labels.append(lbl)  # Append true labels\n\n# Convert lists to numpy arrays\ntest_preds = np.vstack(test_preds)  # Stack all predictions\ntest_labels = np.vstack(test_labels)  # Stack all true labels\n","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-11-17T08:05:14.704814Z","iopub.execute_input":"2024-11-17T08:05:14.706013Z","iopub.status.idle":"2024-11-17T08:07:21.209631Z","shell.execute_reply.started":"2024-11-17T08:05:14.705959Z","shell.execute_reply":"2024-11-17T08:07:21.208677Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc\nimport matplotlib.pyplot as plt\nfrom itertools import cycle\nimport numpy as np\n\n# Assuming you have 5 classes\nn_classes = 5\nlabels = ['0 - No DR', '1 - Mild', '2 - Moderate', '3 - Severe', '4 - Proliferative DR']\n\n# Initialize dictionaries for False Positive Rate (fpr), True Positive Rate (tpr), and roc_auc\nfpr = dict()\ntpr = dict()\nroc_auc = dict()\n\n# Calculate ROC curve and AUC for each class\nfor i in range(n_classes):\n    fpr[i], tpr[i], _ = roc_curve(test_labels[:, i], test_preds[:, i])\n    roc_auc[i] = auc(fpr[i], tpr[i])\n\n# Plot ROC curve for each class\nplt.figure(figsize=(5, 5))\ncolors = cycle(['blue', 'red', 'green', 'orange', 'purple'])\n\nfor i, color in zip(range(n_classes), colors):\n    plt.plot(fpr[i], tpr[i], color=color, lw=2,\n             label='ROC curve of class {0} (area = {1:0.2f})'.format(labels[i], roc_auc[i]))\n\nplt.plot([0, 1], [0, 1], 'k--', lw=2)  # Diagonal line for a random classifier\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('ROC Curve for Multi-Class Classification')\nplt.legend(loc=\"lower right\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-17T08:08:28.654184Z","iopub.execute_input":"2024-11-17T08:08:28.654881Z","iopub.status.idle":"2024-11-17T08:08:29.052769Z","shell.execute_reply.started":"2024-11-17T08:08:28.654843Z","shell.execute_reply":"2024-11-17T08:08:29.051827Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.metrics import roc_curve, auc\nimport matplotlib.pyplot as plt\n\n# Get the true labels and predicted probabilities from the test generator\ntest_labels = []\ntest_preds = []\n\n# Loop through the test generator to get predictions and true labels\nfor i in range(STEP_SIZE_TEST):  # Adjust to match the number of test batches\n    im, lbl = next(test_generator)  # Get a batch of images and labels\n    scores = multi_level_efficientnet_b0.predict(im, batch_size=test_generator.batch_size)  # Predict probabilities\n    test_preds.append(scores)  # Append predicted probabilities\n    test_labels.append(lbl)  # Append true labels\n\n# Convert lists to numpy arrays\ntest_preds = np.vstack(test_preds)  # Stack all predictions\ntest_labels = np.vstack(test_labels)  # Stack all true labels\n\n# Assuming you have 5 classes\nn_classes = 5\n\n# Convert labels to binary format\n# Create binary labels for the true labels\nbinary_labels = np.zeros((test_labels.shape[0], n_classes))\nfor i in range(test_labels.shape[0]):\n    binary_labels[i, int(np.argmax(test_labels[i]))] = 1\n\n# Calculate ROC curve and AUC for the micro-average\nfpr, tpr, _ = roc_curve(binary_labels.ravel(), test_preds.ravel())\nroc_auc = auc(fpr, tpr)\n\n# Plot the micro-average ROC curve\nplt.figure(figsize=(5, 6))","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-11-17T08:08:35.989181Z","iopub.execute_input":"2024-11-17T08:08:35.989855Z","iopub.status.idle":"2024-11-17T08:10:38.001044Z","shell.execute_reply.started":"2024-11-17T08:08:35.989815Z","shell.execute_reply":"2024-11-17T08:10:38.000058Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nplt.plot(fpr, tpr, color='blue', lw=2, label='Micro-average ROC curve (area = {0:0.2f})'.format(roc_auc))\nplt.plot([0, 1], [0, 1], 'k--', lw=2)  # Diagonal line for a random classifier\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Micro-average AUROC Curve for Multi-Class Classification')\nplt.legend(loc=\"lower right\")\nplt.grid()  # Optional: add a grid for better readability\nplt.show()\n\n# Print the micro-average AUC\nprint(f'Micro-average AUC: {roc_auc:.2f}')","metadata":{"execution":{"iopub.status.busy":"2024-11-17T08:10:57.047513Z","iopub.execute_input":"2024-11-17T08:10:57.048311Z","iopub.status.idle":"2024-11-17T08:10:57.325966Z","shell.execute_reply.started":"2024-11-17T08:10:57.048271Z","shell.execute_reply":"2024-11-17T08:10:57.325070Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('auroc score for test data:',auroc)\nprint('specificity overall score for test data:',np.mean(specificity))\nprint('sensitivity overall score for test data:',np.mean(sensitivity))\nprint('specificity  score for test data:',specificity)\nprint('sensitivity  score for test data:',sensitivity)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T08:11:02.542084Z","iopub.execute_input":"2024-11-17T08:11:02.542943Z","iopub.status.idle":"2024-11-17T08:11:02.549183Z","shell.execute_reply.started":"2024-11-17T08:11:02.542902Z","shell.execute_reply":"2024-11-17T08:11:02.548193Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#ensemble the model \nensemble_model =np.mean([predictions,predictions_Ml],axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T08:11:08.727884Z","iopub.execute_input":"2024-11-17T08:11:08.728742Z","iopub.status.idle":"2024-11-17T08:11:08.733497Z","shell.execute_reply.started":"2024-11-17T08:11:08.728704Z","shell.execute_reply":"2024-11-17T08:11:08.732345Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ensemble_final_preds = np.argmax(ensemble_model,axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T08:11:11.600520Z","iopub.execute_input":"2024-11-17T08:11:11.601371Z","iopub.status.idle":"2024-11-17T08:11:11.605459Z","shell.execute_reply.started":"2024-11-17T08:11:11.601333Z","shell.execute_reply":"2024-11-17T08:11:11.604550Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ensemble_acc = accuracy_score(true_labels,ensemble_final_preds)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T08:11:14.505511Z","iopub.execute_input":"2024-11-17T08:11:14.506121Z","iopub.status.idle":"2024-11-17T08:11:14.511525Z","shell.execute_reply.started":"2024-11-17T08:11:14.506084Z","shell.execute_reply":"2024-11-17T08:11:14.510707Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"ensemble accuracy :\",ensemble_acc)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T08:11:17.437370Z","iopub.execute_input":"2024-11-17T08:11:17.438220Z","iopub.status.idle":"2024-11-17T08:11:17.443137Z","shell.execute_reply.started":"2024-11-17T08:11:17.438179Z","shell.execute_reply":"2024-11-17T08:11:17.442216Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import roc_auc_score, roc_curve, auc, recall_score, confusion_matrix\nfrom sklearn.preprocessing import label_binarize\n\n\n# Number of classes\nnum_classes = 5\n\n# One-hot encode the true labels\ntrue_labels_one_hot = label_binarize(true_labels, classes=[0, 1, 2, 3, 4])\n\n# --- AUROC ---\n# Calculate AUROC for each class using one-vs-rest approach\nauroc_ens = roc_auc_score(true_labels_one_hot, ensemble_model, multi_class='ovr')\nprint(f\"Overall AUROC: {auroc}\")\n\n# --- Sensitivity (Recall) ---\nsensitivity_ens = recall_score(true_labels, ensemble_final_preds, average=None)\nprint(f\"Sensitivity per class: {sensitivity_ens}\")\n\n# --- Specificity ---\n# Calculate confusion matrix\ncm = confusion_matrix(true_labels, ensemble_final_preds)\nspecificity_ens = []\nfor i in range(len(cm)):\n    tn = sum(sum(cm)) - (cm[i, :].sum() + cm[:, i].sum() - cm[i, i])\n    fp = cm[:, i].sum() - cm[i, i]\n    specificity_ens.append(tn / (tn + fp))\n\nprint(f\"Specificity per class: {specificity_ens}\")\n\n# Overall Sensitivity and Specificity\noverall_sensitivity_ens = np.mean(sensitivity_ens)\noverall_specificity_ens = np.mean(specificity_ens)\n\nprint(f\"Overall Sensitivity: {overall_sensitivity_ens}\")\nprint(f\"Overall Specificity: {overall_specificity_ens}\")\n\n# --- ROC Curves for each class ---\n# Plot ROC curves for each class\nfpr = dict()\ntpr = dict()\nroc_auc = dict()\n\nplt.figure(figsize=(5, 6))\n\nfor i in range(num_classes):\n    fpr[i], tpr[i], _ = roc_curve(true_labels_one_hot[:, i], ensemble_model[:, i])\n    roc_auc[i] = auc(fpr[i], tpr[i])\n    plt.plot(fpr[i], tpr[i], lw=2, label=f\"Class {i} (area = {roc_auc[i]:.2f})\")\n\n# Plot ROC curve for random guessing\nplt.plot([0, 1], [0, 1], color=\"navy\", lw=2, linestyle=\"--\")\n\n# Plot settings\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel(\"False Positive Rate\")\nplt.ylabel(\"True Positive Rate\")\nplt.title(\"ROC Curves for Ensemble model\")\nplt.legend(loc=\"lower right\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-17T08:12:29.403979Z","iopub.execute_input":"2024-11-17T08:12:29.404380Z","iopub.status.idle":"2024-11-17T08:12:29.810381Z","shell.execute_reply.started":"2024-11-17T08:12:29.404345Z","shell.execute_reply":"2024-11-17T08:12:29.809438Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import roc_curve, auc, roc_auc_score\nfrom sklearn.preprocessing import label_binarize\n\n# Assuming you have the following variables from your ensemble model\n# `ensemble_model`: Probability predictions from the ensemble model\n# `true_labels`: True class labels\nnum_classes = 5  # Number of classes in your classification problem\n\n# One-hot encode the true labels\ntrue_labels_one_hot = label_binarize(true_labels, classes=[0, 1, 2, 3, 4])\n\n# --- Overall AUROC ---\n# Calculate the macro-average AUROC score\noverall_auroc = roc_auc_score(true_labels_one_hot, ensemble_model, average='macro', multi_class='ovr')\nprint(f\"Overall AUROC: {overall_auroc}\")\n\n# --- Macro-average ROC Curve ---\n# Compute ROC curve and ROC area for each class\nfpr = dict()\ntpr = dict()\nroc_auc = dict()\n\nfor i in range(num_classes):\n    fpr[i], tpr[i], _ = roc_curve(true_labels_one_hot[:, i], ensemble_model[:, i])\n    roc_auc[i] = auc(fpr[i], tpr[i])\n\n# Compute micro-average ROC curve and ROC area\nfpr[\"micro\"], tpr[\"micro\"], _ = roc_curve(true_labels_one_hot.ravel(), ensemble_model.ravel())\nroc_auc[\"micro\"] = auc(fpr[\"micro\"], tpr[\"micro\"])\n\n# Compute macro-average ROC curve and ROC area\n# Aggregate all false positive rates\nall_fpr = np.unique(np.concatenate([fpr[i] for i in range(num_classes)]))\n\n# Interpolate all ROC curves at these points\nmean_tpr = np.zeros_like(all_fpr)\nfor i in range(num_classes):\n    mean_tpr += np.interp(all_fpr, fpr[i], tpr[i])\n\n# Average it and compute the AUC\nmean_tpr /= num_classes\n\nfpr[\"macro\"] = all_fpr\ntpr[\"macro\"] = mean_tpr\nroc_auc[\"macro\"] = auc(fpr[\"macro\"], tpr[\"macro\"])\n\n# Plot the macro-average ROC curve\nplt.figure(figsize=(10, 5))\nplt.plot(fpr[\"macro\"], tpr[\"macro\"],\n         label=f\"Macro-average ROC curve (area = {roc_auc['macro']:.2f})\",\n         color=\"navy\", linestyle=\":\", linewidth=4)\n\n# Plot ROC curve for random guessing\nplt.plot([0, 1], [0, 1], color=\"gray\", lw=2, linestyle=\"--\")\n\n# Plot individual class ROC curves (optional)\nfor i in range(num_classes):\n    plt.plot(fpr[i], tpr[i], lw=2, label=f\"Class {i} (area = {roc_auc[i]:.2f})\")\n\n# Plot settings\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel(\"False Positive Rate\")\nplt.ylabel(\"True Positive Rate\")\nplt.title(\" Ensemble Macro-Average ROC Curve and Class-wise ROC Curves\")\nplt.legend(loc=\"lower right\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-17T08:12:37.440729Z","iopub.execute_input":"2024-11-17T08:12:37.441363Z","iopub.status.idle":"2024-11-17T08:12:37.874246Z","shell.execute_reply.started":"2024-11-17T08:12:37.441322Z","shell.execute_reply":"2024-11-17T08:12:37.873326Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# TTA Implementation","metadata":{}},{"cell_type":"markdown","source":"# Note:\n While Test-Time Augmentation (TTA) typically yields the best results and improves accuracy, it requires significant GPU resources. Therefore, instead of using extensive data augmentation, I implemented only horizontal flipping (FLIP_LR). However, running this on my Kaggle environment is not feasible due to limited computational power.\n","metadata":{}},{"cell_type":"markdown","source":"Edafa is a Python wrapper designed to simplify the implementation of Test-Time Augmentation (TTA) for computer vision tasks. Test-Time Augmentation is a technique where, during model evaluation (testing), multiple augmentations of the input image are generated, and the model's predictions are averaged or fused to improve the final performance. This approach is particularly beneficial for tasks like segmentation, super-resolution, and pansharpening, as it can reduce overfitting, increase robustness, and provide better generalization to unseen data.","metadata":{}},{"cell_type":"code","source":"!pip install edafa","metadata":{"execution":{"iopub.status.busy":"2024-11-04T04:52:01.395605Z","iopub.execute_input":"2024-11-04T04:52:01.396454Z","iopub.status.idle":"2024-11-04T04:52:14.186913Z","shell.execute_reply.started":"2024-11-04T04:52:01.396414Z","shell.execute_reply":"2024-11-04T04:52:14.185898Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from edafa import ClassPredictor","metadata":{"execution":{"iopub.status.busy":"2024-11-04T04:52:27.943638Z","iopub.execute_input":"2024-11-04T04:52:27.944063Z","iopub.status.idle":"2024-11-04T04:52:27.955476Z","shell.execute_reply.started":"2024-11-04T04:52:27.944021Z","shell.execute_reply":"2024-11-04T04:52:27.954470Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.models import load_model\nmodel_loaded = load_model('/kaggle/input/efficient-model-400x400-top-model/tensorflow2/default/1/top efficient B0 multi.h5')","metadata":{"execution":{"iopub.status.busy":"2024-11-04T05:00:32.516251Z","iopub.execute_input":"2024-11-04T05:00:32.517191Z","iopub.status.idle":"2024-11-04T05:00:34.521216Z","shell.execute_reply.started":"2024-11-04T05:00:32.517133Z","shell.execute_reply":"2024-11-04T05:00:34.520244Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\nclass TTAPredictor:\n    def __init__(self, model, config):\n        self.model = model\n        self.aug = config.get('aug', 'NO')  # Default to no augmentation\n        self.mean = config.get('mean', 'ARITH')  # Default to arithmetic mean\n\n    def augment_image(self, image):\n        \"\"\"Apply the selected augmentation to the image.\"\"\"\n        if self.aug == 'FLIP_LR':\n            return np.flip(image, axis=2)  # Flip Left-Right\n        return image  # Return the original if no augmentation\n    \n    def predict_tta(self, test_generator):\n        \"\"\"Predict with Test-Time Augmentation (TTA) using only FLIP_LR.\"\"\"\n        predictions = []\n\n        for batch in test_generator:\n            images, _ = batch  # Extract images, ignore labels\n\n            # Apply augmentation and ensure the shape is consistent\n            augmented_batch = np.array([self.augment_image(img) for img in images])\n            assert augmented_batch.shape == images.shape, f\"Shape mismatch: {augmented_batch.shape} vs {images.shape}\"\n\n            # Get predictions for the augmented images\n            preds = self.model.predict(augmented_batch)\n            predictions.append(preds)\n\n        return np.concatenate(predictions, axis=0)\n\n# Configuration for using only FLIP_LR augmentation\nconf = {\n    \"aug\": \"FLIP_LR\",\n    \"mean\": \"ARITH\"  # Arithmetic mean for averaging predictions (if used in ensemble settings)\n}\n\n# Initialize TTA predictor with the model and config\ntta_predictor = TTAPredictor(model_loaded, conf)\n\n# Use the TTA predictor to get augmented predictions\ny_pred_aug = tta_predictor.predict_tta(test_generator)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-04T05:00:42.527811Z","iopub.execute_input":"2024-11-04T05:00:42.528503Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ensemble_acc = accuracy_score(true_labels,ensemble_final_preds)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TTA = accuracy_score(true_labels,ensemble_final_preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T06:15:48.687549Z","iopub.execute_input":"2024-11-05T06:15:48.688401Z","iopub.status.idle":"2024-11-05T06:15:48.729421Z","shell.execute_reply.started":"2024-11-05T06:15:48.688348Z","shell.execute_reply":"2024-11-05T06:15:48.728311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}