{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"},{"sourceId":10478383,"sourceType":"datasetVersion","datasetId":6488230},{"sourceId":10478671,"sourceType":"datasetVersion","datasetId":6488377}],"dockerImageVersionId":30840,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, sys\nimport cv2\nimport pandas as pd\nfrom PIL import Image\nimport json\nimport math\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.applications import DenseNet169\nfrom tensorflow.keras.callbacks import ModelCheckpoint, Callback, EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Sequential\nfrom keras.optimizers import Adam\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score\nfrom sklearn.metrics import roc_curve, auc\nfrom sklearn.utils import shuffle\nfrom collections import Counter\nimport pickle\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.applications.densenet import DenseNet169\nimport scipy\nfrom tqdm import tqdm\nfrom IPython.display import display\nimport warnings\nwarnings.filterwarnings('ignore')\n%matplotlib inline\n\nEPOCHS = 100\nBATCH_SIZE = 64\nSEED = 19071591\nLRATE = 0.00005\nMIN_LRATE = 0.00001\nVERBOSE=1\nDROPOUT = 0.3\n\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:36:16.140207Z","iopub.execute_input":"2025-01-15T15:36:16.140662Z","iopub.status.idle":"2025-01-15T15:36:30.032605Z","shell.execute_reply.started":"2025-01-15T15:36:16.140626Z","shell.execute_reply":"2025-01-15T15:36:30.031953Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"import shutil\ndataset_dir = '/kaggle/working/datasets'\ntrain_images_dir = os.path.join(dataset_dir, 'train_images')\ntrain_csv_path = os.path.join(dataset_dir, 'train.csv')\n\nif not os.path.exists(dataset_dir):\n    os.makedirs(dataset_dir)\n\nif not os.path.exists(train_images_dir):\n    os.makedirs(train_images_dir)\n\ntrain_df = pd.DataFrame(columns=['id_code', 'diagnosis'])\n\ntrain_df2015 = pd.read_csv('/kaggle/input/resized-2015-2019-blindness-detection-images/labels/trainLabels15.csv')\ntrain_df2015 = train_df2015.rename(columns={'image': 'id_code', 'level': 'diagnosis'})\ntrain_df2015 = train_df2015.drop(train_df2015[train_df2015['diagnosis'] == 0].sample(n=25810, replace=False).index)\ntrain_df2015 = train_df2015.drop(train_df2015[train_df2015['diagnosis'] == 1].sample(n=1746, replace=False).index)\ntrain_df2015 = train_df2015.drop(train_df2015[train_df2015['diagnosis'] == 2].sample(n=5224, replace=False).index)\n\ntrain_df2019 = pd.read_csv('/kaggle/input/resized-2015-2019-blindness-detection-images/labels/trainLabels19.csv')\ntrain_df2019 = train_df2019.drop(train_df2019[train_df2019['diagnosis'] == 0].sample(n=738, replace=False).index)\n\ntrain_dfidrid = pd.read_csv('/kaggle/input/idrid-dataset/idrid_labels.csv')\ntrain_dfidrid = train_dfidrid.drop(train_dfidrid[train_dfidrid['diagnosis'] == 0].sample(n=129, replace=False).index)\ntrain_dfidrid = train_dfidrid.drop(train_dfidrid[train_dfidrid['diagnosis'] == 1].sample(n=22, replace=False).index)\ntrain_dfidrid = train_dfidrid.drop(train_dfidrid[train_dfidrid['diagnosis'] == 2].sample(n=156, replace=False).index)\ntrain_dfidrid = train_dfidrid.drop(train_dfidrid[train_dfidrid['diagnosis'] == 3].sample(n=83, replace=False).index)\n\ntrain_df = pd.concat([train_df2015, train_df2019, train_dfidrid], ignore_index=True)\n\nsource_dirs = ['/kaggle/input/resized-2015-2019-blindness-detection-images/resized train 15',\n'/kaggle/input/resized-2015-2019-blindness-detection-images/resized train 19',\n'/kaggle/input/idrid-dataset/Imagenes/Imagenes']\n\nnew_rows = []\nfor index, row in train_df.iterrows():\n    image_id = row['id_code']\n    diagnosis = row['diagnosis']\n\n    found = False  \n    for source_dir in source_dirs:\n        source_path_jpg = os.path.join(source_dir, image_id + '.jpg')\n        source_path_png = os.path.join(source_dir, image_id + '.png')\n\n        if os.path.exists(source_path_jpg):\n            image_name = image_id\n            destination_path = os.path.join(train_images_dir, image_name)\n            shutil.copy(source_path_jpg, destination_path)\n            new_rows.append({'id_code': image_name, 'diagnosis': diagnosis})\n            found = True\n        elif os.path.exists(source_path_png):\n            image_name = image_id\n            destination_path = os.path.join(train_images_dir, image_name)\n            shutil.copy(source_path_png, destination_path)\n            new_rows.append({'id_code': image_name, 'diagnosis': diagnosis})\n            found = True\n\n        if found:\n            break \n\n    if not found:\n        print(f\"Image {image_id} not found in any source directory (neither .jpg .png)\")\n\ntrain_df = pd.DataFrame(new_rows)\ntrain_df.to_csv(train_csv_path, index=False)\n\nnum_images = len(os.listdir(train_images_dir))\nprint(f\"Number of images in train_images: {num_images}\")\"\"\"","metadata":{"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/aptos2019/datasets/train.csv')\nprint(train_df.shape)\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:36:33.520560Z","iopub.execute_input":"2025-01-15T15:36:33.521328Z","iopub.status.idle":"2025-01-15T15:36:33.564579Z","shell.execute_reply.started":"2025-01-15T15:36:33.521288Z","shell.execute_reply":"2025-01-15T15:36:33.563700Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['diagnosis'].hist()\ntrain_df['diagnosis'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:36:35.789470Z","iopub.execute_input":"2025-01-15T15:36:35.789937Z","iopub.status.idle":"2025-01-15T15:36:36.158904Z","shell.execute_reply.started":"2025-01-15T15:36:35.789898Z","shell.execute_reply":"2025-01-15T15:36:36.157862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def crop_image(image, tol = 7):\n    if image.ndim == 2:\n        mask = image > tol\n        return image[np.ix_(mask.any(1), mask.any(0))]\n    elif image.ndim == 3:\n        image_gray = cv2.cvtColor(image, cv2.COLOR_RGB2GRAY)\n        mask = image_gray > tol\n        check_shape = image[:,:,0][np.ix_(mask.any(1), mask.any(0))].shape[0]\n        if(check_shape == 0):\n            return image\n        else:\n            image1 = image [:,:,0][np.ix_(mask.any(1), mask.any(0))]\n            image2 = image [:,:,1][np.ix_(mask.any(1), mask.any(0))]\n            image3 = image [:,:,2][np.ix_(mask.any(1), mask.any(0))]\n            image = np.stack([image1, image2, image3], axis = -1)\n        return image\n\ndef add_black_padding_and_resize(image, img_size):\n    h, w = image.shape[:2]\n    new_h, new_w = img_size, img_size\n    if h > w:\n        scale_factor = img_size / h\n    else:\n        scale_factor = img_size / w\n\n    new_h = int(h * scale_factor)\n    new_w = int(w * scale_factor)\n    resized_image = cv2.resize(image, (new_w, new_h))\n    top = (img_size - new_h) // 2\n    bottom = img_size - new_h - top\n    left = (img_size - new_w) // 2\n    right = img_size - new_w - left\n\n    padded_resized_image = cv2.copyMakeBorder(resized_image, top, bottom, left, right, cv2.BORDER_CONSTANT, value= 0)\n    return padded_resized_image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:36:39.222527Z","iopub.execute_input":"2025-01-15T15:36:39.222855Z","iopub.status.idle":"2025-01-15T15:36:39.230905Z","shell.execute_reply.started":"2025-01-15T15:36:39.222826Z","shell.execute_reply":"2025-01-15T15:36:39.229881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess_image(image, img_size=224):\n    image = Image.open(image)\n    image = np.array(image)\n    image = crop_image(image)\n    image = add_black_padding_and_resize(image, img_size)\n    image = cv2.cvtColor(image, cv2.COLOR_RGB2GRAY)\n    kernel_horizontal = np.array([[1, 1, 1, 1, 1]], dtype=np.uint8)\n\n    kernel_vertical = np.array([[1],\n                                [1],\n                                [1],\n                                [1],\n                                [1]], dtype=np.uint8)\n\n    blackhat_h = cv2.morphologyEx(image, cv2.MORPH_BLACKHAT, kernel_horizontal)\n    blackhat_v = cv2.morphologyEx(image, cv2.MORPH_BLACKHAT, kernel_vertical)\n    kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (9, 9))\n    tophat = cv2.morphologyEx(image, cv2.MORPH_TOPHAT, kernel)\n    blackhat = cv2.morphologyEx(image, cv2.MORPH_BLACKHAT, kernel)\n    image_tophat = cv2.add(image, tophat)\n    image_blackhat = np.maximum(np.maximum(blackhat_h, blackhat_v), blackhat)\n    image = cv2.subtract(image_tophat, image_blackhat)\n    image =np.stack((image,)*3, axis=-1)\n    image = Image.fromarray(image)\n    return image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:36:43.773754Z","iopub.execute_input":"2025-01-15T15:36:43.774081Z","iopub.status.idle":"2025-01-15T15:36:43.780581Z","shell.execute_reply.started":"2025-01-15T15:36:43.774056Z","shell.execute_reply":"2025-01-15T15:36:43.779683Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"N = train_df.shape[0]\nx_train = np.empty((N, 224, 224, 3), dtype=np.uint8)\nfor i, image_id in enumerate(tqdm(train_df['id_code'])):\n    x_train[i, :, :, :] = preprocess_image(f'/kaggle/input/aptos2019/datasets/train_images/{image_id}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:47:18.879712Z","iopub.execute_input":"2025-01-15T15:47:18.880074Z","iopub.status.idle":"2025-01-15T15:49:45.350642Z","shell.execute_reply.started":"2025-01-15T15:47:18.880044Z","shell.execute_reply":"2025-01-15T15:49:45.349825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/test.csv')\nprint(test_df.shape)\ntest_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:49:47.803862Z","iopub.execute_input":"2025-01-15T15:49:47.804201Z","iopub.status.idle":"2025-01-15T15:49:47.816471Z","shell.execute_reply.started":"2025-01-15T15:49:47.804177Z","shell.execute_reply":"2025-01-15T15:49:47.815316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"N = test_df.shape[0]\nx_test = np.empty((N, 224, 224, 3), dtype=np.uint8)\nfor i, image_id in enumerate(tqdm(test_df['id_code'])):\n    x_test[i, :, :, :] = preprocess_image(f'/kaggle/input/aptos2019-blindness-detection/test_images/{image_id}.png')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:49:50.860966Z","iopub.execute_input":"2025-01-15T15:49:50.861257Z","iopub.status.idle":"2025-01-15T15:51:32.995400Z","shell.execute_reply.started":"2025-01-15T15:49:50.861233Z","shell.execute_reply":"2025-01-15T15:51:32.994647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train = pd.get_dummies(train_df['diagnosis']).values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:51:37.189034Z","iopub.execute_input":"2025-01-15T15:51:37.189317Z","iopub.status.idle":"2025-01-15T15:51:37.194557Z","shell.execute_reply.started":"2025-01-15T15:51:37.189295Z","shell.execute_reply":"2025-01-15T15:51:37.193828Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(x_train.shape)\nprint(y_train.shape)\nprint(x_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:51:39.418493Z","iopub.execute_input":"2025-01-15T15:51:39.418792Z","iopub.status.idle":"2025-01-15T15:51:39.423090Z","shell.execute_reply.started":"2025-01-15T15:51:39.418750Z","shell.execute_reply":"2025-01-15T15:51:39.422196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_train, x_val, y_train, y_val = train_test_split(x_train, y_train, test_size=0.15, random_state=SEED, shuffle=True)\nprint(\"đã chia xong\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:51:42.136602Z","iopub.execute_input":"2025-01-15T15:51:42.136950Z","iopub.status.idle":"2025-01-15T15:51:42.379106Z","shell.execute_reply.started":"2025-01-15T15:51:42.136917Z","shell.execute_reply":"2025-01-15T15:51:42.378111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Kích thước tập huấn luyện:\")\nprint(\"x_train:\", x_train.shape)\nprint(\"y_train:\", y_train.shape)\nprint(\"\\nKích thước tập xác thực:\")\nprint(\"x_val:\", x_val.shape)\nprint(\"y_val:\", y_val.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:51:46.290467Z","iopub.execute_input":"2025-01-15T15:51:46.290757Z","iopub.status.idle":"2025-01-15T15:51:46.296452Z","shell.execute_reply.started":"2025-01-15T15:51:46.290734Z","shell.execute_reply":"2025-01-15T15:51:46.295032Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def print_label_counts(y_data, dataset_name=\"\"):\n  print(f\"Số lượng từng nhãn trong tập {dataset_name}:\")\n  unique_labels, counts = np.unique(y_data, return_counts=True)\n  for label, count in zip(unique_labels, counts):\n    print(f\"Nhãn {label}: {count} mẫu\")\n  print()\n\ny_train_labels = np.argmax(y_train, axis=1)\ny_val_labels = np.argmax(y_val, axis=1)\nprint_label_counts(y_train_labels, \"huấn luyện\")\nprint_label_counts(y_val_labels, \"xác thực\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:51:49.929325Z","iopub.execute_input":"2025-01-15T15:51:49.929655Z","iopub.status.idle":"2025-01-15T15:51:49.937357Z","shell.execute_reply.started":"2025-01-15T15:51:49.929626Z","shell.execute_reply":"2025-01-15T15:51:49.936511Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_datagen():\n    return ImageDataGenerator(\n        horizontal_flip=True,  \n        vertical_flip=True,  \n        rotation_range=60,\n        zoom_range=0.15,  \n        fill_mode='constant',\n        cval=0.,\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:51:54.915717Z","iopub.execute_input":"2025-01-15T15:51:54.916032Z","iopub.status.idle":"2025-01-15T15:51:54.920120Z","shell.execute_reply.started":"2025-01-15T15:51:54.916009Z","shell.execute_reply":"2025-01-15T15:51:54.919379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_generator = train_datagen().flow(x_train, y_train, batch_size=BATCH_SIZE, seed=SEED)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:51:57.336237Z","iopub.execute_input":"2025-01-15T15:51:57.336566Z","iopub.status.idle":"2025-01-15T15:51:57.975983Z","shell.execute_reply.started":"2025-01-15T15:51:57.336534Z","shell.execute_reply":"2025-01-15T15:51:57.975288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def precision(y_true, y_pred):\n    y_true_f = tf.cast(y_true, tf.float32) \n    y_pred_f = tf.cast(y_pred, tf.float32) \n    true_positives = tf.reduce_sum(tf.round(tf.clip_by_value(y_true_f * y_pred_f, 0, 1)))\n    predicted_positives = tf.reduce_sum(tf.round(tf.clip_by_value(y_pred_f, 0, 1)))\n    precision = true_positives / (predicted_positives + tf.keras.backend.epsilon())\n    return precision\n\ndef recall(y_true, y_pred):\n    y_true_f = tf.cast(y_true, tf.float32) \n    y_pred_f = tf.cast(y_pred, tf.float32) \n    true_positives = tf.reduce_sum(tf.round(tf.clip_by_value(y_true_f * y_pred_f, 0, 1)))\n    possible_positives = tf.reduce_sum(tf.round(tf.clip_by_value(y_true_f, 0, 1)))\n    recall = true_positives / (possible_positives + tf.keras.backend.epsilon())\n    return recall\n\ndef fbeta_score(y_true, y_pred, beta=1):\n    if beta < 0:\n        raise ValueError('The lowest choosable beta is zero (only precision).')\n    y_true_f = tf.cast(y_true, tf.float32)\n    p = precision(y_true_f, y_pred)\n    r = recall(y_true_f, y_pred)\n    bb = beta ** 2\n\n    def fbeta():\n        num = (1 + bb) * (p * r)\n        den = bb * p + r + tf.keras.backend.epsilon()\n        return num / den\n\n    fbeta_score = tf.cond(\n        tf.equal(tf.reduce_sum(tf.round(tf.clip_by_value(y_true_f, 0, 1))), 0),\n        lambda: 0.0, \n        fbeta         \n    )\n\n    return fbeta_score\n\ndef fmeasure(y_true, y_pred):\n    return fbeta_score(y_true, y_pred, beta=1)\n\ndef mean_pred(y_true, y_pred):\n    return tf.reduce_mean(tf.cast(y_pred, tf.float32))\n\ndef f1_score(y_true, y_pred):\n    p = precision(y_true, y_pred)\n    r = recall(y_true, y_pred)\n    return 2 * (p * r) / (p + r + tf.keras.backend.epsilon())\n\nprint(\"Evaluation metrics defined ...\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T16:22:32.099711Z","iopub.execute_input":"2025-01-15T16:22:32.100040Z","iopub.status.idle":"2025-01-15T16:22:32.109022Z","shell.execute_reply.started":"2025-01-15T16:22:32.100017Z","shell.execute_reply":"2025-01-15T16:22:32.108248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_model():\n    input_tensor = layers.Input(shape=(224, 224, 3))\n    densenet = DenseNet169(\n        weights='/kaggle/input/densenet169/DenseNet-BC-169-32-no-top.h5',\n        include_top=False,\n        input_shape=(224,224,3)\n    )\n    x = densenet(input_tensor)\n    x = layers.GlobalAveragePooling2D()(x)\n    x = layers.Dropout(DROPOUT)(x)\n    output_tensor = layers.Dense(5, activation='softmax')(x)\n\n    model = Model(inputs=input_tensor, outputs=output_tensor)\n\n    model.compile(\n        loss='categorical_crossentropy',\n        optimizer=Adam(learning_rate=LRATE),\n        metrics=['accuracy',mean_pred, precision, recall, f1_score, fbeta_score, fmeasure]\n    )\n    return model\nprint(\"Build model ...\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T16:30:03.260458Z","iopub.execute_input":"2025-01-15T16:30:03.260790Z","iopub.status.idle":"2025-01-15T16:30:03.266827Z","shell.execute_reply.started":"2025-01-15T16:30:03.260748Z","shell.execute_reply":"2025-01-15T16:30:03.266101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras import layers\nmodel = build_model()\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T16:30:05.518843Z","iopub.execute_input":"2025-01-15T16:30:05.519137Z","iopub.status.idle":"2025-01-15T16:30:08.884615Z","shell.execute_reply.started":"2025-01-15T16:30:05.519116Z","shell.execute_reply":"2025-01-15T16:30:08.883862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class KappaMetrics(Callback):\n    def __init__(self, validation_data):\n        super().__init__()\n        self.val_kappas = []\n        self.validation_data = validation_data\n\n    def on_train_begin(self, logs={}):\n        pass\n\n    def on_epoch_end(self, epoch, logs={}):\n        x_val, y_val = self.validation_data\n\n        y_pred = np.argmax(self.model.predict(x_val), axis=1)\n        y_val_true = np.argmax(y_val, axis=1)\n\n        _val_kappa = cohen_kappa_score(\n            y_val_true,\n            y_pred,\n            weights='quadratic'\n        )\n\n        self.val_kappas.append(_val_kappa)\n        print(f\"Epoch: {epoch+1} val_kappa: {_val_kappa:.4f}\")\n\n        if _val_kappa == max(self.val_kappas):\n            print(\"Validation Kappa has improved. Saving model.\")\n            self.model.save('/kaggle/working/model.h5')\n\n        return","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T16:30:50.458552Z","iopub.execute_input":"2025-01-15T16:30:50.458888Z","iopub.status.idle":"2025-01-15T16:30:50.464617Z","shell.execute_reply.started":"2025-01-15T16:30:50.458861Z","shell.execute_reply":"2025-01-15T16:30:50.463701Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"early_stopping = EarlyStopping(monitor='val_loss', patience=15, verbose=VERBOSE, restore_best_weights=True)\nreduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=5, verbose=VERBOSE, min_lr=MIN_LRATE)\n\nkappa_metrics = KappaMetrics(validation_data=(x_val, y_val))\nhistory = model.fit(\n    data_generator,\n    steps_per_epoch=len(x_train) // BATCH_SIZE,\n    epochs=EPOCHS,\n    validation_data=(x_val, y_val),\n    validation_steps=None,\n    callbacks=[kappa_metrics, early_stopping, reduce_lr],\n    verbose=VERBOSE\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T16:54:59.873356Z","iopub.execute_input":"2025-01-15T16:54:59.873662Z","iopub.status.idle":"2025-01-15T17:01:36.518361Z","shell.execute_reply.started":"2025-01-15T16:54:59.873638Z","shell.execute_reply":"2025-01-15T17:01:36.517628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open('/kaggle/working/history.json', 'w') as f:\n    json.dump(history.history, f)\n\nmax_length = max(len(value) for value in history.history.values())\nfor key, value in history.history.items():\n    if len(value) < max_length:\n        history.history[key] = value + [None] * (max_length - len(value))\nhistory_df = pd.DataFrame(history.history)\nprint(history_df.head(EPOCHS))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T17:01:57.489851Z","iopub.execute_input":"2025-01-15T17:01:57.490170Z","iopub.status.idle":"2025-01-15T17:01:57.504620Z","shell.execute_reply.started":"2025-01-15T17:01:57.490144Z","shell.execute_reply":"2025-01-15T17:01:57.503612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\nimport pandas as pd\nwith open('/kaggle/working/history.json', 'r') as f:\n    loaded_history = json.load(f)\n    \nloaded_history_df = pd.DataFrame(loaded_history)\nprint(loaded_history_df.head())\n\n\"\"\"print(loaded_history['loss'])\nprint(loaded_history['val_accuracy'])\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T17:33:24.911152Z","iopub.execute_input":"2025-01-15T17:33:24.911469Z","iopub.status.idle":"2025-01-15T17:33:24.925381Z","shell.execute_reply.started":"2025-01-15T17:33:24.911433Z","shell.execute_reply":"2025-01-15T17:33:24.924632Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ACTUAL_EPOCHS = len(history.history['loss']) if len(history.history['loss']) < EPOCHS else EPOCHS\nepoch_list = list(range(1, ACTUAL_EPOCHS + 1))\n\nf1, (ax1, ax2, ax3, ax4) = plt.subplots(1, 4, figsize=(24, 4))\nt1 = f1.suptitle('CNN Performance', fontsize=12)\nf1.subplots_adjust(top=0.85, wspace=0.3)\n\nax1.plot(epoch_list, history.history['accuracy'], label='Train Accuracy')\nax1.plot(epoch_list, history.history['val_accuracy'], label='Validation Accuracy')\nax1.set_xticks(np.arange(0, ACTUAL_EPOCHS + 1, 5)) \nax1.set_ylabel('Accuracy %')\nax1.set_xlabel('Epoch')\nax1.set_title('Accuracy')\nl1 = ax1.legend(loc=\"best\")\n\nax2.plot(epoch_list, history.history['loss'], label='Train Loss')\nax2.plot(epoch_list, history.history['val_loss'], label='Validation Loss')\nax2.set_xticks(np.arange(0, ACTUAL_EPOCHS + 1, 5)) \nax2.set_ylabel('Loss')\nax2.set_xlabel('Epoch')\nax2.set_title('Loss')\nl2 = ax2.legend(loc=\"best\")\n\n# Biểu đồ các metrics\nax3.plot(epoch_list, history.history['accuracy'], label='Accuracy')\nax3.plot(epoch_list, history.history['precision'], label='Precision')\nax3.plot(epoch_list, history.history['recall'], label='Recall')\nax3.plot(epoch_list, history.history['f1_score'], label='F1 score')\nax3.plot(epoch_list, history.history['fbeta_score'], label='Fbeta score')\nax3.plot(epoch_list, history.history['fmeasure'], label='FMeasure')\nax3.set_xticks(np.arange(0, ACTUAL_EPOCHS + 1, 5)) \nax3.set_ylabel('Score')\nax3.set_xlabel('Epoch')\nax3.set_title('Performance')\nl3 = ax3.legend(loc=\"best\")\n\nax4.plot(epoch_list, kappa_metrics.val_kappas, label='Kappa score')\nax4.set_xticks(np.arange(0, ACTUAL_EPOCHS + 1, 5)) \nax4.set_ylabel('Score')\nax4.set_xlabel('Epoch')\nax4.set_title('Kappa Metrics')\nl4 = ax4.legend(loc=\"best\")\n\nprint(\"Maximum Kappa Score: %s\" % max(kappa_metrics.val_kappas))\nplt.show() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T17:04:19.390479Z","iopub.execute_input":"2025-01-15T17:04:19.390826Z","iopub.status.idle":"2025-01-15T17:04:19.980575Z","shell.execute_reply.started":"2025-01-15T17:04:19.390801Z","shell.execute_reply":"2025-01-15T17:04:19.979795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report, roc_curve, auc, accuracy_score\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom tensorflow.keras.models import load_model\n\nmodel = load_model('/kaggle/working/model.h5', custom_objects={'precision': precision, 'recall': recall, 'f1_score': f1_score, 'fbeta_score': fbeta_score, 'fmeasure': fmeasure, 'mean_pred': mean_pred}) # Load lại model với các hàm custom metrics\n\ny_pred_prob = model.predict(x_val)\ny_val_true = np.argmax(y_val, axis=1)\ny_pred = np.argmax(y_pred_prob, axis=1)\n\ncm = confusion_matrix(y_val_true, y_pred)\nplt.figure(figsize=(8, 6))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues', cbar=False,\n            xticklabels=range(5), yticklabels=range(5))  # Giả sử có 5 lớp\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.title('Confusion Matrix')\nplt.show()\n\nprint(classification_report(y_val_true, y_pred, target_names=[f'Class {i}' for i in range(5)])) \n\naccuracy = accuracy_score(y_val_true, y_pred)\nprint(f\"Accuracy: {accuracy:.4f}\")\n\ncm = confusion_matrix(y_val_true, y_pred)\nfor i in range(5): \n    TP = cm[i, i]\n    FP = cm[:, i].sum() - TP\n    FN = cm[i, :].sum() - TP\n    TN = cm.sum() - (TP + FP + FN)\n\n    sensitivity = TP / (TP + FN) if (TP + FN) != 0 else 0\n    specificity = TN / (TN + FP) if (TN + FP) != 0 else 0\n\n    print(f\"Class {i}:\")\n    print(f\"  Sensitivity: {sensitivity:.4f}\")\n    print(f\"  Specificity: {specificity:.4f}\")\n\nfpr = dict()\ntpr = dict()\nroc_auc = dict()\nfor i in range(5): \n    fpr[i], tpr[i], _ = roc_curve(y_val[:, i], y_pred_prob[:, i])\n    roc_auc[i] = auc(fpr[i], tpr[i])\n\nfpr[\"micro\"], tpr[\"micro\"], _ = roc_curve(y_val.ravel(), y_pred_prob.ravel())\nroc_auc[\"micro\"] = auc(fpr[\"micro\"], tpr[\"micro\"])\n\nall_fpr = np.unique(np.concatenate([fpr[i] for i in range(5)]))\nmean_tpr = np.zeros_like(all_fpr)\nfor i in range(5):\n    mean_tpr += np.interp(all_fpr, fpr[i], tpr[i])\nmean_tpr /= 5\nfpr[\"macro\"] = all_fpr\ntpr[\"macro\"] = mean_tpr\nroc_auc[\"macro\"] = auc(fpr[\"macro\"], tpr[\"macro\"])\n\nplt.figure(figsize=(8, 6))\nplt.plot(fpr[\"micro\"], tpr[\"micro\"],\n         label=f'micro-average ROC curve (area = {roc_auc[\"micro\"]:.2f})',\n         color='deeppink', linestyle=':', linewidth=4)\n\nplt.plot(fpr[\"macro\"], tpr[\"macro\"],\n         label=f'macro-average ROC curve (area = {roc_auc[\"macro\"]:.2f})',\n         color='navy', linestyle=':', linewidth=4)\n\ncolors = ['aqua', 'darkorange', 'cornflowerblue', 'green', 'red']\nfor i, color in zip(range(5), colors):\n    plt.plot(fpr[i], tpr[i], color=color, lw=2,\n             label=f'ROC curve of class {i} (area = {roc_auc[i]:.2f})')\n\nplt.plot([0, 1], [0, 1], 'k--', lw=2)\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic (ROC) Curves')\nplt.legend(loc=\"lower right\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T17:25:20.872693Z","iopub.execute_input":"2025-01-15T17:25:20.873070Z","iopub.status.idle":"2025-01-15T17:25:41.505649Z","shell.execute_reply.started":"2025-01-15T17:25:20.873043Z","shell.execute_reply":"2025-01-15T17:25:41.504871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.models import load_model\nimport cv2\nimport numpy as np\nfrom PIL import Image\n\nmodel = load_model('/kaggle/working/model.h5', custom_objects={'precision': precision, 'recall': recall, 'f1_score': f1_score, 'fbeta_score': fbeta_score, 'fmeasure': fmeasure, 'mean_pred': mean_pred})\n\ndef crop_image(image, tol=7):\n    if image.ndim == 2:\n        mask = image > tol\n        return image[np.ix_(mask.any(1), mask.any(0))]\n    elif image.ndim == 3:\n        image_gray = cv2.cvtColor(image, cv2.COLOR_RGB2GRAY)\n        mask = image_gray > tol\n        check_shape = image[:, :, 0][np.ix_(mask.any(1), mask.any(0))].shape[0]\n        if (check_shape == 0):\n            return image\n        else:\n            image1 = image[:, :, 0][np.ix_(mask.any(1), mask.any(0))]\n            image2 = image[:, :, 1][np.ix_(mask.any(1), mask.any(0))]\n            image3 = image[:, :, 2][np.ix_(mask.any(1), mask.any(0))]\n            image = np.stack([image1, image2, image3], axis=-1)\n        return image\ndef add_black_padding_and_resize(image, img_size):\n    h, w = image.shape[:2]\n    new_h, new_w = img_size, img_size\n    if h > w:\n        scale_factor = img_size / h\n    else:\n        scale_factor = img_size / w\n\n    new_h = int(h * scale_factor)\n    new_w = int(w * scale_factor)\n    resized_image = cv2.resize(image, (new_w, new_h))\n    top = (img_size - new_h) // 2\n    bottom = img_size - new_h - top\n    left = (img_size - new_w) // 2\n    right = img_size - new_w - left\n\n    padded_resized_image = cv2.copyMakeBorder(resized_image, top, bottom, left, right, cv2.BORDER_CONSTANT, value=0)\n    return padded_resized_image\ndef preprocess_image(image_path, img_size=224):  # Changed parameter to image_path\n    image = Image.open(image_path)  # Open image from path\n    image = np.array(image)\n    image = crop_image(image)\n    image = add_black_padding_and_resize(image, img_size)\n    image = cv2.cvtColor(image, cv2.COLOR_RGB2GRAY)\n    kernel_horizontal = np.array([[1, 1, 1, 1, 1]], dtype=np.uint8)\n    kernel_vertical = np.array([[1], [1], [1], [1], [1]], dtype=np.uint8)\n    kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (9, 9))\n    blackhat_h = cv2.morphologyEx(image, cv2.MORPH_BLACKHAT, kernel_horizontal)\n    blackhat_v = cv2.morphologyEx(image, cv2.MORPH_BLACKHAT, kernel_vertical)\n    tophat = cv2.morphologyEx(image, cv2.MORPH_TOPHAT, kernel)\n    blackhat = cv2.morphologyEx(image, cv2.MORPH_BLACKHAT, kernel)\n\n    image_tophat = cv2.add(image, tophat)\n    image_blackhat = np.maximum(np.maximum(blackhat_h, blackhat_v), blackhat)\n    image = cv2.subtract(image_tophat, image_blackhat)\n    image = np.stack((image,) * 3, axis=-1)\n    return image\n\nimage_path = '/kaggle/input/aptos2019/datasets/train_images/30_right.jpg' \nimg = preprocess_image(image_path)\nimg = img / 255.0 \nimg = img.reshape(1, 224, 224, 3)\n\nprediction = model.predict(img)\nprint(\"Raw prediction:\", prediction)\n\npredicted_class = np.argmax(prediction, axis=1)[0]\nprint(\"Predicted class:\", predicted_class)\n\nclass_labels = {\n    0: \"No DR\",\n    1: \"Mild\",\n    2: \"Moderate\",\n    3: \"Severe\",\n    4: \"Proliferative DR\"\n}\nprint(\"Predicted condition:\", class_labels[predicted_class])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T17:21:14.797839Z","iopub.execute_input":"2025-01-15T17:21:14.798181Z","iopub.status.idle":"2025-01-15T17:21:24.869326Z","shell.execute_reply.started":"2025-01-15T17:21:14.798153Z","shell.execute_reply":"2025-01-15T17:21:24.868616Z"}},"outputs":[],"execution_count":null}]}