{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"},{"sourceId":12166161,"sourceType":"datasetVersion","datasetId":7662504},{"sourceId":12166434,"sourceType":"datasetVersion","datasetId":7662688}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install albumentations","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:07:33.405502Z","iopub.execute_input":"2025-07-03T13:07:33.405691Z","iopub.status.idle":"2025-07-03T13:07:37.230310Z","shell.execute_reply.started":"2025-07-03T13:07:33.405674Z","shell.execute_reply":"2025-07-03T13:07:37.229594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'  # 0 = all logs, 1 = info, 2 = warning, 3 = error only\nos.environ['TF_XLA_FLAGS'] = '--tf_xla_auto_jit=0'\nos.environ[\"PYTORCH_CUDA_ALLOC_CONF\"] = \"expandable_segments:True\"\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\n\nprint('✅')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:07:37.231223Z","iopub.execute_input":"2025-07-03T13:07:37.231462Z","iopub.status.idle":"2025-07-03T13:07:37.236617Z","shell.execute_reply.started":"2025-07-03T13:07:37.231439Z","shell.execute_reply":"2025-07-03T13:07:37.235862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n# import sys\n# # Repository source: https://github.com/qubvel/efficientnet\n# sys.path.append(os.path.abspath('../input/efficientnet/efficientnet-master/efficientnet-master/'))\n# from efficientnet import EfficientNetB3\n# print('✅')","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2025-07-03T13:07:37.237291Z","iopub.execute_input":"2025-07-03T13:07:37.237480Z","iopub.status.idle":"2025-07-03T13:07:37.252711Z","shell.execute_reply.started":"2025-07-03T13:07:37.237464Z","shell.execute_reply":"2025-07-03T13:07:37.252053Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.applications import EfficientNetB3\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:07:37.253434Z","iopub.execute_input":"2025-07-03T13:07:37.253714Z","iopub.status.idle":"2025-07-03T13:07:50.152255Z","shell.execute_reply.started":"2025-07-03T13:07:37.253694Z","shell.execute_reply":"2025-07-03T13:07:50.151481Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Standard dependencies\nimport cv2\nimport time\nimport scipy as sp\nimport numpy as np\nimport random as rn\nimport pandas as pd\nfrom tqdm import tqdm\nfrom PIL import Image\nfrom functools import partial\nimport matplotlib.pyplot as plt\n\n# Machine Learning\nimport os\nimport cv2\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nfrom sklearn.utils import shuffle\nfrom albumentations import Compose, HorizontalFlip, VerticalFlip, RandomBrightnessContrast, Rotate, Resize, RandomGamma\n\nimport tensorflow as tf\nimport keras\nfrom tensorflow.keras import initializers\nfrom tensorflow.keras import regularizers\nfrom tensorflow.keras import constraints\nfrom tensorflow.keras import backend as K\nfrom tensorflow.keras.activations import elu\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Layer\nfrom tensorflow.python.keras.engine.input_spec import InputSpec\n\nfrom tensorflow.keras.utils import get_custom_objects\nfrom tensorflow.keras.callbacks import Callback, EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras.layers import Dense, Conv2D, Flatten, GlobalAveragePooling2D, Dropout\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.metrics import cohen_kappa_score\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:07:50.153060Z","iopub.execute_input":"2025-07-03T13:07:50.153600Z","iopub.status.idle":"2025-07-03T13:07:55.258237Z","shell.execute_reply.started":"2025-07-03T13:07:50.153578Z","shell.execute_reply":"2025-07-03T13:07:55.257337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Path specifications\nKAGGLE_DIR = '../input/aptos2019-blindness-detection/'\ntrain_df_path = KAGGLE_DIR + \"train.csv\"\n#test_df_path = KAGGLE_DIR + 'test.csv'\ntrain_img_path = KAGGLE_DIR + \"train_images/\"\n#test_img_path = KAGGLE_DIR + 'test_images/'\nSAVED_MODEL_NAME = '/kaggle/working/efficientnetb3_best_kappa_model.keras'\n\nsave_img_path = '/kaggle/working/images'\nos.makedirs(save_img_path, exist_ok=True)\naug_img_path = \"/kaggle/working/aug_images/\"\nos.makedirs(aug_img_path, exist_ok=True)\n\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:07:55.261246Z","iopub.execute_input":"2025-07-03T13:07:55.261878Z","iopub.status.idle":"2025-07-03T13:07:55.266614Z","shell.execute_reply.started":"2025-07-03T13:07:55.261852Z","shell.execute_reply":"2025-07-03T13:07:55.266056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# === Đọc dữ liệu gốc ===\ntrain_df = pd.read_csv(train_df_path)\n\n# === Chia dữ liệu: 80% train_val, 20% test ===\ntrain_val_df, test_df = train_test_split(\n    train_df,\n    test_size=0.2,\n    stratify=train_df['diagnosis'],\n    random_state=42\n)\n\n# === Thêm đuôi .png vào cột id_code ===\ntrain_val_df['id_code'] = train_val_df['id_code'].astype(str) + \".png\"\ntest_df['id_code'] = test_df['id_code'].astype(str) + \".png\"\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:07:55.267289Z","iopub.execute_input":"2025-07-03T13:07:55.267835Z","iopub.status.idle":"2025-07-03T13:07:55.319524Z","shell.execute_reply.started":"2025-07-03T13:07:55.267792Z","shell.execute_reply":"2025-07-03T13:07:55.318815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Image IDs and Labels (TRAIN + VAL)\")\n# Add extension to id_code\n#train_val_df['id_code'] = train_val_df['id_code'] + \".png\"\nprint(f\"Training images: {train_val_df.shape[0]}\")\ndisplay(train_val_df.head())\n\nprint(\"Image IDs (TEST)\")\n# Add extension to id_code\n#test_df['id_code'] = test_df['id_code'] + \".png\"\nprint(f\"Testing Images: {test_df.shape[0]}\")\ndisplay(test_df.head())\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:07:55.320327Z","iopub.execute_input":"2025-07-03T13:07:55.320565Z","iopub.status.idle":"2025-07-03T13:07:55.342729Z","shell.execute_reply.started":"2025-07-03T13:07:55.320542Z","shell.execute_reply":"2025-07-03T13:07:55.342188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === Lưu file ra /kaggle/working ===\ntrain_val_path = '/kaggle/working/train_val.csv'\ntest_path = '/kaggle/working/test.csv'\n\ntrain_val_df.to_csv(train_val_path, index=False)\ntest_df.to_csv(test_path, index=False)\n\nprint(f\"✅ Saved train_val.csv to {train_val_path}\")\nprint(f\"✅ Saved test.csv to {test_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:07:55.343494Z","iopub.execute_input":"2025-07-03T13:07:55.343741Z","iopub.status.idle":"2025-07-03T13:07:55.355777Z","shell.execute_reply.started":"2025-07-03T13:07:55.343719Z","shell.execute_reply":"2025-07-03T13:07:55.355098Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Specify image size\nIMG_WIDTH = 224\nIMG_HEIGHT = 224\nCHANNELS = 3\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:07:55.356539Z","iopub.execute_input":"2025-07-03T13:07:55.356774Z","iopub.status.idle":"2025-07-03T13:07:55.361658Z","shell.execute_reply.started":"2025-07-03T13:07:55.356753Z","shell.execute_reply":"2025-07-03T13:07:55.361156Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Metric QWK","metadata":{}},{"cell_type":"code","source":"# def get_preds_and_labels(model, generator):\n#     \"\"\"\n#     Get predictions and labels from the generator\n    \n#     :param model: A Keras model object\n#     :param generator: A Keras ImageDataGenerator object\n    \n#     :return: A tuple with two Numpy Arrays. One containing the predictions\n#     and one containing the labels\n#     \"\"\"\n#     preds = []\n#     labels = []\n#     for _ in range(int(np.ceil(generator.samples / BATCH_SIZE))):\n#         x, y = next(generator)\n#         preds.append(model.predict(x))\n#         labels.append(y)\n#     # Flatten list of numpy arrays\n#     return np.concatenate(preds).ravel(), np.concatenate(labels)#.ravel()\n# print('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:07:55.362497Z","iopub.execute_input":"2025-07-03T13:07:55.362660Z","iopub.status.idle":"2025-07-03T13:07:55.376197Z","shell.execute_reply.started":"2025-07-03T13:07:55.362642Z","shell.execute_reply":"2025-07-03T13:07:55.375590Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_preds_and_labels(model, generator):\n    \"\"\"\n    Get predictions and true labels from a data generator.\n    \n    :param model: A compiled Keras model\n    :param generator: A data generator (e.g., ImageDataGenerator or custom generator)\n    \n    :return: preds (softmax output), labels (as integer class indices)\n    \"\"\"\n    steps = int(np.ceil(generator.samples / generator.batch_size))\n    preds = model.predict(generator, steps=steps, verbose=0)\n\n    try:\n        # Nếu có .classes (ImageDataGenerator): trả về trực tiếp\n        labels = generator.classes\n    except AttributeError:\n        # Với custom generator: thu thập labels\n        labels_list = []\n        for i in range(steps):\n            _, y_batch = next(generator)\n            labels_list.append(y_batch)\n\n        labels = np.concatenate(labels_list)\n\n        # Ép về class index nếu là one-hot\n        if isinstance(labels[0], (np.ndarray, list)):  # mỗi label là vector\n            labels = np.argmax(labels, axis=1)\n\n    return preds, labels\n\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:07:55.376866Z","iopub.execute_input":"2025-07-03T13:07:55.377098Z","iopub.status.idle":"2025-07-03T13:07:55.393928Z","shell.execute_reply.started":"2025-07-03T13:07:55.377077Z","shell.execute_reply":"2025-07-03T13:07:55.393372Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Metrics(Callback):\n    def __init__(self, val_generator):\n        super().__init__()\n        self.val_generator = val_generator\n        self.val_kappas = []\n\n    def on_train_begin(self, logs=None):\n        self.val_kappas = []\n\n    def on_epoch_end(self, epoch, logs=None):\n        y_pred, labels = get_preds_and_labels(self.model, self.val_generator)\n        y_pred = y_pred.argmax(axis=1) if y_pred.ndim > 1 else y_pred\n\n        # Không dùng ndim cho labels nữa vì đã xử lý trong get_preds_and_labels\n        # labels đã chắc chắn là vector class index rồi\n\n        val_kappa = cohen_kappa_score(labels, y_pred, weights='quadratic')\n        self.val_kappas.append(val_kappa)\n        print(f\"val_kappa: {round(val_kappa, 4)}\")\n\n        if val_kappa == max(self.val_kappas):\n            print(\"Validation Kappa has improved. Saving model.\")\n\n        if logs is not None:\n            logs['val_kappa'] = val_kappa\n        return\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:07:55.394607Z","iopub.execute_input":"2025-07-03T13:07:55.394880Z","iopub.status.idle":"2025-07-03T13:07:55.407367Z","shell.execute_reply.started":"2025-07-03T13:07:55.394859Z","shell.execute_reply":"2025-07-03T13:07:55.406650Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"# Label distribution\ncounts = train_val_df['diagnosis'].value_counts().sort_index()\nax = counts.plot(kind=\"bar\", figsize=(12, 5), rot=0, color='skyblue')\n\nax.set_title(\"Label Distribution\", fontsize=15, fontweight='bold')\nax.set_xlabel(\"Class\", fontsize=13)\nax.set_ylabel(\"Count\", fontsize=13)\n\n# Hiển thị số trên đầu mỗi cột\nfor i, value in enumerate(counts.values):\n    ax.text(i, value + 10, str(value), ha='center', va='bottom', fontsize=11, fontweight='bold')\n\nplt.tight_layout()\nplt.savefig(os.path.join(save_img_path,'label_distribution.png'))\nplt.show()\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:07:55.408036Z","iopub.execute_input":"2025-07-03T13:07:55.408265Z","iopub.status.idle":"2025-07-03T13:07:55.791000Z","shell.execute_reply.started":"2025-07-03T13:07:55.408244Z","shell.execute_reply":"2025-07-03T13:07:55.790344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def check_duplicates(train_val_df, test_df):\n    train_set = set(train_val_df['id_code'])\n    test_set = set(test_df['id_code'])\n\n    print(\"🔍 Duplicates between train & test:\", len(train_set & test_set))\n\ncheck_duplicates(train_val_df, test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:07:55.791743Z","iopub.execute_input":"2025-07-03T13:07:55.792062Z","iopub.status.idle":"2025-07-03T13:07:55.797439Z","shell.execute_reply.started":"2025-07-03T13:07:55.792019Z","shell.execute_reply":"2025-07-03T13:07:55.796733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from collections import Counter\nimport matplotlib.pyplot as plt\n\nfor name, df in [(\"Train + Val\", train_val_df), (\"Test\", test_df)]:\n    counts = Counter(df['diagnosis'])\n    labels = list(counts.keys())\n    values = list(counts.values())\n    \n    plt.figure(figsize=(8, 4))\n    bars = plt.bar(labels, values, color='skyblue')\n    plt.title(f\"Label Distribution - {name}\")\n    plt.xlabel(\"Class\")\n    plt.ylabel(\"Count\")\n    \n    # Hiển thị giá trị trên từng cột\n    for i, bar in enumerate(bars):\n        height = bar.get_height()\n        plt.text(bar.get_x() + bar.get_width() / 2, height + 10, str(int(height)), \n                 ha='center', va='bottom', fontsize=11, fontweight='bold')\n    \n    plt.tight_layout()\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:07:55.798079Z","iopub.execute_input":"2025-07-03T13:07:55.798238Z","iopub.status.idle":"2025-07-03T13:07:56.128134Z","shell.execute_reply.started":"2025-07-03T13:07:55.798226Z","shell.execute_reply":"2025-07-03T13:07:56.127512Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 5, figsize=(15, 6))\nfor i in range(5):\n    sample = train_val_df[train_val_df['diagnosis'] == i].sample(1)\n    image_name = sample['id_code'].item()\n    if not image_name.endswith('.png'):\n        image_name += '.png'\n\n    img_path = os.path.join(train_img_path, image_name)\n    X = cv2.imread(img_path)\n\n    if X is None:\n        print(f\"[❌] Cannot read image: {img_path}\")\n        ax[i].text(0.5, 0.5, 'Image not found', ha='center', va='center', fontsize=12)\n        ax[i].axis('off')\n        continue\n\n    X = cv2.cvtColor(X, cv2.COLOR_BGR2RGB)\n    ax[i].set_title(f\"Image: {image_name}\\n Label = {sample['diagnosis'].item()}\", \n                    weight='bold', fontsize=10)\n    ax[i].axis('off')\n    ax[i].imshow(X)\n\nplt.tight_layout()\nplt.savefig(os.path.join(save_img_path, 'sample_each_class.png'), dpi=300)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:07:56.128791Z","iopub.execute_input":"2025-07-03T13:07:56.129021Z","iopub.status.idle":"2025-07-03T13:08:03.833239Z","shell.execute_reply.started":"2025-07-03T13:07:56.129005Z","shell.execute_reply":"2025-07-03T13:08:03.832503Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"def crop_image_from_gray(img, tol=7):\n    \"\"\"\n    Applies masks to the orignal image and \n    returns the a preprocessed image with \n    3 channels\n    \n    :param img: A NumPy Array that will be cropped\n    :param tol: The tolerance used for masking\n    \n    :return: A NumPy array containing the cropped image\n    \"\"\"\n    # If for some reason we only have two channels\n    if img.ndim == 2:\n        mask = img > tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    # If we have a normal RGB images\n    elif img.ndim == 3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img > tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n            img = np.stack([img1,img2,img3],axis=-1)\n        return img\n\ndef preprocess_image(image, sigmaX=10):\n    \"\"\"\n    The whole preprocessing pipeline:\n    1. Read in image\n    2. Apply masks\n    3. Resize image to desired size\n    4. Add Gaussian noise to increase Robustness\n    \n    :param img: A NumPy Array that will be cropped\n    :param sigmaX: Value used for add GaussianBlur to the image\n    \n    :return: A NumPy array containing the preprocessed image\n    \"\"\"\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    image = crop_image_from_gray(image)\n    image = cv2.resize(image, (IMG_WIDTH, IMG_HEIGHT))\n    image = cv2.addWeighted (image,4, cv2.GaussianBlur(image, (0,0) ,sigmaX), -4, 128)\n    return image\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:08:03.834072Z","iopub.execute_input":"2025-07-03T13:08:03.834342Z","iopub.status.idle":"2025-07-03T13:08:03.843440Z","shell.execute_reply.started":"2025-07-03T13:08:03.834325Z","shell.execute_reply":"2025-07-03T13:08:03.842695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 5, figsize=(15, 6))\nfor i in range(5):\n    sample = train_val_df[train_val_df['diagnosis'] == i].sample(1)\n    image_name = sample['id_code'].item()\n    if not image_name.endswith('.png'):\n        image_name += '.png'\n\n    image_path = os.path.join(train_img_path, image_name)\n    img_raw = cv2.imread(image_path)\n\n    if img_raw is None:\n        print(f\"[❌] Cannot read image: {image_path}\")\n        ax[i].text(0.5, 0.5, 'Image not found', ha='center', va='center', fontsize=12)\n        ax[i].axis('off')\n        continue\n\n    X = preprocess_image(img_raw)\n    ax[i].set_title(f\"Image: {image_name}\\nLabel = {sample['diagnosis'].item()}\", \n                    weight='bold', fontsize=10)\n    ax[i].axis('off')\n    ax[i].imshow(X)\n\nplt.tight_layout()\nplt.savefig(os.path.join(save_img_path, 'sample_each_class_preprocessed.png'), dpi=300)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:08:03.844320Z","iopub.execute_input":"2025-07-03T13:08:03.844782Z","iopub.status.idle":"2025-07-03T13:08:06.242600Z","shell.execute_reply.started":"2025-07-03T13:08:03.844756Z","shell.execute_reply":"2025-07-03T13:08:06.241863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from collections import Counter\n\n# === In phân phối và kế hoạch augment ===\nlabel_counts = Counter(train_val_df['diagnosis'])\nmax_count = max(label_counts.values())\naugment_plan = {cls: max_count - count for cls, count in label_counts.items() if count < max_count}\n\nprint(f\"[INFO] Original label counts:\")\nfor k, v in sorted(label_counts.items()):\n    print(f\"    Class {k}: {v} samples\")\nprint(f\"[INFO] Auto augment plan:\")\nfor k, v in sorted(augment_plan.items()):\n    print(f\"    Class {k}: need {v} augmented samples\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:08:06.243667Z","iopub.execute_input":"2025-07-03T13:08:06.244186Z","iopub.status.idle":"2025-07-03T13:08:06.251071Z","shell.execute_reply.started":"2025-07-03T13:08:06.244162Z","shell.execute_reply":"2025-07-03T13:08:06.250453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, cv2\nimport numpy as np\nimport pandas as pd\nfrom albumentations import (\n    HorizontalFlip, VerticalFlip, Rotate, RandomBrightnessContrast,\n    RandomGamma, Resize, Compose\n)\nfrom sklearn.utils import shuffle\nfrom collections import Counter\nfrom tqdm import tqdm\n\n# === Augmentation pipeline ===\naugmenter = Compose([\n    HorizontalFlip(p=0.5),\n    VerticalFlip(p=0.5),\n    Rotate(limit=30, p=0.5),\n    RandomBrightnessContrast(p=0.5),\n    RandomGamma(p=0.5),\n    Resize(224, 224)\n])\n\n# === Hàm augment một class ===\ndef augment_class_images(class_df, label, n_to_generate, augmenter, save_dir):\n    new_records = []\n    samples = class_df.copy().reset_index(drop=True)\n    i = 0\n    while len(new_records) < n_to_generate:\n        row = samples.sample(1).iloc[0]\n        # Chuẩn hóa tên file ảnh\n        img_filename = row['id_code'].replace('.png', '') + '.png'\n        img_path = os.path.join(train_img_path, img_filename)\n\n        img = cv2.imread(img_path)\n        if img is None:\n            print(f\"[WARNING] Cannot read image: {img_path}\")\n            continue\n\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        img = cv2.resize(img, (224, 224))\n        aug_img = augmenter(image=img)['image']\n\n        new_name = img_filename.replace(\".png\", f\"_aug{i}.png\")\n        save_path = os.path.join(save_dir, new_name)\n        cv2.imwrite(save_path, cv2.cvtColor(aug_img, cv2.COLOR_RGB2BGR))\n\n        new_records.append({'id_code': new_name, 'diagnosis': label})\n        i += 1\n    return new_records\n\n# === Hàm tổng augment toàn bộ tập train để các lớp bằng nhau ===\ndef augment_dataset_to_balance(df, augmenter, save_dir):\n    label_counts = Counter(df['diagnosis'])\n    max_count = max(label_counts.values())\n\n    print(f\"[INFO] Balancing to {max_count} samples per class\")\n\n    augmented_records = []\n\n    for label in sorted(df['diagnosis'].unique()):\n        class_df = df[df['diagnosis'] == label].reset_index(drop=True)\n        \n        for _, row in class_df.iterrows():\n            # Chuẩn hóa tên file ảnh\n            img_filename = row['id_code'].replace('.png', '') + '.png'\n            img_path = os.path.join(train_img_path, img_filename)\n\n            img = cv2.imread(img_path)\n            if img is None:\n                print(f\"[WARNING] Cannot read image: {img_path}\")\n                continue\n\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n            img = cv2.resize(img, (224, 224))\n            save_path = os.path.join(save_dir, img_filename)\n            cv2.imwrite(save_path, cv2.cvtColor(img, cv2.COLOR_RGB2BGR))\n\n            augmented_records.append({'id_code': img_filename, 'diagnosis': label})\n        \n        # Augment nếu cần\n        n_current = len(class_df)\n        n_to_generate = max_count - n_current\n        if n_to_generate > 0:\n            aug_records = augment_class_images(class_df, label, n_to_generate, augmenter, save_dir)\n            augmented_records.extend(aug_records)\n\n    final_df = pd.DataFrame(augmented_records)\n    return shuffle(final_df, random_state=42)\n\n# === Thực thi augment ===\naugmented_df = augment_dataset_to_balance(train_val_df, augmenter, aug_img_path)\nprint(f\"✅ Augment complete. New balanced dataset shape: {augmented_df.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:08:06.255264Z","iopub.execute_input":"2025-07-03T13:08:06.255725Z","iopub.status.idle":"2025-07-03T13:22:49.130980Z","shell.execute_reply.started":"2025-07-03T13:08:06.255703Z","shell.execute_reply":"2025-07-03T13:22:49.130228Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Chia lại từ augmented_df\ntrain_df, val_df = train_test_split(\n    augmented_df,\n    test_size=0.15,  # bạn muốn 15% cho validation\n    stratify=augmented_df['diagnosis'],\n    random_state=42\n)\n\nprint(f\"Train set: {len(train_df)} samples\")\nprint(f\"Val set:   {len(val_df)} samples\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:22:49.131741Z","iopub.execute_input":"2025-07-03T13:22:49.131983Z","iopub.status.idle":"2025-07-03T13:22:49.140873Z","shell.execute_reply.started":"2025-07-03T13:22:49.131965Z","shell.execute_reply":"2025-07-03T13:22:49.140280Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['id_code'] = train_df['id_code'].apply(lambda x: x if x.endswith('.png') else f\"{x}.png\")\nval_df['id_code'] = val_df['id_code'].apply(lambda x: x if x.endswith('.png') else f\"{x}.png\")\ntest_df['id_code'] = test_df['id_code'].apply(lambda x: x if x.endswith('.png') else f\"{x}.png\")\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:22:49.141474Z","iopub.execute_input":"2025-07-03T13:22:49.141713Z","iopub.status.idle":"2025-07-03T13:22:49.156991Z","shell.execute_reply.started":"2025-07-03T13:22:49.141698Z","shell.execute_reply":"2025-07-03T13:22:49.156257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\nmissing = []\nfor fname in test_df['id_code']:\n    if not Path(os.path.join(train_img_path, fname)).exists():\n        missing.append(fname)\n\nprint(f\"[❌] Missing {len(missing)} images in test set.\")\nif missing:\n    print(\"First few missing:\", missing[:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:22:49.157833Z","iopub.execute_input":"2025-07-03T13:22:49.158052Z","iopub.status.idle":"2025-07-03T13:22:50.042335Z","shell.execute_reply.started":"2025-07-03T13:22:49.158038Z","shell.execute_reply":"2025-07-03T13:22:50.041708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['diagnosis'] = train_df['diagnosis'].astype(str)\nval_df['diagnosis'] = val_df['diagnosis'].astype(str)\ntest_df['diagnosis'] = test_df['diagnosis'].astype(str)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:22:50.043100Z","iopub.execute_input":"2025-07-03T13:22:50.043334Z","iopub.status.idle":"2025-07-03T13:22:50.049951Z","shell.execute_reply.started":"2025-07-03T13:22:50.043308Z","shell.execute_reply":"2025-07-03T13:22:50.049263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# We use a small batch size so we can handle large images easily\nBATCH_SIZE = 32 # 32 hoặc 64\n\ntrain_datagen = ImageDataGenerator(\n    preprocessing_function=preprocess_image,\n    rescale=1/128.)\n\n# Use the dataframe to define train and validation generators\ntrain_generator = train_datagen.flow_from_dataframe(train_df, \n                                                    x_col='id_code', \n                                                    y_col='diagnosis',\n                                                    directory = aug_img_path,\n                                                    target_size=(IMG_WIDTH, IMG_HEIGHT),\n                                                    batch_size=BATCH_SIZE,\n                                                    class_mode='categorical')\n\nval_datagen = ImageDataGenerator(\n    preprocessing_function=preprocess_image,\n    rescale=1/128.)\n    \nval_generator = val_datagen.flow_from_dataframe(val_df, \n                                                  x_col='id_code', \n                                                  y_col='diagnosis',\n                                                  directory = aug_img_path,\n                                                  target_size=(IMG_WIDTH, IMG_HEIGHT),\n                                                  batch_size=BATCH_SIZE,\n                                                  class_mode='categorical')\n\ntest_datagen = ImageDataGenerator(\n    preprocessing_function=preprocess_image,\n    rescale=1/128.)\n\ntest_generator = test_datagen.flow_from_dataframe(\n    test_df,\n    directory=train_img_path,  \n    x_col='id_code',\n    y_col='diagnosis',\n    target_size=(IMG_WIDTH, IMG_HEIGHT),\n    batch_size=BATCH_SIZE,\n    class_mode='categorical',\n    shuffle=False\n)\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:22:50.050579Z","iopub.execute_input":"2025-07-03T13:22:50.050765Z","iopub.status.idle":"2025-07-03T13:22:50.132202Z","shell.execute_reply.started":"2025-07-03T13:22:50.050751Z","shell.execute_reply":"2025-07-03T13:22:50.131685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Đếm số lượng ảnh từng lớp trước và sau\nlabel_counts_before = train_val_df['diagnosis'].value_counts().sort_index()\nlabel_counts_after = train_df['diagnosis'].value_counts().sort_index()\n\n# Tên lớp\nlabel_names = ['No DR', 'Mild', 'Moderate', 'Severe', 'Proliferative DR']\n\n# Vẽ biểu đồ\nfig, axes = plt.subplots(1, 2, figsize=(14, 6))\n\n# === BEFORE AUGMENTATION ===\nbars0 = axes[0].bar(label_names, label_counts_before.values, color='skyblue')\naxes[0].set_title(\"Before Augmentation\", fontsize=15, fontweight='bold')\naxes[0].set_xlabel(\"Class\", fontsize=13)\naxes[0].set_ylabel(\"Count\", fontsize=13)\naxes[0].tick_params(axis='x', labelsize=11)\naxes[0].tick_params(axis='y', labelsize=11)\nfor bar in bars0:\n    height = bar.get_height()\n    axes[0].text(bar.get_x() + bar.get_width()/2, height + 10, f'{int(height)}',\n                 ha='center', va='bottom', fontsize=10, fontweight='bold')\n\n# === AFTER AUGMENTATION ===\nbars1 = axes[1].bar(label_names, label_counts_after.values, color='lightgreen')\naxes[1].set_title(\"After Augmentation\", fontsize=15, fontweight='bold')\naxes[1].set_xlabel(\"Class\", fontsize=13)\naxes[1].set_ylabel(\"Count\", fontsize=13)\naxes[1].tick_params(axis='x', labelsize=11)\naxes[1].tick_params(axis='y', labelsize=11)\nfor bar in bars1:\n    height = bar.get_height()\n    axes[1].text(bar.get_x() + bar.get_width()/2, height + 10, f'{int(height)}',\n                 ha='center', va='bottom', fontsize=10, fontweight='bold')\n\nplt.tight_layout()\n# ✅ Lưu biểu đồ\nplt.savefig(os.path.join(save_img_path, 'augmentation_distribution.png'))\nplt.show()\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:22:50.132928Z","iopub.execute_input":"2025-07-03T13:22:50.133182Z","iopub.status.idle":"2025-07-03T13:22:50.610365Z","shell.execute_reply.started":"2025-07-03T13:22:50.133160Z","shell.execute_reply":"2025-07-03T13:22:50.609604Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Modelling","metadata":{}},{"cell_type":"code","source":"class GroupNormalization(Layer):\n    \"\"\"Group normalization layer\n    Group Normalization divides the channels into groups and computes within each group\n    the mean and variance for normalization. GN's computation is independent of batch sizes,\n    and its accuracy is stable in a wide range of batch sizes\n    # Arguments\n        groups: Integer, the number of groups for Group Normalization.\n        axis: Integer, the axis that should be normalized\n            (typically the features axis).\n            For instance, after a `Conv2D` layer with\n            `data_format=\"channels_first\"`,\n            set `axis=1` in `BatchNormalization`.\n        epsilon: Small float added to variance to avoid dividing by zero.\n        center: If True, add offset of `beta` to normalized tensor.\n            If False, `beta` is ignored.\n        scale: If True, multiply by `gamma`.\n            If False, `gamma` is not used.\n            When the next layer is linear (also e.g. `nn.relu`),\n            this can be disabled since the scaling\n            will be done by the next layer.\n        beta_initializer: Initializer for the beta weight.\n        gamma_initializer: Initializer for the gamma weight.\n        beta_regularizer: Optional regularizer for the beta weight.\n        gamma_regularizer: Optional regularizer for the gamma weight.\n        beta_constraint: Optional constraint for the beta weight.\n        gamma_constraint: Optional constraint for the gamma weight.\n    # Input shape\n        Arbitrary. Use the keyword argument `input_shape`\n        (tuple of integers, does not include the samples axis)\n        when using this layer as the first layer in a model.\n    # Output shape\n        Same shape as input.\n    # References\n        - [Group Normalization](https://arxiv.org/abs/1803.08494)\n    \"\"\"\n\n    def __init__(self,\n                 groups=32,\n                 axis=-1,\n                 epsilon=1e-5,\n                 center=True,\n                 scale=True,\n                 beta_initializer='zeros',\n                 gamma_initializer='ones',\n                 beta_regularizer=None,\n                 gamma_regularizer=None,\n                 beta_constraint=None,\n                 gamma_constraint=None,\n                 **kwargs):\n        super(GroupNormalization, self).__init__(**kwargs)\n        self.supports_masking = True\n        self.groups = groups\n        self.axis = axis\n        self.epsilon = epsilon\n        self.center = center\n        self.scale = scale\n        self.beta_initializer = initializers.get(beta_initializer)\n        self.gamma_initializer = initializers.get(gamma_initializer)\n        self.beta_regularizer = regularizers.get(beta_regularizer)\n        self.gamma_regularizer = regularizers.get(gamma_regularizer)\n        self.beta_constraint = constraints.get(beta_constraint)\n        self.gamma_constraint = constraints.get(gamma_constraint)\n\n    def build(self, input_shape):\n        dim = input_shape[self.axis]\n\n        if dim is None:\n            raise ValueError('Axis ' + str(self.axis) + ' of '\n                             'input tensor should have a defined dimension '\n                             'but the layer received an input with shape ' +\n                             str(input_shape) + '.')\n\n        if dim < self.groups:\n            raise ValueError('Number of groups (' + str(self.groups) + ') cannot be '\n                             'more than the number of channels (' +\n                             str(dim) + ').')\n\n        if dim % self.groups != 0:\n            raise ValueError('Number of groups (' + str(self.groups) + ') must be a '\n                             'multiple of the number of channels (' +\n                             str(dim) + ').')\n\n        self.input_spec = InputSpec(ndim=len(input_shape),\n                                    axes={self.axis: dim})\n        shape = (dim,)\n\n        if self.scale:\n            self.gamma = self.add_weight(shape=shape,\n                                         name='gamma',\n                                         initializer=self.gamma_initializer,\n                                         regularizer=self.gamma_regularizer,\n                                         constraint=self.gamma_constraint)\n        else:\n            self.gamma = None\n        if self.center:\n            self.beta = self.add_weight(shape=shape,\n                                        name='beta',\n                                        initializer=self.beta_initializer,\n                                        regularizer=self.beta_regularizer,\n                                        constraint=self.beta_constraint)\n        else:\n            self.beta = None\n        self.built = True\n\n    def call(self, inputs, **kwargs):\n        input_shape = K.int_shape(inputs)\n        tensor_input_shape = K.shape(inputs)\n\n        # Prepare broadcasting shape.\n        reduction_axes = list(range(len(input_shape)))\n        del reduction_axes[self.axis]\n        broadcast_shape = [1] * len(input_shape)\n        broadcast_shape[self.axis] = input_shape[self.axis] // self.groups\n        broadcast_shape.insert(1, self.groups)\n\n        reshape_group_shape = K.shape(inputs)\n        group_axes = [reshape_group_shape[i] for i in range(len(input_shape))]\n        group_axes[self.axis] = input_shape[self.axis] // self.groups\n        group_axes.insert(1, self.groups)\n\n        # reshape inputs to new group shape\n        group_shape = [group_axes[0], self.groups] + group_axes[2:]\n        group_shape = K.stack(group_shape)\n        inputs = K.reshape(inputs, group_shape)\n\n        group_reduction_axes = list(range(len(group_axes)))\n        group_reduction_axes = group_reduction_axes[2:]\n\n        mean = K.mean(inputs, axis=group_reduction_axes, keepdims=True)\n        variance = K.var(inputs, axis=group_reduction_axes, keepdims=True)\n\n        inputs = (inputs - mean) / (K.sqrt(variance + self.epsilon))\n\n        # prepare broadcast shape\n        inputs = K.reshape(inputs, group_shape)\n        outputs = inputs\n\n        # In this case we must explicitly broadcast all parameters.\n        if self.scale:\n            broadcast_gamma = K.reshape(self.gamma, broadcast_shape)\n            outputs = outputs * broadcast_gamma\n\n        if self.center:\n            broadcast_beta = K.reshape(self.beta, broadcast_shape)\n            outputs = outputs + broadcast_beta\n\n        outputs = K.reshape(outputs, tensor_input_shape)\n\n        return outputs\n\n    def get_config(self):\n        config = {\n            'groups': self.groups,\n            'axis': self.axis,\n            'epsilon': self.epsilon,\n            'center': self.center,\n            'scale': self.scale,\n            'beta_initializer': initializers.serialize(self.beta_initializer),\n            'gamma_initializer': initializers.serialize(self.gamma_initializer),\n            'beta_regularizer': regularizers.serialize(self.beta_regularizer),\n            'gamma_regularizer': regularizers.serialize(self.gamma_regularizer),\n            'beta_constraint': constraints.serialize(self.beta_constraint),\n            'gamma_constraint': constraints.serialize(self.gamma_constraint)\n        }\n        base_config = super(GroupNormalization, self).get_config()\n        return dict(list(base_config.items()) + list(config.items()))\n\n    def compute_output_shape(self, input_shape):\n        return input_shape\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:22:50.611294Z","iopub.execute_input":"2025-07-03T13:22:50.611560Z","iopub.status.idle":"2025-07-03T13:22:50.626845Z","shell.execute_reply.started":"2025-07-03T13:22:50.611533Z","shell.execute_reply":"2025-07-03T13:22:50.626130Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load in EfficientNetB3\neffnet = EfficientNetB3(weights='imagenet',\n                        include_top=False,\n                        input_shape=(IMG_WIDTH, IMG_HEIGHT, CHANNELS))\n# effnet.load_weights('../input/efficientnet-keras-weights-b0b5/efficientnet-b3_imagenet_1000_notop.h5')\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:22:50.627515Z","iopub.execute_input":"2025-07-03T13:22:50.627717Z","iopub.status.idle":"2025-07-03T13:22:54.438791Z","shell.execute_reply.started":"2025-07-03T13:22:50.627702Z","shell.execute_reply":"2025-07-03T13:22:54.438185Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Replace all Batch Normalization layers by Group Normalization layers\nfor i, layer in enumerate(effnet.layers):\n    if \"batch_normalization\" in layer.name:\n        effnet.layers[i] = GroupNormalization(groups=32, axis=-1, epsilon=0.00001)\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:22:54.439461Z","iopub.execute_input":"2025-07-03T13:22:54.439655Z","iopub.status.idle":"2025-07-03T13:22:54.444603Z","shell.execute_reply.started":"2025-07-03T13:22:54.439639Z","shell.execute_reply":"2025-07-03T13:22:54.444038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.utils import class_weight\nfrom tensorflow.keras.saving import register_keras_serializable\nimport tensorflow as tf\nimport numpy as np\n\n# 1. Tính toán class weights từ dữ liệu huấn luyện\nclass_labels = train_df['diagnosis'].values\nclasses = np.unique(class_labels)\nraw_class_weights = class_weight.compute_class_weight(\n    class_weight='balanced',\n    classes=classes,\n    y=class_labels\n)\n\n# 2. Boost các lớp hiếm bằng lũy thừa\nboost_power = 1.5\nalpha_boosted = raw_class_weights ** boost_power\n\n# 3. Chuẩn hóa alpha về tổng = số lớp\nalpha_values = (alpha_boosted / np.sum(alpha_boosted)) * len(classes)\nprint(\"✅ Boosted alpha_values:\", alpha_values)\n\n# 4. Focal Loss với label_smoothing và gamma cao\n@register_keras_serializable()\ndef focal_loss_with_custom_alpha(alpha_list, gamma=2.0, label_smoothing=0.1):\n    alpha = tf.constant(alpha_list, dtype=tf.float32)\n\n    def loss_fn(y_true, y_pred):\n        y_pred = tf.clip_by_value(y_pred, 1e-7, 1.0)\n\n        # Vì generator output là one-hot → dùng luôn\n        y_true_onehot = tf.cast(y_true, tf.float32)\n\n        # Label smoothing\n        num_classes = tf.cast(tf.shape(y_pred)[-1], tf.float32)\n        y_true_smooth = y_true_onehot * (1.0 - label_smoothing) + (label_smoothing / num_classes)\n\n        # Focal loss\n        cross_entropy = -y_true_smooth * tf.math.log(y_pred)\n        focal = tf.pow(1.0 - y_pred, gamma)\n\n        # Alpha per class\n        alpha_factor = tf.reduce_sum(alpha * y_true_onehot, axis=-1, keepdims=True)\n        loss = alpha_factor * focal * cross_entropy\n\n        return tf.reduce_mean(tf.reduce_sum(loss, axis=-1))\n\n    return loss_fn\n    \n# 5. Khởi tạo loss function với tham số rõ ràng\ngamma = 2.0\nlabel_smoothing = 0.1\nloss_fn = focal_loss_with_custom_alpha(alpha_values, gamma=gamma, label_smoothing=label_smoothing)\n\nprint(f'✅ Custom focal loss with boosted alpha, gamma={gamma} and label_smoothing={label_smoothing} ready.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:22:54.445261Z","iopub.execute_input":"2025-07-03T13:22:54.445432Z","iopub.status.idle":"2025-07-03T13:22:54.467535Z","shell.execute_reply.started":"2025-07-03T13:22:54.445410Z","shell.execute_reply":"2025-07-03T13:22:54.466811Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.optimizers import AdamW\n\ndef build_model():\n    \"\"\"\n    A custom implementation of EfficientNetB3\n    for the APTOS 2019 competition\n    (Regression)\n    \"\"\"\n    effnet = EfficientNetB3(include_top=False, weights='imagenet', input_shape=(IMG_HEIGHT, IMG_WIDTH, 3))\n    \n    # Freeze hết, chỉ fine-tune 60 layer cuối\n    for layer in effnet.layers:\n        layer.trainable = False\n    for layer in effnet.layers[-60:]:\n        layer.trainable = True\n        \n    model = Sequential()\n    model.add(effnet)\n    model.add(GlobalAveragePooling2D())\n    model.add(Dense(512, activation='relu'))\n    model.add(Dense(218, activation='relu'))\n    model.add(Dropout(0.25))\n    model.add(Dense(5, activation='softmax'))\n    # model.add(Dense(1, activation=\"linear\"))\n    \n    model.compile(loss=loss_fn, #'mse',\n                  optimizer=AdamW(learning_rate=1e-4, weight_decay=1e-5),\n                  metrics=['mae'])\n    model.summary()\n    return model\n\n# Initialize model\nmodel = build_model()\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:22:54.468293Z","iopub.execute_input":"2025-07-03T13:22:54.468575Z","iopub.status.idle":"2025-07-03T13:22:56.019256Z","shell.execute_reply.started":"2025-07-03T13:22:54.468558Z","shell.execute_reply":"2025-07-03T13:22:56.018546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from keras.callbacks import EarlyStopping, ReduceLROnPlateau, ModelCheckpoint\n\n# Custom QWK metrics callback (nếu đã định nghĩa lớp Metrics như bạn nói)\nkappa_metrics = Metrics(val_generator=val_generator)\n\n# Early stopping - tránh overfitting, phục hồi best weights\nearly_stopping = EarlyStopping(\n    monitor='val_loss',\n    mode='min',\n    verbose=1,\n    patience=5,\n    restore_best_weights=True\n)\n\n# Reduce learning rate khi val_loss không giảm\nreduce_lr = ReduceLROnPlateau(\n    monitor='val_loss',\n    factor=0.3,\n    patience=3,\n    verbose=1,\n    mode='min',\n    min_delta=1e-4,\n    min_lr=1e-6\n)\n\n# Gộp lại callback list\ncallbacks = [kappa_metrics, early_stopping, reduce_lr]\n\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:22:56.020128Z","iopub.execute_input":"2025-07-03T13:22:56.020690Z","iopub.status.idle":"2025-07-03T13:22:56.026200Z","shell.execute_reply.started":"2025-07-03T13:22:56.020670Z","shell.execute_reply":"2025-07-03T13:22:56.025647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nprint(\"TensorFlow version:\", tf.__version__)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:22:56.026860Z","iopub.execute_input":"2025-07-03T13:22:56.027060Z","iopub.status.idle":"2025-07-03T13:22:56.039307Z","shell.execute_reply.started":"2025-07-03T13:22:56.027044Z","shell.execute_reply":"2025-07-03T13:22:56.038504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Begin training\nhistory = model.fit(train_generator,\n          steps_per_epoch=train_generator.samples // BATCH_SIZE,\n          epochs=50,\n          validation_data=val_generator,\n          validation_steps = val_generator.samples // BATCH_SIZE,\n          callbacks=callbacks)\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:22:56.039988Z","iopub.execute_input":"2025-07-03T13:22:56.040213Z","iopub.status.idle":"2025-07-03T13:44:07.715924Z","shell.execute_reply.started":"2025-07-03T13:22:56.040198Z","shell.execute_reply":"2025-07-03T13:44:07.715226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"FULL_MODEL_PATH = '/kaggle/working/efficientnetb3_final_model.keras'\nmodel.save(FULL_MODEL_PATH)\nprint(f\"✅ Saved full model to {FULL_MODEL_PATH}\")\n\nimport pickle\n\nwith open('/kaggle/working/efficientnetb3_training_history.pkl', 'wb') as f:\n    pickle.dump(history.history, f)\nprint(\"✅ Saved training history to efficientnetb3_training_history.pkl\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:44:07.716778Z","iopub.execute_input":"2025-07-03T13:44:07.717433Z","iopub.status.idle":"2025-07-03T13:44:08.855903Z","shell.execute_reply.started":"2025-07-03T13:44:07.717404Z","shell.execute_reply":"2025-07-03T13:44:08.855175Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load saved model\nmodel = tf.keras.models.load_model(FULL_MODEL_PATH, compile=False)\n\nwith open('/kaggle/working/efficientnetb3_training_history.pkl', 'rb') as f:\n    history = pickle.load(f)\n\nIMG_SIZE = 224\nCATEGORIES = ['No DR', 'Mild', 'Moderate', 'Severe', 'Proliferative DR']\n\ndef prepare(filepath):\n    IMG_SIZE = 224\n    img_array = cv2.imread(filepath)\n    if img_array is None:\n        raise FileNotFoundError(f\"❌ Không thể đọc ảnh: {filepath}\")\n    new_array = cv2.resize(img_array, (IMG_SIZE, IMG_SIZE))\n    return new_array.reshape(-1, IMG_SIZE, IMG_SIZE, 3)\ntest_img_path = '/kaggle/working/aug_images/001639a390f0_aug322.png'\n# Regression-based prediction\nprediction = model.predict([prepare(test_img_path)])\nrounded_pred = int(np.rint(prediction).clip(0, 4)[0][0])\nprint(f\"Predicted class: {CATEGORIES[rounded_pred]}\")\nprint(f\"Prediction (raw regression value): {prediction[0][0]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:53:41.890648Z","iopub.execute_input":"2025-07-03T13:53:41.891292Z","iopub.status.idle":"2025-07-03T13:53:54.851881Z","shell.execute_reply.started":"2025-07-03T13:53:41.891265Z","shell.execute_reply":"2025-07-03T13:53:54.851151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\n\nwith open('/kaggle/working/efficientnetb3_training_history.pkl', 'rb') as f:\n    history = pickle.load(f)\n\nprint(type(history))\nprint(history)  # Xem trực tiếp bên trong\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:54:16.866606Z","iopub.execute_input":"2025-07-03T13:54:16.866912Z","iopub.status.idle":"2025-07-03T13:54:16.872326Z","shell.execute_reply.started":"2025-07-03T13:54:16.866891Z","shell.execute_reply":"2025-07-03T13:54:16.871590Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport pandas as pd\n\nhistory_df = pd.DataFrame(history)\n\nfig, axs = plt.subplots(1, 2, figsize=(14, 5))\n\n# Loss (MSE)\naxs[0].plot(history_df['loss'], label='Train Loss ')\naxs[0].plot(history_df['val_loss'], label='Val Loss')\naxs[0].set_title('Loss ', fontsize=14, weight='bold')\naxs[0].set_xlabel('Epoch')\naxs[0].set_ylabel('Loss')\naxs[0].legend()\n\n# MAE\naxs[1].plot(history_df['mae'], label='Train MAE')\naxs[1].plot(history_df['val_mae'], label='Val MAE')\naxs[1].set_title('Mean Absolute Error (MAE)', fontsize=14, weight='bold')\naxs[1].set_xlabel('Epoch')\naxs[1].set_ylabel('MAE')\naxs[1].legend()\n\nplt.tight_layout()\nplt.savefig(os.path.join(save_img_path, 'loss_mae_plot.png'))\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:54:19.565384Z","iopub.execute_input":"2025-07-03T13:54:19.566317Z","iopub.status.idle":"2025-07-03T13:54:20.137973Z","shell.execute_reply.started":"2025-07-03T13:54:19.566289Z","shell.execute_reply":"2025-07-03T13:54:20.137108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_kappa_history = kappa_metrics.val_kappas\n\nplt.figure(figsize=(10, 5))\nplt.plot(val_kappa_history, marker='o', linestyle='-')\nplt.title('Validation Quadratic Weighted Kappa (QWK)', fontsize=16, weight='bold')\nplt.xlabel('Epoch', fontsize=12)\nplt.ylabel('QWK Score', fontsize=12)\nplt.grid(True)\nplt.tight_layout()\nplt.savefig(os.path.join(save_img_path, 'val_kappa_history.png'))\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:54:24.250212Z","iopub.execute_input":"2025-07-03T13:54:24.250705Z","iopub.status.idle":"2025-07-03T13:54:24.552461Z","shell.execute_reply.started":"2025-07-03T13:54:24.250682Z","shell.execute_reply":"2025-07-03T13:54:24.551679Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Evaluation","metadata":{}},{"cell_type":"code","source":"class OptimizedRounder(object):\n    \"\"\"\n    An optimizer for rounding thresholds\n    to maximize Quadratic Weighted Kappa score\n    \"\"\"\n    def __init__(self):\n        self.coef_ = 0\n\n    def _kappa_loss(self, coef, X, y):\n        \"\"\"\n        Get loss according to\n        using current coefficients\n        \n        :param coef: A list of coefficients that will be used for rounding\n        :param X: The raw predictions\n        :param y: The ground truth labels\n        \"\"\"\n        X_p = np.copy(X)\n        for i, pred in enumerate(X_p):\n            if pred < coef[0]:\n                X_p[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                X_p[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                X_p[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                X_p[i] = 3\n            else:\n                X_p[i] = 4\n\n        ll = cohen_kappa_score(y, X_p, weights='quadratic')\n        return -ll\n\n    def fit(self, X, y):\n        \"\"\"\n        Optimize rounding thresholds\n        \n        :param X: The raw predictions\n        :param y: The ground truth labels\n        \"\"\"\n        loss_partial = partial(self._kappa_loss, X=X, y=y)\n        initial_coef = [0.5, 1.5, 2.5, 3.5]\n        self.coef_ = sp.optimize.minimize(loss_partial, initial_coef, method='nelder-mead')\n\n    def predict(self, X, coef):\n        \"\"\"\n        Make predictions with specified thresholds\n        \n        :param X: The raw predictions\n        :param coef: A list of coefficients that will be used for rounding\n        \"\"\"\n        X_p = np.copy(X)\n        for i, pred in enumerate(X_p):\n            if pred < coef[0]:\n                X_p[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                X_p[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                X_p[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                X_p[i] = 3\n            else:\n                X_p[i] = 4\n        return X_p\n\n    def coefficients(self):\n        \"\"\"\n        Return the optimized coefficients\n        \"\"\"\n        return self.coef_['x']\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:54:28.216127Z","iopub.execute_input":"2025-07-03T13:54:28.216580Z","iopub.status.idle":"2025-07-03T13:54:28.225432Z","shell.execute_reply.started":"2025-07-03T13:54:28.216557Z","shell.execute_reply":"2025-07-03T13:54:28.224754Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get predictions on train set\ny_train_preds, train_labels = get_preds_and_labels(model, train_generator)\n\n# Convert one-hot labels to class indices if needed\nif isinstance(train_labels[0], (np.ndarray, list)):\n    train_labels = np.argmax(train_labels, axis=1)\n\n# Convert prediction to class index (rounding only if regression)\nif y_train_preds.shape[1] == 1:\n    y_train_preds = np.rint(y_train_preds).astype(np.uint8).clip(0, 4).flatten()\nelse:\n    y_train_preds = np.argmax(y_train_preds, axis=1)\n\n# Compute training QWK\ntrain_score = cohen_kappa_score(train_labels, y_train_preds, weights=\"quadratic\")\n\n# Get predictions on validation set\ny_val_preds, val_labels = get_preds_and_labels(model, val_generator)\n\n# Convert one-hot labels to class indices if needed\nif isinstance(val_labels[0], (np.ndarray, list)):\n    val_labels = np.argmax(val_labels, axis=1)\n\n# Convert prediction to class index\nif y_val_preds.shape[1] == 1:\n    y_val_preds = np.rint(y_val_preds).astype(np.uint8).clip(0, 4).flatten()\nelse:\n    y_val_preds = np.argmax(y_val_preds, axis=1)\n\n# Compute validation QWK\nval_score = cohen_kappa_score(val_labels, y_val_preds, weights=\"quadratic\")\n\n# Print results\nprint(f\"The Training Cohen Kappa Score is: {round(train_score, 5)}\")\nprint(f\"The Validation Cohen Kappa Score is: {round(val_score, 5)}\")\nprint('✅')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T14:29:14.106382Z","iopub.execute_input":"2025-07-03T14:29:14.106953Z","iopub.status.idle":"2025-07-03T14:30:07.667228Z","shell.execute_reply.started":"2025-07-03T14:29:14.106925Z","shell.execute_reply":"2025-07-03T14:30:07.666506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report\n\nprint(\"Train Confusion Matrix:\")\nprint(confusion_matrix(train_labels, y_train_preds))\nprint(\"Train Classification Report:\")\nprint(classification_report(train_labels, y_train_preds))\n\nprint(\"Val Confusion Matrix:\")\nprint(confusion_matrix(val_labels, y_val_preds))\nprint(\"Val Classification Report:\")\nprint(classification_report(val_labels, y_val_preds))\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T14:39:24.843298Z","iopub.execute_input":"2025-07-03T14:39:24.844132Z","iopub.status.idle":"2025-07-03T14:39:24.873139Z","shell.execute_reply.started":"2025-07-03T14:39:24.844105Z","shell.execute_reply":"2025-07-03T14:39:24.872558Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Optimize on validation data and evaluate again\ny_val_preds, val_labels = get_preds_and_labels(model, val_generator)\noptR = OptimizedRounder()\noptR.fit(y_val_preds, val_labels)\ncoefficients = optR.coefficients()\nopt_val_predictions = optR.predict(y_val_preds, coefficients)\nnew_val_score = cohen_kappa_score(val_labels, opt_val_predictions, weights=\"quadratic\")\n\nprint(f\"Optimized Thresholds:\\n{coefficients}\\n\")\nprint(f\"The Validation Quadratic Weighted Kappa (QWK)\\n\\\nwith optimized rounding thresholds is: {round(new_val_score, 5)}\\n\")\nprint(f\"This is an improvement of {round(new_val_score - val_score, 5)}\\n\\\nover the unoptimized rounding\")\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:55:33.412502Z","iopub.status.idle":"2025-07-03T13:55:33.412757Z","shell.execute_reply.started":"2025-07-03T13:55:33.412643Z","shell.execute_reply":"2025-07-03T13:55:33.412655Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img_array = preprocess_image(img_path)\npredictions = model.predict(img_array)\npred_raw = predictions[0][0]\nprint(f\"Raw output before thresholding: {pred_raw}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.utils.multiclass import unique_labels\n\ndef plot_confusion_matrix(y_true, y_pred, classes,\n                          normalize=False,\n                          title=None,\n                          cmap=plt.cm.Blues):\n    \"\"\"\n    This function prints and plots the confusion matrix.\n    Normalization can be applied by setting `normalize=True`.\n    \"\"\"\n    if not title:\n        if normalize:\n            title = 'Normalized confusion matrix'\n        else:\n            title = 'Confusion matrix, without normalization'\n\n    # Compute confusion matrix\n    cm = confusion_matrix(y_true, y_pred)\n    # Only use the labels that appear in the data\n    classes = classes[unique_labels(y_true, y_pred)]\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n        print(\"Normalized confusion matrix\")\n    else:\n        print('Confusion matrix, without normalization')\n\n    print(cm)\n\n    fig, ax = plt.subplots()\n    im = ax.imshow(cm, interpolation='nearest', cmap=cmap)\n    ax.figure.colorbar(im, ax=ax)\n    # We want to show all ticks...\n    ax.set(xticks=np.arange(cm.shape[1]),\n           yticks=np.arange(cm.shape[0]),\n           # ... and label them with the respective list entries\n           xticklabels=classes, yticklabels=classes,\n           title=title,\n           ylabel='True label',\n           xlabel='Predicted label')\n\n    # Rotate the tick labels and set their alignment.\n    plt.setp(ax.get_xticklabels(), rotation=45, ha=\"right\",\n             rotation_mode=\"anchor\")\n\n    # Loop over data dimensions and create text annotations.\n    fmt = '.2f' if normalize else 'd'\n    thresh = cm.max() / 2.\n    for i in range(cm.shape[0]):\n        for j in range(cm.shape[1]):\n            ax.text(j, i, format(cm[i, j], fmt),\n                    ha=\"center\", va=\"center\",\n                    color=\"white\" if cm[i, j] > thresh else \"black\")\n    fig.tight_layout()\n    return ax\n\n\nnp.set_printoptions(precision=2)\nprint('✅')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_confusion_matrix(val_labels, opt_val_predictions, classes=np.array(['0', '1', '2', '3', '4']),\n                      title='Confusion matrix, without normalization')\n\nplt.savefig(os.path.join(save_img_path, 'val_conf_matrix.png'))\nplt.show()\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T00:38:36.753278Z","iopub.execute_input":"2025-06-30T00:38:36.753589Z","iopub.status.idle":"2025-06-30T00:38:37.264683Z","shell.execute_reply.started":"2025-06-30T00:38:36.753569Z","shell.execute_reply":"2025-06-30T00:38:37.263640Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport cv2\nimport os\n\n# === Load EfficientNetB3 model đã train ===\nmodel = keras.models.load_model(FULL_MODEL_PATH)\n\n# Dummy call để model có .input và .output\n_ = model(tf.zeros((1, IMG_HEIGHT, IMG_WIDTH, 3)))\n\n# === Grad-CAM heatmap ===\ndef make_gradcam_heatmap(img_array, model, last_conv_layer_name, pred_index=None):\n    # Tách EfficientNetB3 backbone\n    backbone = model.get_layer('efficientnetb3')\n    last_conv_layer = backbone.get_layer(last_conv_layer_name)\n    feature_extractor = keras.Model(inputs=backbone.input, outputs=last_conv_layer.output)\n\n    # Build classifier\n    classifier_input = keras.Input(shape=last_conv_layer.output.shape[1:])\n    x = classifier_input\n    for layer in model.layers[1:]:\n        x = layer(x)\n    classifier_model = keras.Model(inputs=classifier_input, outputs=x)\n\n    with tf.GradientTape() as tape:\n        conv_outputs = feature_extractor(img_array)\n        tape.watch(conv_outputs)\n        preds = classifier_model(conv_outputs)\n        if pred_index is None:\n            pred_index = tf.argmax(preds[0])\n        class_channel = preds[:, pred_index]\n\n    grads = tape.gradient(class_channel, conv_outputs)\n    pooled_grads = tf.reduce_mean(grads, axis=(0, 1, 2))\n    conv_outputs = conv_outputs[0]\n    heatmap = conv_outputs @ pooled_grads[..., tf.newaxis]\n    heatmap = tf.squeeze(heatmap)\n    heatmap = tf.maximum(heatmap, 0) / tf.math.reduce_max(heatmap)\n    return heatmap.numpy()\n\n# === Preprocess ảnh ===\ndef preprocess_image(img_path):\n    img = keras.utils.load_img(img_path, target_size=(IMG_HEIGHT, IMG_WIDTH))\n    img_array = keras.utils.img_to_array(img)\n    img_array = np.expand_dims(img_array, axis=0)\n    return img_array\n\n# === Overlay Grad-CAM ===\ndef save_and_display_gradcam(img_path, heatmap, true_label, save_path, alpha=0.5):\n    img = keras.utils.load_img(img_path)\n    img = keras.utils.img_to_array(img)\n    \n    os.makedirs(save_path, exist_ok=True)\n\n    heatmap = np.uint8(255 * heatmap)\n    heatmap = cv2.applyColorMap(heatmap, cv2.COLORMAP_JET)\n    heatmap = cv2.resize(heatmap, (img.shape[1], img.shape[0]))\n    heatmap = cv2.cvtColor(heatmap, cv2.COLOR_BGR2RGB)\n\n    superimposed_img = heatmap * alpha + img\n    superimposed_img = np.clip(superimposed_img, 0, 255).astype('uint8')\n\n    predictions = model.predict(preprocess_image(img_path))\n    pred = int(np.rint(predictions).clip(0, 4)[0][0])\n\n    plt.figure(figsize=(15, 5))\n    plt.subplot(1, 3, 1)\n    plt.imshow(img.astype('uint8'))\n    plt.title('Original')\n    plt.axis('off')\n\n    plt.subplot(1, 3, 2)\n    plt.imshow(heatmap)\n    plt.title('Grad-CAM')\n    plt.axis('off')\n\n    plt.subplot(1, 3, 3)\n    plt.imshow(superimposed_img)\n    plt.title(f\"Overlayed\\nTrue: {true_label}, Pred: {pred}\")\n    plt.axis('off')\n\n    plt.tight_layout()\n    filename = os.path.basename(img_path).replace('.png', '_gradcam.png')\n    plt.savefig(os.path.join(save_path, filename), dpi=300)\n    plt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T00:32:51.200574Z","iopub.execute_input":"2025-06-30T00:32:51.200984Z","iopub.status.idle":"2025-06-30T00:33:05.324064Z","shell.execute_reply.started":"2025-06-30T00:32:51.200955Z","shell.execute_reply":"2025-06-30T00:33:05.322932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import tensorflow as tf\n# from tensorflow import keras\n# import numpy as np\n# import matplotlib.pyplot as plt\n# import cv2\n\n# # Load model\n# model = keras.models.load_model('/content/drive/MyDrive/NCKH/Book Chapter/model/vgg16_final.keras')\n\n# # Build model với dummy input\n# dummy_input = tf.zeros((1, 224, 224, 3))\n# _ = model(dummy_input)\n\n# # Hàm tạo Grad-CAM heatmap\n# def make_gradcam_heatmap(img_array, model, last_conv_layer_name, pred_index=None):\n#     # Lấy phần backbone VGG16\n#     vgg_model = model.get_layer('vgg16')\n\n#     # Tách các layer\n#     last_conv_layer = vgg_model.get_layer(last_conv_layer_name)\n#     feature_extractor = keras.Model(vgg_model.input, last_conv_layer.output)\n\n#     # Phần classifier\n#     classifier_input = keras.Input(shape=last_conv_layer.output.shape[1:])\n#     x = classifier_input\n#     for layer in model.layers[1:]:  # Bỏ vgg16, lấy các lớp sau\n#         x = layer(x)\n#     classifier_model = keras.Model(classifier_input, x)\n\n#     # Gradient Tape\n#     with tf.GradientTape() as tape:\n#         conv_outputs = feature_extractor(img_array)\n#         tape.watch(conv_outputs)\n#         preds = classifier_model(conv_outputs)\n#         if pred_index is None:\n#             pred_index = tf.argmax(preds[0])\n#         class_channel = preds[:, pred_index]\n\n#     # Tính gradient\n#     grads = tape.gradient(class_channel, conv_outputs)\n#     pooled_grads = tf.reduce_mean(grads, axis=(0, 1, 2))\n\n#     conv_outputs = conv_outputs[0]\n#     heatmap = conv_outputs @ pooled_grads[..., tf.newaxis]\n#     heatmap = tf.squeeze(heatmap)\n\n#     # Chuẩn hóa heatmap\n#     heatmap = tf.maximum(heatmap, 0) / tf.math.reduce_max(heatmap)\n#     return heatmap.numpy()\n\n# # Load ảnh để Grad-CAM\n# def preprocess_image(img_path, target_size=(224,224)):\n#     img = keras.utils.load_img(img_path, target_size=target_size)\n#     img_array = keras.utils.img_to_array(img)\n#     img_array = np.expand_dims(img_array, axis=0)\n#     # img_array = img_array / 255.0  # normalize nếu cần\n#     return img_array\n\n# # Overlay heatmap lên ảnh gốc\n# def save_and_display_gradcam(img_path, heatmap, true_label, svg_path, alpha=0.8, class_names=['basophil', 'erythroblast', 'monocyte', 'myeloblast', 'seg_neutrophil']):\n#     # Load ảnh gốc\n#     img = keras.utils.load_img(img_path)\n#     img = keras.utils.img_to_array(img)\n\n#     # Chuyển heatmap về dạng đúng\n#     heatmap = np.uint8(255 * heatmap)\n#     heatmap = cv2.applyColorMap(heatmap, cv2.COLORMAP_JET)\n\n#     # Resize heatmap về cùng kích thước ảnh gốc\n#     heatmap = cv2.resize(heatmap, (img.shape[1], img.shape[0]))\n\n#     # Chuyển đổi heatmap từ BGR sang RGB\n#     heatmap = cv2.cvtColor(heatmap, cv2.COLOR_BGR2RGB)\n\n#     # Overlay heatmap lên ảnh gốc\n#     superimposed_img = heatmap * alpha + img\n#     superimposed_img = np.clip(superimposed_img, 0, 255).astype('uint8')\n\n#     # Dự đoán nhãn của ảnh\n#     predictions = model.predict(preprocess_image(img_path))\n#     predicted_class = np.argmax(predictions)\n#     predicted_label = class_names[predicted_class] if class_names else f\"Class {predicted_class}\"\n\n#     # Hiển thị ảnh gốc, heatmap và ảnh overlay\n#     plt.figure(figsize=(15,5))\n\n#     # Ảnh gốc\n#     plt.subplot(1, 3, 1)\n#     plt.imshow(img.astype('uint8'))\n#     plt.title('Original Image')\n#     plt.axis('off')\n\n#     # Heatmap\n#     plt.subplot(1, 3, 2)\n#     plt.imshow(heatmap)\n#     plt.title('Grad-CAM Heatmap')\n#     plt.axis('off')\n\n#     # Ảnh overlay\n#     plt.subplot(1, 3, 3)\n#     plt.imshow(superimposed_img)\n#     plt.title(f'Overlayed Image\\nTrue: {true_label}, Pred: {predicted_label}')\n#     plt.axis('off')\n\n#     plt.tight_layout()\n\n#     # Save the figure as SVG\n#     plt.savefig(svg_path, format='svg')\n\n#     plt.show()\n\n# # =============== CÁCH GỌI ===============\n\n# # Đường dẫn ảnh bạn muốn test\n# img_path = '/content/drive/MyDrive/NCKH/Book Chapter/Blood Cell images for Cancer detection/basophil/BA_396998.jpg'  # sửa đúng file của bạn\n# # img_path = '/content/drive/MyDrive/NCKH/Book Chapter/Blood Cell images for Cancer detection/seg_neutrophil/NGS_0022.jpg'\n# # Tiền xử lý ảnh\n# img_array = preprocess_image(img_path)\n\n# # Tạo heatmap\n# heatmap = make_gradcam_heatmap(img_array, model, last_conv_layer_name=\"block5_conv3\")\n\n# # Overlay lên ảnh\n# save_and_display_gradcam(img_path, heatmap, true_label='basophil',svg_path='gradcam_output.svg')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-29T23:11:04.792929Z","iopub.status.idle":"2025-06-29T23:11:04.793365Z","shell.execute_reply.started":"2025-06-29T23:11:04.793149Z","shell.execute_reply":"2025-06-29T23:11:04.793187Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === Ví dụ sử dụng ===\nimg_path = '/kaggle/working/aug_images/001639a390f0.png'\nimg_array = preprocess_image(img_path)\nheatmap = make_gradcam_heatmap(img_array, model, last_conv_layer_name='top_conv')\nsave_and_display_gradcam(img_path, heatmap, true_label=2, save_path='/kaggle/working/gradcam_results')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import classification_report\n\n# Sau khi đã có dự đoán rời rạc (0–4) và nhãn thật:\nprint(\"Classification Report (Validation Set):\")\nprint(classification_report(val_labels, opt_val_predictions, digits=4, target_names=[\n    'No DR', 'Mild', 'Moderate', 'Severe', 'Proliferative DR']))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T00:39:01.034392Z","iopub.execute_input":"2025-06-30T00:39:01.034729Z","iopub.status.idle":"2025-06-30T00:39:01.052550Z","shell.execute_reply.started":"2025-06-30T00:39:01.034704Z","shell.execute_reply":"2025-06-30T00:39:01.051516Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.predict(np.expand_dims(preprocess_image(cv2.imread(val_df.iloc[0]['id_code'])), axis=0))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T00:39:05.185764Z","iopub.execute_input":"2025-06-30T00:39:05.186087Z","iopub.status.idle":"2025-06-30T00:39:05.233415Z","shell.execute_reply.started":"2025-06-30T00:39:05.186066Z","shell.execute_reply":"2025-06-30T00:39:05.232034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\n\nsample_idx = random.sample(range(len(val_df)), 5)\n\nfor i in sample_idx:\n    img_path = os.path.join(aug_img_path, val_df.iloc[i]['id_code'])  # ảnh augment\n    true_label = int(val_df.iloc[i]['diagnosis'])\n    pred_label = int(opt_val_predictions[i])  # đã tối ưu hóa threshold rồi\n\n    print(f\"Image: {img_path} | True: {true_label} | Pred: {pred_label}\")\n    generate_gradcam(img_path, model, class_idx=pred_label, save_path=save_img_path)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-29T23:11:04.804976Z","iopub.status.idle":"2025-06-29T23:11:04.805322Z","shell.execute_reply.started":"2025-06-29T23:11:04.805140Z","shell.execute_reply":"2025-06-29T23:11:04.805157Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Test","metadata":{}},{"cell_type":"code","source":"# Dự đoán trên test set\ny_test_preds, test_labels = get_preds_and_labels(model, test_generator)\ny_test_preds = np.rint(y_test_preds).astype(np.uint8).clip(0, 4)\n\n# QWK gốc\ntest_score = cohen_kappa_score(test_labels, y_test_preds, weights=\"quadratic\")\nprint(f\"The Test Cohen Kappa Score is: {round(test_score, 5)}\")\n\n# Tối ưu hóa threshold\noptR = OptimizedRounder()\noptR.fit(y_test_preds, test_labels)\ncoefficients = optR.coefficients()\nopt_test_predictions = optR.predict(y_test_preds, coefficients)\nnew_test_score = cohen_kappa_score(test_labels, opt_test_predictions, weights=\"quadratic\")\n\nprint(f\"Optimized Thresholds:\\n{coefficients}\\n\")\nprint(f\"The Test Quadratic Weighted Kappa (QWK)\\n\\\nwith optimized rounding thresholds is: {round(new_test_score, 5)}\\n\")\nprint(f\"Improvement over unoptimized rounding: {round(new_test_score - test_score, 5)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-29T23:11:04.806914Z","iopub.status.idle":"2025-06-29T23:11:04.807275Z","shell.execute_reply.started":"2025-06-29T23:11:04.807113Z","shell.execute_reply":"2025-06-29T23:11:04.807126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Đánh giá mô hình trên tập test\nloss, accuracy = model.evaluate(X_test, y_test)\nprint(f'Model Accuracy : {accuracy * 100}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-29T23:11:04.809657Z","iopub.status.idle":"2025-06-29T23:11:04.810016Z","shell.execute_reply.started":"2025-06-29T23:11:04.809875Z","shell.execute_reply":"2025-06-29T23:11:04.809888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Dự đoán trên tập test\npred = np.argmax(model.predict(X_test), axis = -1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-29T23:11:04.811400Z","iopub.status.idle":"2025-06-29T23:11:04.811733Z","shell.execute_reply.started":"2025-06-29T23:11:04.811599Z","shell.execute_reply":"2025-06-29T23:11:04.811613Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Confusion Matrix\nplot_confusion_matrix(test_labels, opt_test_predictions, classes=np.array(['0', '1', '2', '3', '4']),\n                      title='Test Confusion Matrix (Optimized)')\n\nplt.savefig(os.path.join(save_img_path, 'test_conf_matrix.png'))\nplt.show()\n\n# Classification Report\nprint(\"Classification Report (Test Set):\")\nprint(classification_report(test_labels, opt_test_predictions, digits=4, target_names=[\n    'No DR', 'Mild', 'Moderate', 'Severe', 'Proliferative DR']))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-29T23:11:04.813408Z","iopub.status.idle":"2025-06-29T23:11:04.813763Z","shell.execute_reply.started":"2025-06-29T23:11:04.813579Z","shell.execute_reply":"2025-06-29T23:11:04.813591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#confusion matrix\ncf = confusion_matrix(y_test, pred, normalize = 'true')\nsns.heatmap(cf, annot. = True, cmap = 'crest');\nplt.xlabel('PREDICTIONS');\nplt.ylabel('ACTUAL');","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-29T23:11:04.815511Z","iopub.status.idle":"2025-06-29T23:11:04.815950Z","shell.execute_reply.started":"2025-06-29T23:11:04.815729Z","shell.execute_reply":"2025-06-29T23:11:04.815743Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === Ví dụ sử dụng ===\nimg_path = '/kaggle/working/aug_images/001639a390f0.png'\nimg_array = preprocess_image(img_path)\nheatmap = make_gradcam_heatmap(img_array, model, last_conv_layer_name='top_conv')\nsave_and_display_gradcam(img_path, heatmap, true_label=2, save_path='/kaggle/working/gradcam_results')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-29T23:11:04.817275Z","iopub.status.idle":"2025-06-29T23:11:04.818115Z","shell.execute_reply.started":"2025-06-29T23:11:04.817935Z","shell.execute_reply":"2025-06-29T23:11:04.817951Z"}},"outputs":[],"execution_count":null}]}