{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":14774,"databundleVersionId":875431}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install albumentations","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T15:40:37.985377Z","iopub.execute_input":"2025-08-12T15:40:37.985696Z","iopub.status.idle":"2025-08-12T15:40:40.991864Z","shell.execute_reply.started":"2025-08-12T15:40:37.985651Z","shell.execute_reply":"2025-08-12T15:40:40.991077Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'  # 0 = all logs, 1 = info, 2 = warning, 3 = error only\nos.environ['TF_XLA_FLAGS'] = '--tf_xla_auto_jit=0'\nos.environ[\"PYTORCH_CUDA_ALLOC_CONF\"] = \"expandable_segments:True\"\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\n\nprint('✅')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-12T15:40:40.993770Z","iopub.execute_input":"2025-08-12T15:40:40.994071Z","iopub.status.idle":"2025-08-12T15:40:40.999725Z","shell.execute_reply.started":"2025-08-12T15:40:40.994036Z","shell.execute_reply":"2025-08-12T15:40:40.998831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Standard dependencies\nimport cv2\nimport time\nimport scipy as sp\nimport numpy as np\nimport random as rn\nimport pandas as pd\nfrom tqdm import tqdm\nfrom PIL import Image\nfrom functools import partial\nimport matplotlib.pyplot as plt\n\n# Machine Learning\nimport os\nimport cv2\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nfrom sklearn.utils import shuffle\nfrom albumentations import Compose, HorizontalFlip, VerticalFlip, RandomBrightnessContrast, Rotate, Resize, RandomGamma\n\nimport tensorflow as tf\nimport keras\nfrom tensorflow.keras import initializers\nfrom tensorflow.keras import regularizers\nfrom tensorflow.keras import constraints\nfrom tensorflow.keras import backend as K\nfrom tensorflow.keras.activations import elu\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Layer\nfrom tensorflow.python.keras.engine.input_spec import InputSpec\n\nfrom tensorflow.keras.utils import get_custom_objects\nfrom tensorflow.keras.callbacks import Callback, EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras.layers import Dense, Conv2D, Flatten, GlobalAveragePooling2D, Dropout\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.metrics import cohen_kappa_score\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T15:40:41.000506Z","iopub.execute_input":"2025-08-12T15:40:41.001213Z","iopub.status.idle":"2025-08-12T15:40:41.024816Z","shell.execute_reply.started":"2025-08-12T15:40:41.001191Z","shell.execute_reply":"2025-08-12T15:40:41.024109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Path specifications\nKAGGLE_DIR = '../input/aptos2019-blindness-detection/'\ndf = pd.read_csv(os.path.join(KAGGLE_DIR + \"train.csv\"))\n#test_df_path = KAGGLE_DIR + 'test.csv'\nimg_dir = KAGGLE_DIR + \"train_images/\"\ndf['image_path'] = df['id_code'].apply(lambda x: os.path.join(img_dir, f\"{x}.png\"))\n#test_img_path = KAGGLE_DIR + 'test_images/'\n#SAVED_MODEL_NAME = '/kaggle/working/efficientnetb3_best_kappa_model.keras'\n\nsave_img_path = '/kaggle/working/images'\nos.makedirs(save_img_path, exist_ok=True)\n# aug_img_path = \"/kaggle/working/aug_images/\"\n# os.makedirs(aug_img_path, exist_ok=True)\n\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T15:40:41.025598Z","iopub.execute_input":"2025-08-12T15:40:41.025807Z","iopub.status.idle":"2025-08-12T15:40:41.051451Z","shell.execute_reply.started":"2025-08-12T15:40:41.025793Z","shell.execute_reply":"2025-08-12T15:40:41.050712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in range(5):\n    df[f\"ovr_{i}\"] = df['diagnosis'].apply(lambda x: 1 if x == i else 0)\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T15:40:41.053301Z","iopub.execute_input":"2025-08-12T15:40:41.053493Z","iopub.status.idle":"2025-08-12T15:40:41.065652Z","shell.execute_reply.started":"2025-08-12T15:40:41.053479Z","shell.execute_reply":"2025-08-12T15:40:41.064966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.utils import resample\nfrom sklearn.model_selection import train_test_split\n\n# Bước 1: Chia train và val_test (train khoảng 70%)\ntrain_df, val_test_df = train_test_split(\n    df,\n    test_size=0.30,  # val + test chiếm 30%\n    stratify=df['diagnosis'],\n    shuffle=True,\n    random_state=42\n)\n\n# Bước 2: Chia val và test từ val_test_df theo tỉ lệ 1:2 (val = 10%, test = 20%)\nval_df, test_df = train_test_split(\n    val_test_df,\n    test_size=2/3,  # test = 2 phần, val = 1 phần → test ~20%, val ~10%\n    stratify=val_test_df['diagnosis'],\n    shuffle=True,\n    random_state=42\n)\n\nmax_count = train_df[\"diagnosis\"].value_counts().max()\noversampled_dfs = []\n\nfor i in range(5):\n    cls_df = train_df[train_df[\"diagnosis\"] == i]\n    if len(cls_df) < max_count:\n        cls_df = resample(cls_df, replace=True, n_samples=max_count, random_state=42)\n    oversampled_dfs.append(cls_df)\n\noversampled_train_df = pd.concat(oversampled_dfs).sample(frac=1, random_state=42).reset_index(drop=True)\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T15:40:41.066502Z","iopub.execute_input":"2025-08-12T15:40:41.066765Z","iopub.status.idle":"2025-08-12T15:40:41.098825Z","shell.execute_reply.started":"2025-08-12T15:40:41.066744Z","shell.execute_reply":"2025-08-12T15:40:41.098247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Image IDs and Labels (TRAIN)\")\n# Add extension to id_code\n#train_val_df['id_code'] = train_val_df['id_code'] + \".png\"\nprint(f\"Training images: {train_df.shape[0]}\")\ndisplay(train_df.head())\n\nprint(\"Image IDs (TEST)\")\n# Add extension to id_code\n#test_df['id_code'] = test_df['id_code'] + \".png\"\nprint(f\"Testing Images: {test_df.shape[0]}\")\ndisplay(test_df.head())\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T15:40:41.099467Z","iopub.execute_input":"2025-08-12T15:40:41.099643Z","iopub.status.idle":"2025-08-12T15:40:41.115598Z","shell.execute_reply.started":"2025-08-12T15:40:41.099629Z","shell.execute_reply":"2025-08-12T15:40:41.115069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Đếm số lượng ảnh từng lớp trước và sau\nbefore_counts = train_df['diagnosis'].value_counts().sort_index()\nafter_counts = oversampled_train_df['diagnosis'].value_counts().sort_index()\n\n# Vẽ biểu đồ cột trên cùng một hàng\nfig, axes = plt.subplots(1, 2, figsize=(12, 5), sharey=True)\n\n# Trước khi oversample\naxes[0].bar(before_counts.index.astype(str), before_counts.values, color='skyblue')\naxes[0].set_title(\"Before Oversampling\")\naxes[0].set_xlabel(\"Class\")\naxes[0].set_ylabel(\"Image Count\")\n\n# Thêm số trên đỉnh cột\nfor i, v in enumerate(before_counts.values):\n    axes[0].text(i, v + 20, str(v), ha='center', va='bottom', fontsize=10)\n\n# Sau khi oversample\naxes[1].bar(after_counts.index.astype(str), after_counts.values, color='salmon')\naxes[1].set_title(\"After Oversampling\")\naxes[1].set_xlabel(\"Class\")\n\n# Thêm số trên đỉnh cột\nfor i, v in enumerate(after_counts.values):\n    axes[1].text(i, v + 20, str(v), ha='center', va='bottom', fontsize=10)\n\nplt.tight_layout()\nplt.show()\nplt.savefig('/kaggle/working/comparison.png')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T15:40:41.116398Z","iopub.execute_input":"2025-08-12T15:40:41.116637Z","iopub.status.idle":"2025-08-12T15:40:41.400143Z","shell.execute_reply.started":"2025-08-12T15:40:41.116615Z","shell.execute_reply":"2025-08-12T15:40:41.399468Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Specify image size\nIMG_WIDTH = 224\nIMG_HEIGHT = 224\nCHANNELS = 3\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T15:40:41.400982Z","iopub.execute_input":"2025-08-12T15:40:41.401221Z","iopub.status.idle":"2025-08-12T15:40:41.405536Z","shell.execute_reply.started":"2025-08-12T15:40:41.401204Z","shell.execute_reply":"2025-08-12T15:40:41.404985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def check_missing(df, name):\n    missing = df[~df['image_path'].apply(os.path.exists)]\n    print(f\"{name}: {len(missing)} ảnh bị thiếu.\")\n    if not missing.empty:\n        print(missing[['id_code', 'image_path']].head(5))\n    return missing\n\nmissing_train = check_missing(train_df, \"Train\")\nmissing_val = check_missing(val_df, \"Validation\")\nmissing_oversample = check_missing(oversampled_train_df, \"Oversampled Train\")\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T15:40:41.406245Z","iopub.execute_input":"2025-08-12T15:40:41.406449Z","iopub.status.idle":"2025-08-12T15:40:46.448585Z","shell.execute_reply.started":"2025-08-12T15:40:41.406434Z","shell.execute_reply":"2025-08-12T15:40:46.447977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.utils import Sequence\nimport cv2\nimport numpy as np\nfrom albumentations import (\n    Compose, Resize, HorizontalFlip, CLAHE,\n    RandomBrightnessContrast, Rotate, Normalize\n)\nfrom albumentations.core.composition import OneOf\n\n# Augment mạnh cho training\ntrain_transform = Compose([\n    Resize(224, 224),\n    HorizontalFlip(p=0.5),\n    CLAHE(p=0.3),\n    RandomBrightnessContrast(p=0.5),\n    Rotate(limit=10, p=0.5),\n    Normalize()\n])\n\n# Augment nhẹ hoặc chỉ resize cho validation/test\nval_transform = Compose([\n    Resize(224, 224),\n    Normalize()\n])\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T15:40:46.449457Z","iopub.execute_input":"2025-08-12T15:40:46.449774Z","iopub.status.idle":"2025-08-12T15:40:46.459975Z","shell.execute_reply.started":"2025-08-12T15:40:46.449749Z","shell.execute_reply":"2025-08-12T15:40:46.459330Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.utils import Sequence\nimport cv2\nimport numpy as np\n\nclass DR_OvR_Generator(Sequence):\n    def __init__(self, df, batch_size=32, transform=None, shuffle=True):\n        self.df = df.reset_index(drop=True)\n        self.batch_size = batch_size\n        self.transform = transform\n        self.shuffle = shuffle\n        self.indices = np.arange(len(self.df))\n        self.on_epoch_end()\n\n    def __len__(self):\n        return int(np.ceil(len(self.df) / self.batch_size))\n\n    def __getitem__(self, index):\n        batch_indices = self.indices[index * self.batch_size:(index + 1) * self.batch_size]\n        batch_df = self.df.iloc[batch_indices]\n\n        X = np.empty((len(batch_df), 224, 224, 3), dtype=np.float32)\n        y = np.empty((len(batch_df), 5), dtype=np.float32)\n\n        for i, (_, row) in enumerate(batch_df.iterrows()):\n            path = row['image_path']\n            image = cv2.imread(path)\n\n            if image is None:\n                print(f\"❗ Không thể load ảnh: {path}\")\n                image = np.zeros((224, 224, 3), dtype=np.uint8)\n\n            elif len(image.shape) != 3 or image.shape[2] != 3:\n                print(f\"❗ Ảnh không đúng định dạng RGB: {path}, shape: {image.shape}\")\n                image = np.zeros((224, 224, 3), dtype=np.uint8)\n\n            else:\n                try:\n                    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n                except Exception as e:\n                    print(f\"❗ Lỗi chuyển màu RGB ảnh {path}: {e}\")\n                    image = np.zeros((224, 224, 3), dtype=np.uint8)\n\n            if self.transform:\n                try:\n                    image = self.transform(image=image)['image']\n                except Exception as e:\n                    print(f\"❗ Lỗi augment ảnh {path}: {e}\")\n                    image = np.zeros((224, 224, 3), dtype=np.uint8)\n\n            X[i] = image\n            y[i] = [row[f\"ovr_{j}\"] for j in range(5)]\n\n        return X, {f\"ovr_{j}\": y[:, j] for j in range(5)}\n\n    def on_epoch_end(self):\n        if self.shuffle:\n            np.random.shuffle(self.indices)\nprint('✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T15:40:46.460705Z","iopub.execute_input":"2025-08-12T15:40:46.460933Z","iopub.status.idle":"2025-08-12T15:40:46.477495Z","shell.execute_reply.started":"2025-08-12T15:40:46.460909Z","shell.execute_reply":"2025-08-12T15:40:46.476796Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras import layers, models\n\ndef build_resnet_ovr(input_shape=(224,224,3), fine_tune_at=60):\n    base = ResNet50(include_top=False, input_shape=input_shape, weights='imagenet')\n    for layer in base.layers:\n        layer.trainable = False\n    for layer in base.layers[-fine_tune_at:]:\n        layer.trainable = True\n\n    x = layers.GlobalAveragePooling2D()(base.output)\n    x = layers.Dense(512, activation='relu')(x)\n    x = layers.Dense(128, activation='relu')(x)\n    x = layers.Dropout(0.3)(x)\n    outputs = [layers.Dense(1, activation='sigmoid', name=f\"ovr_{i}\")(x) for i in range(5)]\n    return models.Model(inputs=base.input, outputs=outputs)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T15:40:46.478240Z","iopub.execute_input":"2025-08-12T15:40:46.478467Z","iopub.status.idle":"2025-08-12T15:40:46.495693Z","shell.execute_reply.started":"2025-08-12T15:40:46.478452Z","shell.execute_reply":"2025-08-12T15:40:46.494940Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras import backend as K\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras import layers, models\n\ndef binary_focal_loss(gamma=2.0, alpha=0.25):\n    def focal_loss(y_true, y_pred):\n        epsilon = K.epsilon()\n        y_pred = K.clip(y_pred, epsilon, 1. - epsilon)\n        cross_entropy = -y_true * K.log(y_pred) - (1 - y_true) * K.log(1 - y_pred)\n        weight = alpha * y_true * K.pow(1 - y_pred, gamma) + (1 - alpha) * (1 - y_true) * K.pow(y_pred, gamma)\n        return K.mean(weight * cross_entropy, axis=-1)\n    return focal_loss","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T15:40:46.498462Z","iopub.execute_input":"2025-08-12T15:40:46.498675Z","iopub.status.idle":"2025-08-12T15:40:46.510080Z","shell.execute_reply.started":"2025-08-12T15:40:46.498645Z","shell.execute_reply":"2025-08-12T15:40:46.509182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"losses = {f\"ovr_{i}\": binary_focal_loss() for i in range(5)}\nmetrics = {f\"ovr_{i}\": 'AUC' for i in range(5)}  # ✅ mỗi output cần metric riêng\n\nmodel = build_resnet_ovr()\nmodel.compile(optimizer='adam', loss=losses, metrics=metrics)\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T15:40:46.510760Z","iopub.execute_input":"2025-08-12T15:40:46.510950Z","iopub.status.idle":"2025-08-12T15:40:48.197312Z","shell.execute_reply.started":"2025-08-12T15:40:46.510936Z","shell.execute_reply":"2025-08-12T15:40:48.196581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prior terms for QWK evaluation\norig_counts = train_df['diagnosis'].value_counts().sort_index().to_numpy().astype(np.float32)\norig_prior = orig_counts / orig_counts.sum()\nos_counts = oversampled_train_df['diagnosis'].value_counts().sort_index().to_numpy().astype(np.float32)\nos_prior = os_counts / os_counts.sum()\nratio_prior = os_prior / orig_prior\n\ndef prior_correct_and_normalize(p_ovr, ratio):\n    p = p_ovr / ratio\n    p = p / p.sum(axis=1, keepdims=True)\n    return p","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T15:40:48.198093Z","iopub.execute_input":"2025-08-12T15:40:48.198299Z","iopub.status.idle":"2025-08-12T15:40:48.205063Z","shell.execute_reply.started":"2025-08-12T15:40:48.198283Z","shell.execute_reply":"2025-08-12T15:40:48.204356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# QWK callback on validation\nclass QWKCallback(tf.keras.callbacks.Callback):\n    def __init__(self, val_gen, y_true, ratio_prior):\n        super().__init__()\n        self.val_gen = val_gen\n        self.y_true = y_true\n        self.ratio_prior = ratio_prior\n        self.best_qwk = -1.0\n\n    def on_epoch_end(self, epoch, logs=None):\n        p_list = self.model.predict(self.val_gen, verbose=0)\n        p_val = np.concatenate(p_list, axis=1)\n        p_val = np.clip(p_val, 1e-6, 1-1e-6)\n        p_val = prior_correct_and_normalize(p_val, self.ratio_prior)\n        y_hat = p_val.argmax(axis=1)\n        qwk = cohen_kappa_score(self.y_true, y_hat, weights='quadratic')\n        print(f\"\\nval_qwk: {qwk:.4f}\")\n        logs = logs or {}\n        logs['val_qwk'] = qwk\n        if qwk > self.best_qwk:\n            self.best_qwk = qwk\n\nqwk_cb = QWKCallback(\n    val_gen=DR_OvR_Generator(val_df, 32, val_transform, shuffle=False),\n    y_true=val_df[\"diagnosis\"].values,\n    ratio_prior=ratio_prior\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T15:40:48.205696Z","iopub.execute_input":"2025-08-12T15:40:48.206223Z","iopub.status.idle":"2025-08-12T15:40:48.220071Z","shell.execute_reply.started":"2025-08-12T15:40:48.206205Z","shell.execute_reply":"2025-08-12T15:40:48.219350Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau, ModelCheckpoint\n\ncallbacks = [\n    EarlyStopping(\n        monitor='val_loss',\n        patience=5,\n        restore_best_weights=True,\n        verbose=1),\n    ReduceLROnPlateau(\n        monitor='val_loss',\n        factor=0.5,\n        patience=2,\n        verbose=1,\n        min_lr=1e-4),\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T15:40:48.220825Z","iopub.execute_input":"2025-08-12T15:40:48.221114Z","iopub.status.idle":"2025-08-12T15:40:48.236345Z","shell.execute_reply.started":"2025-08-12T15:40:48.221098Z","shell.execute_reply":"2025-08-12T15:40:48.235642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nmissing_files = []\nfor path in df['image_path']:\n    if not os.path.exists(path):\n        missing_files.append(path)\n\nprint(f\"Tổng số ảnh bị thiếu: {len(missing_files)}\")\nprint(\"Ví dụ các ảnh thiếu:\", missing_files[:5])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T15:40:48.237078Z","iopub.execute_input":"2025-08-12T15:40:48.237368Z","iopub.status.idle":"2025-08-12T15:40:48.887329Z","shell.execute_reply.started":"2025-08-12T15:40:48.237353Z","shell.execute_reply":"2025-08-12T15:40:48.886717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_gen = DR_OvR_Generator(oversampled_train_df, batch_size=32, transform=train_transform)\nval_gen = DR_OvR_Generator(val_df, batch_size=32, transform=val_transform, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T15:40:48.888071Z","iopub.execute_input":"2025-08-12T15:40:48.888253Z","iopub.status.idle":"2025-08-12T15:40:48.893617Z","shell.execute_reply.started":"2025-08-12T15:40:48.888239Z","shell.execute_reply":"2025-08-12T15:40:48.892925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=20,\n    callbacks=callbacks\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T15:40:48.894436Z","iopub.execute_input":"2025-08-12T15:40:48.894630Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\nimport os\n\n# Tạo thư mục output nếu chưa tồn tại\noutput_dir = '/kaggle/working/output/ResNet'\nos.makedirs(output_dir, exist_ok=True)\n\n# Lưu history vào file JSON\nwith open(output_dir + 'history_1.json', 'w') as f:\n    json.dump(history.history, f)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Lưu trọng số mô hình\nmodel.save_weights(os.path.join(output+dir,'/my_model.weights.h5'))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\nimport matplotlib.pyplot as plt\n\n# Đọc history từ file JSON\nwith open(output_dir + 'history_1.json', 'r') as f:\n    history = json.load(f)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nbest_epoch = np.argmin(history['val_loss']) + 1\nprint(f\"Best epoch selected by EarlyStopping: {best_epoch}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import cohen_kappa_score, confusion_matrix\n\ny_true = val_df[\"diagnosis\"].values\ny_pred_probs = model.predict(DR_OvR_Generator(val_df, 32, val_transform, shuffle=False))\ny_pred_probs = np.concatenate(y_pred_probs, axis=1)\ny_pred = np.argmax(y_pred_probs, axis=1)\n\nqwk = cohen_kappa_score(y_true, y_pred, weights='quadratic')\nprint(f\"QWK = {qwk:.4f}\")\nprint(confusion_matrix(y_true, y_pred))\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nimport cv2\n\ndef make_gradcam_heatmap(img_array, model, class_index, conv_layer_name=\"conv5_block3_out\"):\n    \"\"\"\n    Args:\n        img_array: shape (1, H, W, 3)\n        model: trained model with multiple outputs (OvR)\n        class_index: int in [0, 4] → which OvR output to use\n        conv_layer_name: usually \"top_conv\" for EfficientNet\n    Returns:\n        heatmap: np.array (H, W)\n    \"\"\"\n    grad_model = tf.keras.models.Model(\n        [model.inputs],\n        [model.get_layer(conv_layer_name).output, model.outputs[class_index]]\n    )\n\n    with tf.GradientTape() as tape:\n        conv_outputs, predictions = grad_model(img_array)\n        loss = predictions\n\n    grads = tape.gradient(loss, conv_outputs)\n    pooled_grads = tf.reduce_mean(grads, axis=(0, 1, 2))\n    conv_outputs = conv_outputs[0]\n\n    # Weighted sum of channels\n    heatmap = tf.reduce_sum(tf.multiply(pooled_grads, conv_outputs), axis=-1)\n    heatmap = tf.maximum(heatmap, 0) / tf.reduce_max(heatmap)\n    return heatmap.numpy()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_gradcam(img_path, model, class_index, conv_layer_name=\"conv5_block3_out\", alpha=0.4):\n    \"\"\"\n    Args:\n        img_path: path to original .png image\n        class_index: OvR index (0-4)\n    \"\"\"\n    # Load & preprocess ảnh\n    img = cv2.imread(img_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img_resized = cv2.resize(img, (224, 224))\n    input_array = tf.keras.applications.efficientnet.preprocess_input(img_resized.astype(np.float32))\n    input_array = np.expand_dims(input_array, axis=0)\n\n    # GradCAM\n    heatmap = make_gradcam_heatmap(input_array, model, class_index, conv_layer_name)\n\n    # Resize heatmap về đúng size ảnh gốc\n    heatmap = cv2.resize(heatmap, (img.shape[1], img.shape[0]))\n    heatmap = np.uint8(255 * heatmap)\n\n    heatmap_color = cv2.applyColorMap(heatmap, cv2.COLORMAP_JET)\n    overlay = cv2.addWeighted(img, 1 - alpha, heatmap_color, alpha, 0)\n\n    # Plot\n    plt.figure(figsize=(10, 4))\n    plt.subplot(1, 2, 1)\n    plt.imshow(img)\n    plt.title(\"Original\")\n    plt.axis(False)\n\n    plt.subplot(1, 2, 2)\n    plt.imshow(overlay)\n    plt.title(f\"GradCAM - OvR_{class_index}\")\n    plt.axis(False)\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nimport cv2\nfrom tensorflow.keras.applications.resnet50 import preprocess_input\n\nimg_path = test_df.iloc[0]['image_path']\n\n# Load and preprocess the image\nimg = cv2.imread(img_path)\nimg = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\nimg_resized = cv2.resize(img, (224, 224))  # ResNet50 dùng 224x224\nimg_array = preprocess_input(img_resized.astype(np.float32))\n\n# Dự đoán\ny_pred = model.predict(np.expand_dims(img_array, axis=0))  # trả về list 5 nhánh sigmoid\n\n# Plot Grad-CAM (giữ nguyên class_index nếu muốn highlight nhánh số 2)\nplot_gradcam(img_path, model, class_index=2)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Lấy xác suất trên val\np_list = model.predict(DR_OvR_Generator(val_df, 32, val_transform, shuffle=False),\n                       workers=4, use_multiprocessing=True, verbose=0)\np_val = np.concatenate(p_list, axis=1)  # shape (N,5)\np_val = np.clip(p_val, 1e-6, 1-1e-6)\ny_true = val_df[\"diagnosis\"].values\n\n# Các hàm suy luận và chỉ số\ndef pred_argmax_raw(p): return p.argmax(axis=1)\ndef pred_argmax_prior(p, ratio): return prior_correct_and_normalize(p, ratio).argmax(axis=1)\n\ndef pred_expected_grade_round(p, ratio=None):\n    if ratio is not None:\n        p = prior_correct_and_normalize(p, ratio)\n    ks = np.arange(p.shape[1], dtype=np.float32)\n    score = (p * ks).sum(axis=1)\n    return np.rint(score).astype(int).clip(0, p.shape[1]-1)\n\ndef off_by_one_acc(y, yhat): return np.mean(np.abs(y - yhat) <= 1)\ndef label_mae(y, yhat): return np.mean(np.abs(y - yhat))\n\ndef weighted_error_qwk_style(y, yhat, K=5):\n    from sklearn.metrics import confusion_matrix\n    W = np.zeros((K, K), dtype=np.float32)\n    for i in range(K):\n        for j in range(K):\n            W[i, j] = ((i - j) ** 2) / ((K - 1) ** 2)\n    cm = confusion_matrix(y, yhat, labels=np.arange(K)).astype(np.float32)\n    return (W * cm).sum() / cm.sum()\n\nmethods = {\n    \"argmax_raw\":            lambda p: pred_argmax_raw(p),\n    \"argmax_prior\":          lambda p: pred_argmax_prior(p, ratio_prior),\n    \"exp_grade_round\":       lambda p: pred_expected_grade_round(p, ratio=None),\n    \"exp_grade_round_prior\": lambda p: pred_expected_grade_round(p, ratio=ratio_prior),\n}\n\nrows = []\nfor name, fn in methods.items():\n    y_hat = fn(p_val)\n    rows.append({\n        \"Method\": name,\n        \"QWK\": cohen_kappa_score(y_true, y_hat, weights='quadratic'),\n        \"Accuracy\": (y_true == y_hat).mean(),\n        \"OffByOne\": off_by_one_acc(y_true, y_hat),\n        \"MAE\": label_mae(y_true, y_hat),\n        \"WeightedError\": weighted_error_qwk_style(y_true, y_hat, K=5)\n    })\n\ndf_summary = pd.DataFrame(rows).sort_values(\"QWK\", ascending=False)\nprint(df_summary)\n\n# Lưu bảng\nos.makedirs(output_dir, exist_ok=True)\nsummary_path = os.path.join(output_dir, \"summary_metrics_qwk.csv\")\ndf_summary.to_csv(summary_path, index=False)\nprint(f\"Saved: {summary_path}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}