{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"sourceType":"competition"},{"sourceId":3438806,"sourceType":"datasetVersion","datasetId":2071681},{"sourceId":281462858,"sourceType":"kernelVersion"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# --- CELL 1: IMPORTS & CONFIG ---\nimport pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models, Model\nfrom tensorflow.keras.callbacks import ModelCheckpoint, ReduceLROnPlateau, EarlyStopping\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import class_weight\nimport matplotlib.pyplot as plt\nimport os\nimport cv2\nimport albumentations as A\nimport json\n\n# CẤU HÌNH (GIỮ NGUYÊN NHƯ ĐÃ NỘP)\nBASE_PATH = '/kaggle/input/cassava-leaf-disease-classification'\nTRAIN_IMG_PATH = os.path.join(BASE_PATH, 'train_images')\nTEST_IMG_PATH = os.path.join(BASE_PATH, 'test_images')\n\nIMG_SIZE = 320   # Kích thước ảnh\nBATCH_SIZE = 16\nN_CLASSES = 5\nEPOCHS = 25      \n\nprint(f\"✅ TensorFlow Version: {tf.__version__}\")\nprint(f\"✅ Thiết lập cấu hình: IMG_SIZE={IMG_SIZE}, BATCH_SIZE={BATCH_SIZE}, EPOCHS={EPOCHS}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T15:29:46.845377Z","iopub.execute_input":"2025-12-01T15:29:46.845584Z","iopub.status.idle":"2025-12-01T15:30:11.87036Z","shell.execute_reply.started":"2025-12-01T15:29:46.845566Z","shell.execute_reply":"2025-12-01T15:30:11.86966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CELL 2: DATA PREPARATION ---\n\n# 1. Tải dữ liệu\nprint(\">>> Đang tải dữ liệu và chia tập Train/Val...\")\ndf_train = pd.read_csv(os.path.join(BASE_PATH, 'train.csv'))\n\n# 2. Tính Class Weights (Để xử lý mất cân bằng)\nclass_weights_arr = class_weight.compute_class_weight(\n    class_weight='balanced',\n    classes=np.unique(df_train['label']),\n    y=df_train['label']\n)\nclass_weights = dict(enumerate(class_weights_arr))\nprint(\"\\n⚖️ Class Weights (Trọng số lớp):\")\nfor k, v in class_weights.items():\n    print(f\"   Lớp {k}: {v:.4f}\")\n\n# 3. Chia dữ liệu (Stratified Split)\ntrain_df, val_df = train_test_split(df_train, test_size=0.2, random_state=42, stratify=df_train['label'])\nprint(f\"\\n📊 Số lượng ảnh Train: {len(train_df)}\")\nprint(f\"📊 Số lượng ảnh Val:   {len(val_df)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T15:31:25.077981Z","iopub.execute_input":"2025-12-01T15:31:25.078307Z","iopub.status.idle":"2025-12-01T15:31:25.131129Z","shell.execute_reply.started":"2025-12-01T15:31:25.078284Z","shell.execute_reply":"2025-12-01T15:31:25.130515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CELL 3: DATA GENERATOR & AUGMENTATION (ĐÃ SỬA LỖI) ---\n\n# Pipeline Train\nAUGMENTATIONS = A.Compose([\n    A.Resize(height=IMG_SIZE, width=IMG_SIZE, p=1.0), \n    A.HorizontalFlip(p=0.5),\n    A.VerticalFlip(p=0.5),\n    A.RandomRotate90(p=0.5),\n    A.ShiftScaleRotate(p=0.5),\n    A.HueSaturationValue(p=0.5),\n    A.CoarseDropout(p=0.3),\n    A.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225], p=1.0),\n], p=1.0)\n\n# Pipeline Val/Test\nVAL_AUGMENTATIONS = A.Compose([\n    A.Resize(height=IMG_SIZE, width=IMG_SIZE, p=1.0),\n    A.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225], p=1.0),\n], p=1.0)\n\nclass CassavaDataGenerator(tf.keras.utils.Sequence):\n    def __init__(self, df, base_path, batch_size, img_size, n_classes, augmentations, shuffle=True):\n        self.df = df.reset_index(drop=True)\n        self.base_path = base_path\n        self.batch_size = batch_size\n        self.img_size = img_size\n        self.n_classes = n_classes\n        self.augmentations = augmentations\n        self.shuffle = shuffle\n        self.indices = self.df.index.tolist()\n        self.on_epoch_end()\n\n    def __len__(self):\n        return int(np.ceil(len(self.indices) / self.batch_size))\n\n    def on_epoch_end(self):\n        if self.shuffle:\n            np.random.shuffle(self.indices)\n\n    def __getitem__(self, index):\n        batch_indices = self.indices[index*self.batch_size:(index+1)*self.batch_size]\n        batch_df = self.df.iloc[batch_indices]\n        X = np.zeros((len(batch_df), self.img_size, self.img_size, 3), dtype=np.float32)\n        Y = np.zeros((len(batch_df), self.n_classes), dtype=np.float32)\n        for i, (idx, row) in enumerate(batch_df.iterrows()):\n            img_path = os.path.join(self.base_path, row['image_id'])\n            img = cv2.imread(img_path)\n            if img is None: continue\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n            if self.augmentations:\n                augmented = self.augmentations(image=img)\n                img = augmented['image']\n            X[i] = img\n            Y[i] = tf.keras.utils.to_categorical(row['label'], num_classes=self.n_classes)\n        return X, Y\n\n# Khởi tạo Generators\nprint(\">>> Đang khởi tạo Data Generators...\")\ntrain_gen = CassavaDataGenerator(train_df, TRAIN_IMG_PATH, BATCH_SIZE, IMG_SIZE, N_CLASSES, AUGMENTATIONS, shuffle=True)\nval_gen = CassavaDataGenerator(val_df, TRAIN_IMG_PATH, BATCH_SIZE, IMG_SIZE, N_CLASSES, VAL_AUGMENTATIONS, shuffle=False)\nprint(\"✅ Generator đã sẵn sàng.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T15:33:43.319943Z","iopub.execute_input":"2025-12-01T15:33:43.320341Z","iopub.status.idle":"2025-12-01T15:33:43.342579Z","shell.execute_reply.started":"2025-12-01T15:33:43.320307Z","shell.execute_reply":"2025-12-01T15:33:43.341726Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CELL 4: MODEL ARCHITECTURE ---\n\ndef build_model():\n    # Tìm weights offline (Ưu tiên Noisy Student)\n    WEIGHTS_PATH = '/kaggle/input/efficientnet-keras-weights/noisy-student/b3_notop.h5'\n    \n    if not os.path.exists(WEIGHTS_PATH):\n        WEIGHTS_PATH = '/kaggle/input/efficientnet-keras-weights/imagenet_1000/b3_notop.h5'\n    \n    if os.path.exists(WEIGHTS_PATH):\n        print(f\"✅ Đã tìm thấy Weights: {WEIGHTS_PATH}\")\n        weights_mode = None \n    else:\n        print(\"❌ CẢNH BÁO: Không tìm thấy Weights offline! Hãy kiểm tra lại Input.\")\n        weights_mode = None\n\n    # Khởi tạo Base Model\n    base_model = tf.keras.applications.EfficientNetB3(\n        weights=weights_mode, \n        include_top=False, \n        input_shape=(IMG_SIZE, IMG_SIZE, 3)\n    )\n    \n    # Load weights thủ công\n    if os.path.exists(WEIGHTS_PATH):\n        try:\n            base_model.load_weights(WEIGHTS_PATH, by_name=True, skip_mismatch=True)\n            print(\"✅ Đã load weights vào model thành công.\")\n        except Exception as e:\n            print(f\"⚠️ Lỗi load weights: {e}\")\n\n    # Fine-tuning: Mở khóa toàn bộ\n    base_model.trainable = True \n    \n    # Custom Head\n    x = base_model.output\n    x = layers.GlobalAveragePooling2D()(x)\n    x = layers.Dropout(0.5)(x)\n    predictions = layers.Dense(N_CLASSES, activation='softmax')(x)\n    \n    model = Model(inputs=base_model.input, outputs=predictions)\n    \n    # Compile với Learning Rate nhỏ\n    optimizer = tf.keras.optimizers.Adam(learning_rate=1e-4)\n    \n    model.compile(\n        optimizer=optimizer,\n        loss='categorical_crossentropy',\n        metrics=['accuracy']\n    )\n    return model\n\nprint(\">>> Đang xây dựng Model...\")\nmodel = build_model()\nprint(\"✅ Model đã được biên dịch.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T15:34:13.571312Z","iopub.execute_input":"2025-12-01T15:34:13.57162Z","iopub.status.idle":"2025-12-01T15:34:17.428956Z","shell.execute_reply.started":"2025-12-01T15:34:13.571596Z","shell.execute_reply":"2025-12-01T15:34:17.42833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CELL 5: TRAINING PROCESS ---\n\n# Định nghĩa Callbacks\ncheckpoint = ModelCheckpoint(\n    'cassava_best.keras', \n    monitor='val_accuracy', \n    save_best_only=True, \n    mode='max', \n    verbose=1\n)\nreduce_lr = ReduceLROnPlateau(\n    monitor='val_loss', \n    factor=0.2, \n    patience=2, \n    min_lr=1e-6, \n    verbose=1\n)\nearly_stop = EarlyStopping(\n    monitor='val_loss', \n    patience=5, \n    restore_best_weights=True, \n    verbose=1\n)\n\nprint(f\">>> Bắt đầu huấn luyện {EPOCHS} epochs trên GPU...\")\nhistory = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=EPOCHS,\n    callbacks=[checkpoint, reduce_lr, early_stop],\n    class_weight=class_weights, # Áp dụng trọng số lớp\n    verbose=1\n)\n\n# Lưu lại lịch sử huấn luyện ra file CSV (để lỡ có tắt máy thì vẫn còn số liệu vẽ)\nhistory_df = pd.DataFrame(history.history)\nhistory_df.to_csv('history.csv', index=False)\nprint(\"✅ Huấn luyện hoàn tất. Đã lưu history.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T15:34:45.521811Z","iopub.execute_input":"2025-12-01T15:34:45.522531Z","iopub.status.idle":"2025-12-01T16:30:34.368119Z","shell.execute_reply.started":"2025-12-01T15:34:45.522507Z","shell.execute_reply":"2025-12-01T16:30:34.367501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- VẼ HÌNH MINH HỌA AUGMENTATION  ---\nimport matplotlib.pyplot as plt\nimport cv2\nimport os\nimport albumentations as A\nimport numpy as np\n\n# 1. Cấu hình các phép biến đổi\nAUGMENTATIONS = A.Compose([\n    A.Resize(height=320, width=320, p=1.0),\n    A.HorizontalFlip(p=0.5),\n    A.VerticalFlip(p=0.5),\n    A.RandomRotate90(p=0.5),\n    A.ShiftScaleRotate(p=0.5),\n    A.HueSaturationValue(p=0.5),\n    A.CoarseDropout(max_holes=8, max_height=32, max_width=32, p=0.5),\n])\n\n# 2. Chọn ảnh mẫu\nBASE_PATH = '/kaggle/input/cassava-leaf-disease-classification'\nimg_id = '2216849948.jpg' \nimg_path = os.path.join(BASE_PATH, 'test_images', img_id)\n\n# Fallback nếu không tìm thấy ảnh test\nif not os.path.exists(img_path):\n    img_path = os.path.join(BASE_PATH, 'train_images', '1000015157.jpg') \n\n# 3. Đọc ảnh\nimage = cv2.imread(img_path)\nimage = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n# 4. Vẽ lưới 3x3 (Sử dụng plt.subplots để dễ chỉnh khoảng cách)\nfig, axes = plt.subplots(3, 3, figsize=(12, 14)) # Tăng chiều cao lên 14\nfig.suptitle(\"Figure B1: Data Augmentation Examples\\n(Minh họa Tăng cường dữ liệu)\", fontsize=16, y=0.98)\n\n# Dãn khoảng cách giữa các hình ra\nplt.subplots_adjust(hspace=0.3, wspace=0.1) \n\naxes = axes.flatten()\n\nfor i in range(9):\n    ax = axes[i]\n    if i == 0:\n        # Ảnh gốc\n        aug_img = cv2.resize(image, (320, 320))\n        title = \"Original (Resized)\"\n    else:\n        # Ảnh biến đổi\n        aug_img = AUGMENTATIONS(image=image)['image']\n        title = f\"Augmented {i}\"\n    \n    ax.imshow(aug_img)\n    ax.set_title(title, fontsize=12)\n    ax.axis(\"off\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T16:46:09.450434Z","iopub.execute_input":"2025-12-01T16:46:09.451054Z","iopub.status.idle":"2025-12-01T16:46:10.354405Z","shell.execute_reply.started":"2025-12-01T16:46:09.451028Z","shell.execute_reply":"2025-12-01T16:46:10.35317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\nimport os\nimport cv2\nimport albumentations as A\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, confusion_matrix, roc_curve, auc\nfrom sklearn.preprocessing import label_binarize\nfrom itertools import cycle\n\n# --- CẤU HÌNH (GIỮ NGUYÊN) ---\nBASE_PATH = '/kaggle/input/cassava-leaf-disease-classification'\nIMG_SIZE = 320\nBATCH_SIZE = 16\nN_CLASSES = 5\nLABELS = [\"CBB\", \"CBSD\", \"CGM\", \"CMD\", \"Healthy\"]\n\n# --- TỰ ĐỘNG TÌM FILE CẦN THIẾT ---\nprint(\"🔍 Đang tìm file kết quả...\")\nhistory_path = 'history.csv'\nmodel_path = 'cassava_best.keras'\n\nif not os.path.exists(history_path):\n    for root, dirs, files in os.walk('/kaggle/input'):\n        for file in files:\n            if file == 'history.csv': history_path = os.path.join(root, file)\n            if file.endswith('.keras') and 'best' in file: model_path = os.path.join(root, file)\n\nprint(f\"📄 File History: {history_path}\")\nprint(f\"🧠 File Model: {model_path}\")\n\n# ====================================================\n# 1. VẼ FIGURE 3: BIỂU ĐỒ HUẤN LUYỆN\n# ====================================================\nif os.path.exists(history_path):\n    print(\"\\n📈 Đang vẽ Figure 3 (Training History)...\")\n    history_df = pd.read_csv(history_path)\n    epochs = range(1, len(history_df) + 1)\n    \n    plt.figure(figsize=(15, 6))\n    plt.subplot(1, 2, 1)\n    plt.plot(epochs, history_df['accuracy'], 'b-', label='Train Acc')\n    plt.plot(epochs, history_df['val_accuracy'], 'r-', label='Val Acc')\n    plt.title('Figure 3a: Accuracy')\n    plt.legend(); plt.grid(True, alpha=0.3)\n    \n    plt.subplot(1, 2, 2)\n    plt.plot(epochs, history_df['loss'], 'b-', label='Train Loss')\n    plt.plot(epochs, history_df['val_loss'], 'r-', label='Val Loss')\n    plt.title('Figure 3b: Loss')\n    plt.legend(); plt.grid(True, alpha=0.3)\n    plt.show()\nelse:\n    print(\"❌ Không tìm thấy history.csv.\")\n\n# ====================================================\n# CHUẨN BỊ DỮ LIỆU TEST\n# ====================================================\nif os.path.exists(model_path):\n    print(\"\\n⏳ Đang load model & dữ liệu...\")\n    try:\n        model = tf.keras.models.load_model(model_path)\n        df = pd.read_csv(os.path.join(BASE_PATH, 'train.csv'))\n        _, val_df = train_test_split(df, test_size=0.2, random_state=42, stratify=df['label'])\n        val_df_sample = val_df.iloc[:1000] # Lấy 1000 ảnh test\n        \n        VAL_AUG = A.Compose([\n            A.Resize(height=IMG_SIZE, width=IMG_SIZE, p=1.0),\n            A.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225], p=1.0),\n        ], p=1.0)\n\n        # Lưu ảnh gốc (để vẽ lỗi sai sau này)\n        original_images = []\n        y_true = []\n        y_pred_probs = [] # Lưu xác suất để vẽ ROC\n        \n        print(f\"⏳ Đang chạy dự đoán trên {len(val_df_sample)} ảnh...\")\n        for i, row in val_df_sample.iterrows():\n            path = os.path.join(BASE_PATH, 'train_images', row['image_id'])\n            img_raw = cv2.imread(path)\n            if img_raw is None: continue\n            img_rgb = cv2.cvtColor(img_raw, cv2.COLOR_BGR2RGB)\n            \n            # Preprocess cho model\n            img_input = VAL_AUG(image=img_rgb)['image']\n            img_input = np.expand_dims(img_input, axis=0)\n            \n            pred = model.predict(img_input, verbose=0)\n            \n            original_images.append(img_rgb) # Lưu ảnh gốc\n            y_true.append(row['label'])\n            y_pred_probs.append(pred[0])\n\n        y_true = np.array(y_true)\n        y_pred_probs = np.array(y_pred_probs)\n        y_pred = np.argmax(y_pred_probs, axis=1)\n\n        # ====================================================\n        # 2. VẼ FIGURE 4: CONFUSION MATRIX (NORMALIZED)\n        # ====================================================\n        print(\"\\n📉 Đang vẽ Figure 4 & 5 (Confusion Matrix)...\")\n        cm = confusion_matrix(y_true, y_pred)\n        # Tính phần trăm\n        cm_norm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n        \n        plt.figure(figsize=(16, 6))\n        \n        # Matrix đếm số lượng\n        plt.subplot(1, 2, 1)\n        sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=LABELS, yticklabels=LABELS)\n        plt.title(\"Figure 4: Confusion Matrix (Counts)\")\n        plt.ylabel('True'); plt.xlabel('Predicted')\n        \n        # Matrix phần trăm (MỚI)\n        plt.subplot(1, 2, 2)\n        sns.heatmap(cm_norm, annot=True, fmt='.2f', cmap='Greens', xticklabels=LABELS, yticklabels=LABELS)\n        plt.title(\"Figure 5: Normalized Confusion Matrix (%)\")\n        plt.ylabel('True'); plt.xlabel('Predicted')\n        \n        plt.tight_layout()\n        plt.show()\n\n        # ====================================================\n        # 3. VẼ FIGURE 6: ROC CURVES (MỚI)\n        # ====================================================\n        print(\"\\n📈 Đang vẽ Figure 6 (ROC Curves)...\")\n        y_true_bin = label_binarize(y_true, classes=[0, 1, 2, 3, 4])\n        n_classes = y_true_bin.shape[1]\n        \n        fpr = dict()\n        tpr = dict()\n        roc_auc = dict()\n        \n        for i in range(n_classes):\n            fpr[i], tpr[i], _ = roc_curve(y_true_bin[:, i], y_pred_probs[:, i])\n            roc_auc[i] = auc(fpr[i], tpr[i])\n            \n        plt.figure(figsize=(10, 8))\n        colors = cycle(['blue', 'red', 'green', 'purple', 'orange'])\n        for i, color in zip(range(n_classes), colors):\n            plt.plot(fpr[i], tpr[i], color=color, lw=2,\n                     label='ROC curve of class {0} (area = {1:0.2f})'.format(LABELS[i], roc_auc[i]))\n\n        plt.plot([0, 1], [0, 1], 'k--', lw=2)\n        plt.xlim([0.0, 1.0])\n        plt.ylim([0.0, 1.05])\n        plt.xlabel('False Positive Rate')\n        plt.ylabel('True Positive Rate')\n        plt.title('Figure 6: Multi-class ROC Curves')\n        plt.legend(loc=\"lower right\")\n        plt.show()\n\n        # ====================================================\n        # 4. VẼ FIGURE 7: TOP CONFIDENT ERRORS (MỚI)\n        # ====================================================\n        print(\"\\n⚠️ Đang tìm các lỗi sai nghiêm trọng nhất (Top Confident Errors)...\")\n        \n        # Tìm các index dự đoán sai\n        incorrect_indices = np.where(y_pred != y_true)[0]\n        \n        # Lấy độ tin cậy của các lần đoán sai này\n        incorrect_confidences = [np.max(y_pred_probs[i]) for i in incorrect_indices]\n        \n        # Sắp xếp để lấy những ca sai mà tự tin nhất (Model \"ảo tưởng\" nhất)\n        sorted_indices = np.argsort(incorrect_confidences)[::-1] # Giảm dần\n        top_errors = [incorrect_indices[i] for i in sorted_indices[:9]] # Lấy top 9\n        \n        plt.figure(figsize=(12, 12))\n        for i, idx in enumerate(top_errors):\n            ax = plt.subplot(3, 3, i + 1)\n            # Resize ảnh gốc về nhỏ để vẽ cho nhanh\n            img_show = cv2.resize(original_images[idx], (320, 320))\n            plt.imshow(img_show)\n            \n            true_name = LABELS[y_true[idx]]\n            pred_name = LABELS[y_pred[idx]]\n            conf = np.max(y_pred_probs[idx]) * 100\n            \n            plt.title(f\"True: {true_name}\\nPred: {pred_name} ({conf:.1f}%)\", color='red', fontsize=10)\n            plt.axis(\"off\")\n            \n        plt.suptitle(\"Figure 7: Top 9 Confident Errors (Những ca sai 'tự tin' nhất)\\n\\n\", fontsize=16, y=0.96)\n        plt.tight_layout()\n        plt.show()\n        \n        # In bảng report cuối cùng\n        print(\"\\n📋 CLASSIFICATION REPORT:\")\n        print(classification_report(y_true, y_pred, target_names=LABELS))\n\n    except Exception as e:\n        print(f\"❌ Lỗi: {e}\")\nelse:\n    print(\"❌ Không tìm thấy file model.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T17:31:52.085042Z","iopub.execute_input":"2025-12-01T17:31:52.085342Z","iopub.status.idle":"2025-12-01T17:33:28.097867Z","shell.execute_reply.started":"2025-12-01T17:31:52.085319Z","shell.execute_reply":"2025-12-01T17:33:28.097035Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CELL 9: SUBMISSION ---\nprint(\">>> Đang tạo file submit...\")\n\n# Load lại model tốt nhất\nmodel.load_weights('cassava_best.keras')\n\nsample_sub = pd.read_csv(os.path.join(BASE_PATH, 'sample_submission.csv'))\npreds = []\n\nfor image_id in sample_sub['image_id']:\n    img_path = os.path.join(TEST_IMG_PATH, image_id)\n    img = cv2.imread(img_path)\n    if img is None: \n        preds.append(3)\n        continue\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    augmented = VAL_AUGMENTATIONS(image=img)\n    img = np.expand_dims(augmented['image'], axis=0)\n    \n    prediction = model.predict(img, verbose=0)\n    preds.append(np.argmax(prediction))\n\nsample_sub['label'] = preds\nsample_sub.to_csv('submission.csv', index=False)\nprint(\"\\n✅ XONG! FILE 'submission.csv' ĐÃ SẴN SÀNG.\")\nprint(sample_sub.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T16:52:57.725333Z","iopub.execute_input":"2025-12-01T16:52:57.725662Z","iopub.status.idle":"2025-12-01T16:53:07.787541Z","shell.execute_reply.started":"2025-12-01T16:52:57.725638Z","shell.execute_reply":"2025-12-01T16:53:07.786847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nimport cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport os\nimport random\n\n# Thử import albumentations, xử lý trường hợp chưa có (dù Kaggle offline thường có sẵn)\ntry:\n    import albumentations as A\nexcept ImportError:\n    print(\"❌ Thư viện albumentations chưa được cài đặt! \")\n\n# --- 1. CẤU HÌNH (ĐỘC LẬP) ---\n# Các thông số này được fix cứng để khớp với model đã train\nIMG_SIZE = 320 \nLABELS = {\n    0: \"Bệnh Cháy Lá (CBB)\",\n    1: \"Bệnh Sọc Nâu (CBSD)\",\n    2: \"Bệnh Đốm Xanh (CGM)\",\n    3: \"Bệnh Khảm Lá (CMD)\",\n    4: \"Khỏe Mạnh (Healthy)\"\n}\n\n\nPREPROCESS = A.Compose([\n    A.Resize(height=IMG_SIZE, width=IMG_SIZE, p=1.0),\n    A.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225], p=1.0),\n], p=1.0)\n\n# --- 2. TỰ ĐỘNG TÌM MODEL ---\n# Đoạn này cực kỳ quan trọng để \"sáng mai chạy vẫn được\"\n# Nó sẽ tự đi lục lọi trong thư mục Input để tìm file .keras của bạn\nprint(\"🔍 Đang quét tìm file model (.keras)...\")\nmodel_path = None\n\n# 1. Tìm trong thư mục Input (nơi chứa model đã lưu từ phiên trước - Add Input)\nfor root, dirs, files in os.walk('/kaggle/input'):\n    for file in files:\n        if file.endswith('.keras') and ('best' in file or 'model' in file):\n            model_path = os.path.join(root, file)\n            print(f\"✅ Đã tìm thấy model trong Input: {model_path}\")\n            break\n    if model_path: break\n\n# 2. Tìm trong thư mục hiện tại (nếu bạn vừa train xong chưa tắt máy)\nif not model_path and os.path.exists('cassava_best.keras'):\n    model_path = 'cassava_best.keras'\n    print(f\"✅ Đã tìm thấy model trong thư mục làm việc: {model_path}\")\n\n# --- 3. LOAD MODEL & HÀM DỰ ĐOÁN ---\nif model_path:\n    print(\"⏳ Đang nạp model vào bộ nhớ ...\")\n    try:\n        # Load model mà không cần custom_objects vì dùng hàm chuẩn\n        model = tf.keras.models.load_model(model_path)\n        print(\"🎉 Model đã sẵn sàng để dự đoán!\")\n    except Exception as e:\n        print(f\"❌ Lỗi khi load model: {e}\")\n        model = None\nelse:\n    print(\"❌ Không tìm thấy file model nào\")\n    model = None\n\ndef predict_and_show(image_path):\n    if model is None: \n        print(\"🚫 Chưa có model, không thể dự đoán.\")\n        return\n\n    if not os.path.exists(image_path):\n        print(f\"❌ Không tìm thấy file ảnh tại: {image_path}\")\n        return\n\n    try:\n        # Đọc ảnh gốc\n        original_img = cv2.imread(image_path)\n        if original_img is None:\n            print(\"❌ File ảnh bị lỗi hoặc không phải ảnh hợp lệ.\")\n            return\n        original_img = cv2.cvtColor(original_img, cv2.COLOR_BGR2RGB)\n\n        # Xử lý ảnh (Resize + Normalize)\n        augmented = PREPROCESS(image=original_img)\n        processed_img = augmented['image']\n        \n        # Thêm chiều batch (1, 320, 320, 3)\n        input_tensor = np.expand_dims(processed_img, axis=0)\n\n        # Dự đoán\n        probs = model.predict(input_tensor, verbose=0)[0]\n        pred_index = np.argmax(probs)\n        confidence = probs[pred_index] * 100\n        pred_label = LABELS[pred_index]\n\n        # Hiển thị kết quả\n        plt.figure(figsize=(8, 6))\n        plt.imshow(original_img)\n        plt.axis('off')\n        \n        # Màu chữ tiêu đề: Xanh (tự tin cao), Đỏ (tự tin thấp)\n        color = 'green' if confidence > 70 else 'red'\n        \n        plt.title(f\"KẾT QUẢ DỰ ĐOÁN:\\n{pred_label}\\nĐộ tin cậy: {confidence:.2f}%\", \n                  color=color, fontsize=14, fontweight='bold', backgroundcolor='white')\n        plt.show()\n\n        # In thanh phần trăm chi tiết\n        print(f\"\\n📊 Phân tích chi tiết cho ảnh: {os.path.basename(image_path)}\")\n        print(\"-\" * 60)\n        for i, prob in enumerate(probs):\n            # Vẽ thanh tiến trình bằng text cho đẹp\n            bar_len = int(prob * 40)\n            bar = \"█\" * bar_len + \"░\" * (40 - bar_len)\n            print(f\"{LABELS[i]:<35} : {bar} {prob*100:.2f}%\")\n        print(\"-\" * 60)\n        \n    except Exception as e:\n        print(f\"❌ Có lỗi xảy ra khi xử lý ảnh: {e}\")\n\n# --- 4. CHẠY THỬ (TEST) ---\ndef predict_random_9_images():\n    if model is None: return\n\n    BASE_PATH = '/kaggle/input/cassava-leaf-disease-classification'\n    TRAIN_DIR = os.path.join(BASE_PATH, 'train_images')\n\n    if not os.path.exists(TRAIN_DIR):\n        print(\"⚠️ Không tìm thấy thư mục ảnh. Kiểm tra lại đường dẫn dataset.\")\n        return\n\n    # Lấy danh sách tất cả ảnh\n    all_images = os.listdir(TRAIN_DIR)\n    \n    # Chọn ngẫu nhiên 9 ảnh\n    random_images = random.sample(all_images, 9)\n    \n    print(f\"\\n🎲 Đang dự đoán 9 ảnh ngẫu nhiên...\")\n\n    # Tạo lưới vẽ 3x3\n    fig, axes = plt.subplots(3, 3, figsize=(15, 15))\n    fig.suptitle(\"KẾT QUẢ DỰ ĐOÁN NGẪU NHIÊN 9 ẢNH\", fontsize=16, y=0.95, fontweight='bold')\n    plt.subplots_adjust(hspace=0.4, wspace=0.1) # Chỉnh khoảng cách để không bị đè chữ\n\n    axes = axes.flatten()\n\n    for i, img_name in enumerate(random_images):\n        img_path = os.path.join(TRAIN_DIR, img_name)\n        ax = axes[i]\n\n        try:\n            # Đọc ảnh gốc\n            original_img = cv2.imread(img_path)\n            original_img = cv2.cvtColor(original_img, cv2.COLOR_BGR2RGB)\n\n            # Xử lý ảnh\n            augmented = PREPROCESS(image=original_img)\n            processed_img = augmented['image']\n            input_tensor = np.expand_dims(processed_img, axis=0)\n\n            # Dự đoán\n            probs = model.predict(input_tensor, verbose=0)[0]\n            pred_index = np.argmax(probs)\n            confidence = probs[pred_index] * 100\n            pred_label = LABELS[pred_index]\n\n            # Vẽ lên subplot\n            ax.imshow(original_img)\n            ax.axis('off')\n            \n            # Màu chữ: Xanh (tự tin cao), Vàng (khá), Đỏ (thấp)\n            if confidence > 80: color = 'green'\n            elif confidence > 50: color = '#d4ac0d' # Màu vàng đậm\n            else: color = 'red'\n\n            # Tiêu đề ảnh (Tên bệnh + Độ tin cậy)\n            # Dùng .split('(')[0] để lấy tên ngắn gọn cho đỡ dài dòng\n            short_name = pred_label.split('(')[0].strip()\n            title = f\"{short_name}\\n({confidence:.1f}%)\"\n            \n            ax.set_title(title, color=color, fontsize=11, fontweight='bold')\n\n        except Exception as e:\n            print(f\"Lỗi xử lý ảnh {img_name}: {e}\")\n            ax.axis('off')\n    \n    plt.show()\n\n# --- CHẠY HÀM ---\npredict_random_9_images()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T17:04:57.978367Z","iopub.execute_input":"2025-12-01T17:04:57.978976Z","iopub.status.idle":"2025-12-01T17:05:18.70016Z","shell.execute_reply.started":"2025-12-01T17:04:57.978951Z","shell.execute_reply":"2025-12-01T17:05:18.698816Z"}},"outputs":[],"execution_count":null}]}