{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.10","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":25563,"databundleVersionId":2094376,"sourceType":"competition"},{"sourceId":2240903,"sourceType":"datasetVersion","datasetId":1337312}],"dockerImageVersionId":30096,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nfrom pathlib import Path\nfrom PIL import Image\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\n\n# --- 1. CONFIGURATION  ---\nclass config:\n    # Đường dẫn gốc\n    DATA_DIR = Path('../input/plant-pathology-2021-fgvc8/train_images')\n    CSV_PATH = Path('../input/plant-pathology-2021-fgvc8/train.csv')\n    \n    # Nơi lưu ảnh sau khi resize (Working directory)\n    ROOT_DIR = Path('./data')\n    \n    # Thông số ảnh\n    INPUT_HEIGHT = 224\n    INPUT_WIDTH = 224\n    \n    # Training params\n    BATCH_SIZE = 32\n    DEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    \n    # Sẽ cập nhật sau khi đọc dữ liệu\n    NUM_CLASSES = 0 \n    LABELS = []\n\n# --- 2. DATA SPLITTING FUNCTION  ---\ndef split_df(csv_dir):\n    \n    df = pd.read_csv(csv_dir)\n    \n    # Mã hóa nhãn (Labels)\n    le = LabelEncoder()\n    df['labels_n'] = le.fit_transform(df['labels'].values)\n    \n    # Cập nhật Config\n    config.LABELS = list(le.classes_)\n    config.NUM_CLASSES = len(le.classes_)\n    \n    # Cập nhật đường dẫn ảnh trỏ về thư mục resize\n    # Lưu ý: Lúc này ảnh chưa có ở đó, bước sau sẽ tạo\n    df['image'] = df['image'].apply(lambda x: str(config.ROOT_DIR / x))\n\n    \n    train_df, dummy_df = train_test_split(\n        df, \n        train_size=0.7, \n        shuffle=True, \n        random_state=42, \n        stratify=df['labels_n']\n    )\n\n    # 2. Tách Temp thành Valid (15%) và Test (15%)\n    valid_df, test_df = train_test_split(\n        dummy_df, \n        train_size=0.5, \n        shuffle=True, \n        random_state=42, \n        stratify=dummy_df['labels_n']\n    )\n\n    return train_df, valid_df, test_df\n\n# Thực hiện chia dữ liệu\nprint(\"Đang đọc và chia dữ liệu...\")\ntrain_df, valid_df, test_df = split_df(config.CSV_PATH)\n\nprint(f\"Train size: {len(train_df)}\")\nprint(f\"Valid size: {len(valid_df)}\")\nprint(f\"Test size : {len(test_df)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-01T17:10:38.112159Z","iopub.execute_input":"2026-01-01T17:10:38.112536Z","iopub.status.idle":"2026-01-01T17:10:38.245143Z","shell.execute_reply.started":"2026-01-01T17:10:38.112506Z","shell.execute_reply":"2026-01-01T17:10:38.244141Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import multiprocessing\nfrom joblib import Parallel, delayed\n\n# --- RESIZE IMAGES (PARALLEL VERSION) ---\n# Tạo thư mục chứa dữ liệu\nif not os.path.exists(config.ROOT_DIR):\n    os.makedirs(config.ROOT_DIR)\n\nprint(f\"Đang thực hiện Resize ảnh trên {multiprocessing.cpu_count()} nhân CPU...\")\n\n\ndef resize_one_image(img_path_str):\n    # Lấy tên file gốc\n    file_name = os.path.basename(img_path_str)\n    src_path = config.DATA_DIR / file_name\n    dst_path = Path(img_path_str)\n    \n    # Chỉ resize nếu chưa có\n    if not dst_path.exists():\n        try:\n            with Image.open(src_path) as img:\n                img = img.resize((config.INPUT_WIDTH, config.INPUT_HEIGHT), Image.BILINEAR)\n                img.save(dst_path)\n        except Exception as e:\n            pass\n\n# Lấy danh sách ảnh\nall_images = pd.concat([train_df, valid_df, test_df])['image'].values\n\n\n\nParallel(n_jobs=-1, backend=\"threading\")(\n    delayed(resize_one_image)(path) for path in tqdm(all_images, desc=\"Resizing Parallel\")\n)\n\nprint(\"\\nResize hoàn tất! Dữ liệu đã sẵn sàng tại ./data\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-01T17:10:38.246538Z","iopub.execute_input":"2026-01-01T17:10:38.246811Z","iopub.status.idle":"2026-01-01T17:33:43.605137Z","shell.execute_reply.started":"2026-01-01T17:10:38.246785Z","shell.execute_reply":"2026-01-01T17:33:43.603987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- DATASET CLASS ---\nclass PlantDataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df = df\n        self.transform = transform\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, index):\n        row = self.df.iloc[index]\n        image_path = row['image']\n        label = row['labels_n']\n        \n        # Mở ảnh\n        image = Image.open(image_path).convert(\"RGB\")\n        \n        if self.transform:\n            image = self.transform(image)\n            \n        return image, torch.tensor(label, dtype=torch.long)\n\n# --- TRANSFORMS ---\n# Chuẩn ImageNet\nmean = [0.485, 0.456, 0.406]\nstd = [0.229, 0.224, 0.225]\n\ntrain_transforms = transforms.Compose([\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomVerticalFlip(),\n    transforms.RandomRotation(15),\n    transforms.ColorJitter(brightness=0.1, contrast=0.1),\n    transforms.ToTensor(),\n    transforms.Normalize(mean, std)\n])\n\ntest_transforms = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize(mean, std)\n])\n\n# --- DATALOADERS ---\ntrain_ds = PlantDataset(train_df, transform=train_transforms)\nvalid_ds = PlantDataset(valid_df, transform=test_transforms)\ntest_ds = PlantDataset(test_df, transform=test_transforms)\n\ntrain_loader = DataLoader(train_ds, batch_size=config.BATCH_SIZE, shuffle=True, num_workers=2)\nvalid_loader = DataLoader(valid_ds, batch_size=config.BATCH_SIZE, shuffle=False, num_workers=2)\ntest_loader = DataLoader(test_ds, batch_size=config.BATCH_SIZE, shuffle=False, num_workers=2)\n\nprint(\"Dataloaders đã sẵn sàng.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-01T17:33:43.607525Z","iopub.execute_input":"2026-01-01T17:33:43.607865Z","iopub.status.idle":"2026-01-01T17:33:43.620329Z","shell.execute_reply.started":"2026-01-01T17:33:43.607826Z","shell.execute_reply":"2026-01-01T17:33:43.619382Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- MODEL DEFINITION: DENSENET ---\nclass DenseLayer(nn.Module):\n    def __init__(self, in_channels, growth_rate):\n        super(DenseLayer, self).__init__()\n        self.bn1 = nn.BatchNorm2d(in_channels)\n        self.relu = nn.ReLU(inplace=True)\n        self.conv1 = nn.Conv2d(in_channels, 4 * growth_rate, kernel_size=1, bias=False)\n        self.bn2 = nn.BatchNorm2d(4 * growth_rate)\n        self.conv2 = nn.Conv2d(4 * growth_rate, growth_rate, kernel_size=3, padding=1, bias=False)\n\n    def forward(self, x):\n        out = self.conv1(self.relu(self.bn1(x)))\n        out = self.conv2(self.relu(self.bn2(out)))\n        return torch.cat([x, out], 1)\n\nclass DenseBlock(nn.Module):\n    def __init__(self, num_layers, in_channels, growth_rate):\n        super(DenseBlock, self).__init__()\n        layers = []\n        for i in range(num_layers):\n            layers.append(DenseLayer(in_channels + i * growth_rate, growth_rate))\n        self.layer = nn.Sequential(*layers)\n\n    def forward(self, x):\n        return self.layer(x)\n\nclass TransitionLayer(nn.Module):\n    def __init__(self, in_channels, out_channels):\n        super(TransitionLayer, self).__init__()\n        self.bn = nn.BatchNorm2d(in_channels)\n        self.relu = nn.ReLU(inplace=True)\n        self.conv = nn.Conv2d(in_channels, out_channels, kernel_size=1, bias=False)\n        self.avg_pool = nn.AvgPool2d(kernel_size=2, stride=2)\n\n    def forward(self, x):\n        out = self.conv(self.relu(self.bn(x)))\n        out = self.avg_pool(out)\n        return out\n\nclass ResearcherDenseNet(nn.Module):\n    def __init__(self, num_classes, growth_rate=32):\n        super(ResearcherDenseNet, self).__init__()\n        # Initial Conv\n        self.features = nn.Sequential(\n            nn.Conv2d(3, 64, kernel_size=7, stride=2, padding=3, bias=False),\n            nn.BatchNorm2d(64),\n            nn.ReLU(inplace=True),\n            nn.MaxPool2d(kernel_size=3, stride=2, padding=1)\n        )\n        \n        num_features = 64\n        block_config = (6, 12, 16) # Cấu hình 3 blocks\n        \n        for i, num_layers in enumerate(block_config):\n            block = DenseBlock(num_layers, num_features, growth_rate)\n            self.features.add_module(f'denseblock{i+1}', block)\n            num_features = num_features + num_layers * growth_rate\n            \n            if i != len(block_config) - 1:\n                out_features = num_features // 2\n                trans = TransitionLayer(num_features, out_features)\n                self.features.add_module(f'transition{i+1}', trans)\n                num_features = out_features\n\n        self.final_bn = nn.BatchNorm2d(num_features)\n        self.classifier = nn.Linear(num_features, num_classes)\n\n    def forward(self, x):\n        out = self.features(x)\n        out = torch.relu(self.final_bn(out))\n        out = torch.nn.functional.adaptive_avg_pool2d(out, (1, 1))\n        out = torch.flatten(out, 1)\n        out = self.classifier(out)\n        return out\n\n# Khởi tạo Model\nmodel = ResearcherDenseNet(num_classes=config.NUM_CLASSES).to(config.DEVICE)\nprint(\"Model DenseNet đã được khởi tạo.\")\n\n# --- TRAINING LOOP ---\ncriterion = nn.CrossEntropyLoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=1e-3)\nscheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.1, patience=2)\n\nEPOCHS = 10 \nbest_acc = 0.0\n\nprint(f\"Bắt đầu Training {EPOCHS} Epochs...\")\n\nfor epoch in range(EPOCHS):\n    model.train()\n    running_loss = 0.0\n    correct = 0\n    total = 0\n    \n    # Train\n    loop = tqdm(train_loader, desc=f\"Epoch {epoch+1}/{EPOCHS}\")\n    for images, labels in loop:\n        images, labels = images.to(config.DEVICE), labels.to(config.DEVICE)\n        \n        optimizer.zero_grad()\n        outputs = model(images)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        \n        running_loss += loss.item()\n        _, predicted = outputs.max(1)\n        total += labels.size(0)\n        correct += predicted.eq(labels).sum().item()\n        \n        loop.set_postfix(loss=loss.item())\n    \n    train_acc = 100 * correct / total\n    avg_loss = running_loss / len(train_loader)\n    \n    # Validation\n    model.eval()\n    val_correct = 0\n    val_total = 0\n    with torch.no_grad():\n        for images, labels in valid_loader:\n            images, labels = images.to(config.DEVICE), labels.to(config.DEVICE)\n            outputs = model(images)\n            _, predicted = outputs.max(1)\n            val_total += labels.size(0)\n            val_correct += predicted.eq(labels).sum().item()\n            \n    val_acc = 100 * val_correct / val_total\n    \n    print(f\"Epoch {epoch+1}: Loss={avg_loss:.4f} | Train Acc={train_acc:.2f}% | Valid Acc={val_acc:.2f}%\")\n    \n    scheduler.step(avg_loss)\n    \n    if val_acc > best_acc:\n        best_acc = val_acc\n        torch.save(model.state_dict(), \"best_model.pth\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-01T17:33:43.621555Z","iopub.execute_input":"2026-01-01T17:33:43.621925Z","iopub.status.idle":"2026-01-01T18:01:15.294934Z","shell.execute_reply.started":"2026-01-01T17:33:43.621891Z","shell.execute_reply":"2026-01-01T18:01:15.293895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import classification_report, confusion_matrix, accuracy_score\nimport numpy as np\nimport os\nimport torch\nimport warnings\nfrom tqdm import tqdm\n\nwarnings.filterwarnings('ignore')\n\nprint(\"--- ĐANG CHẠY ĐÁNH GIÁ TRÊN TẬP TEST ---\")\n\n# 1. DANH SÁCH 6 NHÃN MỤC TIÊU (GOM GỌN)\nFINAL_LABELS = ['complex', 'frog_eye_leaf_spot', 'healthy', 'powdery_mildew', 'rust', 'scab']\n\n# 2. HÀM GOM NHÓM (QUY TẮC GHÉP BỆNH)\ndef simplify_label(original_label):\n    # Quy tắc: Nếu trong tên bệnh có từ khóa thì gán về bệnh đó\n    if 'scab' in original_label: return 'scab'\n    if 'rust' in original_label: return 'rust'\n    if 'powdery_mildew' in original_label: return 'powdery_mildew'\n    if 'frog_eye_leaf_spot' in original_label: return 'frog_eye_leaf_spot'\n    if 'healthy' in original_label: return 'healthy'\n    return 'complex' \n\n# 3. Load Model & Dự đoán\nif os.path.exists(\"best_model.pth\"):\n    model.load_state_dict(torch.load(\"best_model.pth\"))\n    print(\"-> Đã nạp model tốt nhất (Best Weights).\")\nelse:\n    print(\"-> Cảnh báo: Đang dùng model hiện tại.\")\n\nmodel.eval()\n\ny_true_raw = []\ny_pred_raw = []\n\nwith torch.no_grad():\n    for images, labels in tqdm(test_loader, desc=\"Predicting\"):\n        images, labels = images.to(config.DEVICE), labels.to(config.DEVICE)\n        outputs = model(images)\n        _, predicted = outputs.max(1)\n        y_true_raw.extend(labels.cpu().numpy())\n        y_pred_raw.extend(predicted.cpu().numpy())\n\n# 4. XỬ LÝ KẾT QUẢ (GOM VỀ 6 LOẠI)\nraw_label_names = config.LABELS \ny_true_clean = [simplify_label(raw_label_names[i]) for i in y_true_raw]\ny_pred_clean = [simplify_label(raw_label_names[i]) for i in y_pred_raw]\n\n# 5. IN KẾT QUẢ ACCURACY \nfinal_acc = accuracy_score(y_true_clean, y_pred_clean)\nprint(\"\\n\" + \"#\"*60)\nprint(f\"### KẾT QUẢ ĐỘ CHÍNH XÁC (TEST ACCURACY): {final_acc * 100:.2f}% ###\")\nprint(\"#\"*60 + \"\\n\")\n\n# 6. BÁO CÁO CHI TIẾT\nprint(\"BÁO CÁO CHI TIẾT TỪNG LOẠI BỆNH:\")\nprint(classification_report(y_true_clean, y_pred_clean, target_names=FINAL_LABELS, zero_division=0))\n\n# 7. VẼ MA TRẬN NHẦM LẪN\nprint(\"Confusion Matrix\")\ncm = confusion_matrix(y_true_clean, y_pred_clean, labels=FINAL_LABELS)\n\nplt.figure(figsize=(10, 8))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Greens', \n            xticklabels=FINAL_LABELS, \n            yticklabels=FINAL_LABELS)\n\nplt.xlabel('Model Dự Đoán (Predicted)', fontsize=12)\nplt.ylabel('Thực Tế (Truth)', fontsize=12)\nplt.title(f'Confusion Matrix (Accuracy: {final_acc*100:.2f}%)', fontsize=14)\nplt.xticks(rotation=45)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-01T18:01:15.296596Z","iopub.execute_input":"2026-01-01T18:01:15.296891Z","iopub.status.idle":"2026-01-01T18:01:25.213667Z","shell.execute_reply.started":"2026-01-01T18:01:15.296857Z","shell.execute_reply":"2026-01-01T18:01:25.212839Z"}},"outputs":[],"execution_count":null}]}