{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":150545,"sourceType":"datasetVersion","datasetId":70909}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\nfrom tensorflow.keras.optimizers import Adam\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport os\nimport shutil\nimport math","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-09T14:07:04.140897Z","iopub.execute_input":"2025-09-09T14:07:04.141185Z","iopub.status.idle":"2025-09-09T14:07:18.300244Z","shell.execute_reply.started":"2025-09-09T14:07:04.141137Z","shell.execute_reply":"2025-09-09T14:07:18.299685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Cài đặt cơ bản ---\nIMAGE_SIZE = (128, 128) # Kích thước ảnh đầu vào cho mô hình\nBATCH_SIZE = 32\nNUM_CLASSES = None # Sẽ được xác định tự động\nEPOCHS = 10 # Số lượng epochs để huấn luyện. Bạn có thể tăng lên nếu muốn độ chính xác cao hơn.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-09T14:07:18.30136Z","iopub.execute_input":"2025-09-09T14:07:18.301776Z","iopub.status.idle":"2025-09-09T14:07:18.30586Z","shell.execute_reply.started":"2025-09-09T14:07:18.301757Z","shell.execute_reply":"2025-09-09T14:07:18.305212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dir = '/kaggle/input/plantdisease/PlantVillage' ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-09T14:07:18.306624Z","iopub.execute_input":"2025-09-09T14:07:18.306896Z","iopub.status.idle":"2025-09-09T14:07:18.479726Z","shell.execute_reply.started":"2025-09-09T14:07:18.306871Z","shell.execute_reply":"2025-09-09T14:07:18.479049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 1. Tiền xử lý dữ liệu và Data Augmentation ---\n\n# Tạo ImageDataGenerator cho tập huấn luyện và kiểm tra\n# Rescale pixel values từ [0, 255] về [0, 1]\n# Data augmentation cho tập huấn luyện để tăng khả năng tổng quát hóa\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    rotation_range=20, # Xoay ảnh 20 độ\n    width_shift_range=0.2, # Dịch chuyển ảnh theo chiều ngang\n    height_shift_range=0.2, # Dịch chuyển ảnh theo chiều dọc\n    validation_split=0.2 # Chia 20% dữ liệu cho tập validation\n)\n\n# Chỉ rescale cho tập validation/test, không augmentation\nvalidation_datagen = ImageDataGenerator(\n    rescale=1./255,\n    validation_split=0.2\n)\n\n# Load dữ liệu từ các thư mục\nprint(\"Loading training data...\")\ntrain_generator = train_datagen.flow_from_directory(\n    data_dir,\n    target_size=IMAGE_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode='categorical', # Vì có nhiều lớp bệnh\n    subset='training' # Chỉ lấy tập huấn luyện\n)\n\nprint(\"\\nLoading validation data...\")\nvalidation_generator = validation_datagen.flow_from_directory(\n    data_dir,\n    target_size=IMAGE_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode='categorical', # Vì có nhiều lớp bệnh\n    subset='validation' # Chỉ lấy tập validation\n)\n\nNUM_CLASSES = train_generator.num_classes\nclass_names = list(train_generator.class_indices.keys())\nprint(f\"\\nDetected {NUM_CLASSES} classes: {class_names}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-09T14:07:18.480419Z","iopub.execute_input":"2025-09-09T14:07:18.480597Z","iopub.status.idle":"2025-09-09T14:07:32.322498Z","shell.execute_reply.started":"2025-09-09T14:07:18.480582Z","shell.execute_reply":"2025-09-09T14:07:32.32191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 2. Xây dựng Mô hình CNN (Convolutional Neural Network) ---\n\n# Sử dụng mô hình Sequential\nmodel = Sequential([\n    # Lớp Convolutional đầu tiên\n    Conv2D(32, (3, 3), activation='relu', input_shape=(IMAGE_SIZE[0], IMAGE_SIZE[1], 3)),\n    MaxPooling2D(pool_size=(2, 2)),\n    \n    # Lớp Convolutional thứ hai\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D(pool_size=(2, 2)),\n    \n    # Lớp Convolutional thứ ba\n    Conv2D(128, (3, 3), activation='relu'),\n    MaxPooling2D(pool_size=(2, 2)),\n    \n    # San phẳng output để đưa vào lớp Dense\n    Flatten(),\n    \n    # Lớp Dense ẩn\n    Dense(512, activation='relu'),\n    Dropout(0.5), # Regularization để tránh overfitting\n    \n    # Lớp đầu ra (output layer)\n    Dense(NUM_CLASSES, activation='softmax') # Softmax cho phân loại đa lớp\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-09T14:07:32.324103Z","iopub.execute_input":"2025-09-09T14:07:32.3247Z","iopub.status.idle":"2025-09-09T14:07:34.38655Z","shell.execute_reply.started":"2025-09-09T14:07:32.324675Z","shell.execute_reply":"2025-09-09T14:07:34.385774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 3. Compile Mô hình ---\nmodel.compile(optimizer=Adam(learning_rate=0.001), # Tối ưu hóa Adam\n              loss='categorical_crossentropy', # Hàm mất mát cho phân loại đa lớp\n              metrics=['accuracy'])\n\nmodel.summary() # Hiển thị tóm tắt cấu trúc mô hình","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-09T14:07:34.387342Z","iopub.execute_input":"2025-09-09T14:07:34.387603Z","iopub.status.idle":"2025-09-09T14:07:34.412622Z","shell.execute_reply.started":"2025-09-09T14:07:34.387581Z","shell.execute_reply":"2025-09-09T14:07:34.412084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 4. Huấn luyện Mô hình ---\nprint(\"\\nStarting model training...\")\nhistory = model.fit(\n    train_generator,\n    steps_per_epoch=math.ceil(train_generator.samples / BATCH_SIZE),\n    epochs=EPOCHS,\n    validation_data=validation_generator,\n    validation_steps=math.ceil(validation_generator.samples / BATCH_SIZE)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-09T14:07:34.413278Z","iopub.execute_input":"2025-09-09T14:07:34.413559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 5. Đánh giá Mô hình ---\nprint(\"\\nEvaluating model on validation data...\")\nloss, accuracy = model.evaluate(validation_generator)\nprint(f\"Validation Loss: {loss:.4f}\")\nprint(f\"Validation Accuracy: {accuracy:.4f}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 6. Trực quan hóa kết quả huấn luyện ---\nplt.figure(figsize=(12, 4))\nplt.subplot(1, 2, 1)\nplt.plot(history.history['accuracy'], label='Training Accuracy')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\nplt.title('Training and Validation Accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\n\nplt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='Training Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Training and Validation Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 7. Lưu Mô hình (tùy chọn) ---\nmodel_save_path = 'plant_disease_classifier.h5'\nmodel.save(model_save_path)\nprint(f\"\\nModel saved to {model_save_path}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 8. Ví dụ dự đoán (tùy chọn) ---\nprint(\"\\nExample Prediction:\")\n\n# Ví dụ:\nimage_path = '/kaggle/input/plantdisease/PlantVillage/Tomato_mosaic_virus/0a86d9a9-178b-491c-b5f7-872f23246a06___UMD_Powd.M 003_flipLR.JPG' \n\nfrom tensorflow.keras.preprocessing import image\n\n\nimg = image.load_img(image_path, target_size=IMAGE_SIZE)\n\nsample_image = image.img_to_array(img)\n\nsample_image_expanded = np.expand_dims(sample_image, axis=0) / 255.0 \n\ntrue_label_name = 'Tomato_mosaic_virus' \n\npredictions = model.predict(sample_image_expanded)\npredicted_class_index = np.argmax(predictions[0])\npredicted_class_name = class_names[predicted_class_index]\nconfidence = np.max(predictions[0]) * 100\n\nprint(f\"True Label: {true_label_name}\")\nprint(f\"Predicted Label: {predicted_class_name} (Confidence: {confidence:.2f}%)\")\n\nplt.imshow(sample_image / 255.0) # Hiển thị ảnh đã chuẩn hóa\nplt.title(f\"True: {true_label_name}\\nPredicted: {predicted_class_name} ({confidence:.2f}%)\")\nplt.axis('off')\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}