{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":13836,"databundleVersionId":1718836,"isSourceIdPinned":false},{"sourceType":"datasetVersion","sourceId":15093554,"datasetId":9663611,"databundleVersionId":15978173}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Import Libraries","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-08T17:10:37.191962Z","iopub.execute_input":"2026-03-08T17:10:37.192974Z","iopub.status.idle":"2026-03-08T17:10:56.393748Z","shell.execute_reply.started":"2026-03-08T17:10:37.192940Z","shell.execute_reply":"2026-03-08T17:10:56.392932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport json, os, warnings\nwarnings.filterwarnings('ignore')\n\nimport tensorflow as tf\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\n\nprint(f\"✅ TensorFlow: {tf.__version__}\")\nprint(f\"✅ GPU: {len(tf.config.list_physical_devices('GPU')) > 0}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T17:11:15.550219Z","iopub.execute_input":"2026-03-08T17:11:15.550773Z","iopub.status.idle":"2026-03-08T17:11:24.065850Z","shell.execute_reply.started":"2026-03-08T17:11:15.550741Z","shell.execute_reply":"2026-03-08T17:11:24.065082Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Cell 2 — Load Data","metadata":{}},{"cell_type":"code","source":"import json, pandas as pd\n\nBASE_DIR = '/kaggle/input/competitions/cassava-leaf-disease-classification'\n\nwith open(f'{BASE_DIR}/label_num_to_disease_map.json') as f:\n    label_map = json.load(f)\n\nprint(\"🌿 Disease Classes:\")\nfor k, v in label_map.items():\n    print(f\"  Class {k}: {v}\")\n\ndf = pd.read_csv(f'{BASE_DIR}/train.csv')\ndf['disease_name'] = df['label'].astype(str).map(label_map)\nprint(f\"\\n📊 Total images: {len(df)}\")\nprint(df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T17:11:32.738831Z","iopub.execute_input":"2026-03-08T17:11:32.739396Z","iopub.status.idle":"2026-03-08T17:11:32.771864Z","shell.execute_reply.started":"2026-03-08T17:11:32.739353Z","shell.execute_reply":"2026-03-08T17:11:32.771073Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Cell 3 — Visualize Distribution","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nlabel_map_local = {\n    0: 'CBB', 1: 'CBSD', 2: 'CGM', 3: 'CMD', 4: 'Healthy'\n}\ndf['disease_name'] = df['label'].map(label_map_local)\n\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(14, 5))\n\ndf['label'].value_counts().sort_index().plot(kind='bar', ax=ax1, color='steelblue')\nax1.set_title('Class Distribution')\nax1.set_xlabel('Disease Class')\nax1.set_ylabel('Count')\nax1.tick_params(axis='x', rotation=0)\n\ndf['disease_name'].value_counts().plot(kind='pie', ax=ax2, autopct='%1.1f%%')\nax2.set_title('Disease Distribution (%)')\nax2.set_ylabel('')\n\nplt.tight_layout()\nplt.show()\nprint(f\"Total images: {len(df)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T19:31:10.502025Z","iopub.execute_input":"2026-03-08T19:31:10.502715Z","iopub.status.idle":"2026-03-08T19:31:10.709395Z","shell.execute_reply.started":"2026-03-08T19:31:10.502684Z","shell.execute_reply":"2026-03-08T19:31:10.708837Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Cell 4 — Show Sample Images","metadata":{}},{"cell_type":"code","source":"import os\nfrom PIL import Image\n\nimg_dir = f'{BASE_DIR}/train_images'\nfig, axes = plt.subplots(2, 5, figsize=(15, 7))\n\nfor label_id in range(5):\n    samples = df[df['label'] == label_id]['image_id'].values[:2]\n    for i, img_name in enumerate(samples):\n        img = Image.open(os.path.join(img_dir, img_name))\n        axes[i][label_id].imshow(img)\n        axes[i][label_id].set_title(f\"Class {label_id}\\n{label_map[str(label_id)][:20]}\", fontsize=8)\n        axes[i][label_id].axis('off')\n\nplt.suptitle('Sample Images per Disease Class', fontsize=14, fontweight='bold')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T19:31:23.758319Z","iopub.execute_input":"2026-03-08T19:31:23.758901Z","iopub.status.idle":"2026-03-08T19:31:24.767801Z","shell.execute_reply.started":"2026-03-08T19:31:23.758872Z","shell.execute_reply":"2026-03-08T19:31:24.767059Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Cell 5 — Prepare Data","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport numpy as np\n\n# Define ก่อนเลย!\nIMG_SIZE = 380\nBATCH_SIZE = 16\nimg_dir = '/kaggle/input/competitions/cassava-leaf-disease-classification/train_images'\n\ntrain_df, val_df = train_test_split(df, test_size=0.2, stratify=df['label'], random_state=42)\nprint(f\"Training: {len(train_df)} images\")\nprint(f\"Validation: {len(val_df)} images\")\n\nclasses = np.array(sorted(df['label'].unique()))\nweights = compute_class_weight('balanced', classes=classes, y=df['label'])\nclass_weight = dict(zip(classes, weights))\nprint(\"Class weights:\", class_weight)\n\ntrain_datagen = ImageDataGenerator(rescale=1./255, rotation_range=40, horizontal_flip=True, vertical_flip=True, zoom_range=0.2)\nval_datagen = ImageDataGenerator(rescale=1./255)\n\ntrain_df['label'] = train_df['label'].astype(str)\nval_df['label'] = val_df['label'].astype(str)\n\ntrain_gen = train_datagen.flow_from_dataframe(train_df, directory=img_dir, x_col='image_id', y_col='label', target_size=(IMG_SIZE, IMG_SIZE), batch_size=BATCH_SIZE, class_mode='categorical')\nval_gen = val_datagen.flow_from_dataframe(val_df, directory=img_dir, x_col='image_id', y_col='label', target_size=(IMG_SIZE, IMG_SIZE), batch_size=BATCH_SIZE, class_mode='categorical')\nprint(\"✅ Data ready!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T19:27:42.076364Z","iopub.execute_input":"2026-03-08T19:27:42.077028Z","iopub.status.idle":"2026-03-08T19:28:14.507409Z","shell.execute_reply.started":"2026-03-08T19:27:42.076995Z","shell.execute_reply":"2026-03-08T19:28:14.506749Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Cell 6 — Build Model (Transfer Learning)","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.applications import EfficientNetB4\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout, BatchNormalization\nfrom tensorflow.keras.models import Model\n\nbase_model = EfficientNetB4(\n    weights='imagenet',\n    include_top=False,\n    input_shape=(IMG_SIZE, IMG_SIZE, 3)\n)\n\n# Unfreeze 30 layers สุดท้าย\nbase_model.trainable = True\nfor layer in base_model.layers[:-30]:\n    layer.trainable = False\n\nx = base_model.output\nx = GlobalAveragePooling2D()(x)\nx = BatchNormalization()(x)\nx = Dense(256, activation='relu')(x)\nx = Dropout(0.5)(x)\npredictions = Dense(5, activation='softmax')(x)\n\nmodel = Model(inputs=base_model.input, outputs=predictions)\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=1e-4),\n    loss='categorical_crossentropy',\n    metrics=['accuracy']\n)\nprint(\"✅ EfficientNetB4 ready!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T19:28:19.174827Z","iopub.execute_input":"2026-03-08T19:28:19.175165Z","iopub.status.idle":"2026-03-08T19:28:21.177823Z","shell.execute_reply.started":"2026-03-08T19:28:19.175126Z","shell.execute_reply":"2026-03-08T19:28:21.177137Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":" Cell 7 — Train Model","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\n\ncallbacks = [\n    EarlyStopping(patience=5, restore_best_weights=True, verbose=1),\n    ReduceLROnPlateau(factor=0.5, patience=3, verbose=1)\n]\n\nhistory = model.fit(\n    train_gen,\n    epochs=15,\n    validation_data=val_gen,\n    callbacks=callbacks,\n    class_weight=class_weight\n)\nprint(\"✅ Training complete!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T17:11:45.197664Z","iopub.execute_input":"2026-03-08T17:11:45.198413Z","iopub.status.idle":"2026-03-08T18:53:02.217046Z","shell.execute_reply.started":"2026-03-08T17:11:45.198383Z","shell.execute_reply":"2026-03-08T18:53:02.216422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.save('/kaggle/working/cassava_model.h5')\nprint(\"✅ Model saved!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T18:57:47.890404Z","iopub.execute_input":"2026-03-08T18:57:47.891118Z","iopub.status.idle":"2026-03-08T18:57:49.088561Z","shell.execute_reply.started":"2026-03-08T18:57:47.891086Z","shell.execute_reply":"2026-03-08T18:57:49.087757Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Cell 8 — Plot Results","metadata":{}},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(14, 5))\n\nax1.plot(history.history['accuracy'], label='Train', color='blue')\nax1.plot(history.history['val_accuracy'], label='Validation', color='orange')\nax1.set_title('Model Accuracy')\nax1.set_xlabel('Epoch')\nax1.set_ylabel('Accuracy')\nax1.legend()\n\nax2.plot(history.history['loss'], label='Train', color='blue')\nax2.plot(history.history['val_loss'], label='Validation', color='orange')\nax2.set_title('Model Loss')\nax2.set_xlabel('Epoch')\nax2.set_ylabel('Loss')\nax2.legend()\n\nplt.tight_layout()\nplt.show()\n\nbest_acc = max(history.history['val_accuracy'])\nprint(f\"🎯 Best Validation Accuracy: {best_acc*100:.2f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T18:58:00.242516Z","iopub.execute_input":"2026-03-08T18:58:00.242852Z","iopub.status.idle":"2026-03-08T18:58:00.515660Z","shell.execute_reply.started":"2026-03-08T18:58:00.242825Z","shell.execute_reply":"2026-03-08T18:58:00.514979Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Cell 9 — Predict Sample Image","metadata":{}},{"cell_type":"code","source":"import os, numpy as np, matplotlib.pyplot as plt\nfrom PIL import Image\nimport json, pandas as pd\nfrom tensorflow.keras.models import load_model\n\nmodel = load_model('/kaggle/working/cassava_model.h5')\nBASE_DIR = '/kaggle/input/competitions/cassava-leaf-disease-classification'\nwith open(f'{BASE_DIR}/label_num_to_disease_map.json') as f:\n    label_map = json.load(f)\ndf = pd.read_csv(f'{BASE_DIR}/train.csv')\nimg_dir = '/kaggle/input/competitions/cassava-leaf-disease-classification/train_images'\nIMG_SIZE = 380\n\ndef predict_disease(image_path, true_label):\n    img = Image.open(image_path).resize((IMG_SIZE, IMG_SIZE))\n    img_array = np.expand_dims(np.array(img)/255.0, axis=0)\n    pred = model.predict(img_array, verbose=0)\n    class_id = np.argmax(pred)\n    confidence = pred[0][class_id] * 100\n    disease = label_map[str(class_id)]\n    plt.figure(figsize=(5, 4))\n    plt.imshow(Image.open(image_path))\n    plt.title(f\"True: {label_map[str(true_label)]}\\nPredicted: {disease} ({confidence:.1f}%)\", fontweight='bold')\n    plt.axis('off')\n    plt.show()\n    print(f\"True: {label_map[str(true_label)]} | Predicted: {disease} | Confidence: {confidence:.1f}%\")\n    return class_id\n\nfor label_id in range(5):\n    samples = df[df['label']==label_id]['image_id'].values[:500]\n    found = False\n    for img_name in samples:\n        test_img = f'{img_dir}/{img_name}'\n        img = Image.open(test_img).resize((IMG_SIZE, IMG_SIZE))\n        img_array = np.expand_dims(np.array(img)/255.0, axis=0)\n        pred = model.predict(img_array, verbose=0)\n        class_id = np.argmax(pred)\n        if class_id == label_id:\n            predict_disease(test_img, label_id)\n            found = True\n            break\n    if not found:\n        print(f\"❌ false {label_map[str(label_id)]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T19:13:03.863412Z","iopub.execute_input":"2026-03-08T19:13:03.864048Z","iopub.status.idle":"2026-03-08T19:14:48.110529Z","shell.execute_reply.started":"2026-03-08T19:13:03.864010Z","shell.execute_reply":"2026-03-08T19:14:48.109761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\n\nBASE_DIR = '/kaggle/input/competitions/cassava-leaf-disease-classification'\nimg_dir = f'{BASE_DIR}/train_images'\n\n# ดึงรูปตัวอย่างโรคละ 1 รูป\nfor label_id in range(5):\n    sample = df[df['label']==label_id]['image_id'].iloc[10]\n    src = f'{img_dir}/{sample}'\n    dst = f'/kaggle/working/class_{label_id}_{label_map[str(label_id)][:10]}.jpg'\n    shutil.copy(src, dst)\n    print(f\"✅ Saved: {dst}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}