{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 🌿 Cassava Leaf Disease Classification  \n### Smart Leaf Disease Detection System using EfficientNetB0 & Transfer Learning\n\n---\n\n## **Project Overview**\nThis notebook presents a complete deep-learning pipeline for identifying cassava leaf diseases using the official dataset from the *Kaggle Cassava Leaf Disease Classification* challenge.  \nThe workflow relies on **EfficientNetB0**, **transfer learning**, and a two-stage fine-tuning strategy to build a reliable image-based classifier that works under real farm conditions.\n\n---\n\n## **Objectives**\n- Load and explore the cassava dataset  \n- Build a classifier using **EfficientNetB0** as the backbone  \n- Apply **data augmentation**, **class balancing**, and **focal loss**  \n- Train the model in two stages: frozen backbone → fine-tuning  \n- Evaluate performance using accuracy, F1-score, and confusion matrix  \n- Visualize model decisions using **Grad-CAM**  \n- Generate the final **submission.csv** file for Kaggle  \n\n---\n\n## **Dataset Description**\nThe dataset contains **21,367 RGB images** of cassava leaves collected under natural agricultural environments.  \nEach image belongs to one of five disease categories:\n\n| Label | Disease Description |\n|-------|----------------------|\n| **0** | Cassava Bacterial Blight (CBB) |\n| **1** | Cassava Brown Streak Disease (CBSD) |\n| **2** | Cassava Green Mottle (CGM) |\n| **3** | Cassava Mosaic Disease (CMD) |\n| **4** | Healthy |\n\n---\n\n## **Modeling Approach**\nThe model is trained using a two-phase strategy:\n\n### **Stage 1 — Frozen EfficientNetB0 Backbone**\n- Load EfficientNetB0 pretrained on ImageNet  \n- Freeze all convolutional layers  \n- Train only the classification head  \n\n### **Stage 2 — Fine-Tuning**\n- Unfreeze upper backbone layers  \n- Retrain with a **lower learning rate**  \n- Improve feature extraction and overall accuracy  \n\nThis workflow provides stable training early on and better specialization later.\n\n---\n\n## **Notebook Structure**\n- Exploratory Data Analysis (EDA)  \n- Image statistics and visual inspection  \n- Data pipeline and augmentation  \n- Model construction (EfficientNetB0)  \n- Two-stage training  \n- Evaluation metrics and confusion matrix  \n- Grad-CAM interpretability  \n- Test inference & submission file  \n\n---\n","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# import paskagees ","metadata":{"execution":{"iopub.status.busy":"2025-11-23T14:14:33.124701Z","iopub.status.idle":"2025-11-23T14:14:33.124932Z","shell.execute_reply.started":"2025-11-23T14:14:33.124813Z","shell.execute_reply":"2025-11-23T14:14:33.124824Z"}}},{"cell_type":"code","source":"# to hidden any warrrnign \nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:45:49.020028Z","iopub.execute_input":"2025-11-28T11:45:49.020581Z","iopub.status.idle":"2025-11-28T11:45:49.024586Z","shell.execute_reply.started":"2025-11-28T11:45:49.020558Z","shell.execute_reply":"2025-11-28T11:45:49.023773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, sys, warnings\n\ndef init_env():\n    \"\"\"\n    Suppress TensorFlow, protobuf, and general runtime warnings.\n    Ensures cleaner notebook logs and avoids unnecessary noise.\n    \"\"\"\n    warnings.filterwarnings(\"ignore\")\n    os.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'\n    \n    # Silence annoying protobuf warnings in Kaggle environment\n    sys.stderr = open(os.devnull, 'w')\n    try:\n        import google.protobuf.message_factory as mf\n        if not hasattr(mf.MessageFactory, \"GetPrototype\"):\n            mf.MessageFactory.GetPrototype = lambda self, desc: None\n    except:\n        pass\ninit_env()\ninit_env()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:45:49.025864Z","iopub.execute_input":"2025-11-28T11:45:49.026128Z","iopub.status.idle":"2025-11-28T11:45:49.131726Z","shell.execute_reply.started":"2025-11-28T11:45:49.026099Z","shell.execute_reply":"2025-11-28T11:45:49.131300Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport json\nimport cv2\nimport gc\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set(style=\"whitegrid\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:45:49.132624Z","iopub.execute_input":"2025-11-28T11:45:49.132903Z","iopub.status.idle":"2025-11-28T11:45:50.414987Z","shell.execute_reply.started":"2025-11-28T11:45:49.132880Z","shell.execute_reply":"2025-11-28T11:45:50.414553Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import backend as K\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.layers import (\n    Dense, Dropout, BatchNormalization,\n    GlobalAveragePooling2D, Input\n)\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import (\n    EarlyStopping, ModelCheckpoint,\n    ReduceLROnPlateau, CSVLogger,\n    LearningRateScheduler\n)\n\n\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom tensorflow.keras.layers import LayerNormalization\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:45:50.416988Z","iopub.execute_input":"2025-11-28T11:45:50.417440Z","iopub.status.idle":"2025-11-28T11:46:06.558001Z","shell.execute_reply.started":"2025-11-28T11:45:50.417421Z","shell.execute_reply":"2025-11-28T11:46:06.557414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report\nfrom sklearn.metrics import balanced_accuracy_score\n\nfrom sklearn.model_selection import train_test_split\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:06.558836Z","iopub.execute_input":"2025-11-28T11:46:06.559268Z","iopub.status.idle":"2025-11-28T11:46:06.668213Z","shell.execute_reply.started":"2025-11-28T11:46:06.559249Z","shell.execute_reply":"2025-11-28T11:46:06.667651Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"TensorFlow version:\", tf.__version__)\nprint(\"Num GPUs:\", len(tf.config.list_physical_devices('GPU')))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:06.669374Z","iopub.execute_input":"2025-11-28T11:46:06.669612Z","iopub.status.idle":"2025-11-28T11:46:07.264130Z","shell.execute_reply.started":"2025-11-28T11:46:06.669586Z","shell.execute_reply":"2025-11-28T11:46:07.263469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#  determin raandom \ntf.random.set_seed(42)\nnp.random.seed(42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:07.265109Z","iopub.execute_input":"2025-11-28T11:46:07.265415Z","iopub.status.idle":"2025-11-28T11:46:07.283852Z","shell.execute_reply.started":"2025-11-28T11:46:07.265388Z","shell.execute_reply":"2025-11-28T11:46:07.283307Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PATHS ","metadata":{}},{"cell_type":"markdown","source":"________________","metadata":{}},{"cell_type":"code","source":"ROOT = \"/kaggle/input/\"\nDATASET = \"cassava-leaf-disease-classification\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:07.284698Z","iopub.execute_input":"2025-11-28T11:46:07.284883Z","iopub.status.idle":"2025-11-28T11:46:07.299493Z","shell.execute_reply.started":"2025-11-28T11:46:07.284869Z","shell.execute_reply":"2025-11-28T11:46:07.298944Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_DIR = os.path.join(ROOT, DATASET)\nTRAIN_IMG_DIR = os.path.join(DATA_DIR, \"train_images\")\nTEST_IMG_DIR = os.path.join(DATA_DIR, \"test_images\")\n\nTRAIN_CSV = os.path.join(DATA_DIR, \"train.csv\")\nLABEL_JSON = os.path.join(DATA_DIR, \"label_num_to_disease_map.json\")\nSUBMISSION_TEMPLATE = os.path.join(DATA_DIR, \"sample_submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:07.300381Z","iopub.execute_input":"2025-11-28T11:46:07.300680Z","iopub.status.idle":"2025-11-28T11:46:07.322041Z","shell.execute_reply.started":"2025-11-28T11:46:07.300657Z","shell.execute_reply":"2025-11-28T11:46:07.321501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# FOCAL LOSS (for imbalanced data)\nNUM_CLASSES = 5  # we have 5 cassava classes\n\ndef sparse_focal_loss(gamma=2.0):\n    \"\"\"\n    Sparse Categorical Focal Loss for multi-class problems.\n    y_true: integer class labels (0..4)\n    y_pred: softmax probabilities from the model\n    \"\"\"\n    def loss_fn(y_true, y_pred):\n        # make sure y_true is int\n        y_true = tf.cast(y_true, tf.int32)\n\n        # one-hot encode: (batch,) -> (batch, NUM_CLASSES)\n        y_true_oh = tf.one_hot(y_true, depth=NUM_CLASSES)\n\n        # avoid log(0)\n        y_pred_clipped = tf.clip_by_value(y_pred, 1e-7, 1.0 - 1e-7)\n\n        # standard cross-entropy\n        ce = -tf.reduce_sum(y_true_oh * tf.math.log(y_pred_clipped), axis=-1)\n\n        # probability of the true class\n        p_t = tf.reduce_sum(y_true_oh * y_pred, axis=-1)\n\n        # focal factor => samples that are already easy (p_t high) get lower weight\n        focal_factor = tf.pow(1.0 - p_t, gamma)\n\n        return focal_factor * ce\n\n    return loss_fn\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:07.324927Z","iopub.execute_input":"2025-11-28T11:46:07.325226Z","iopub.status.idle":"2025-11-28T11:46:07.336660Z","shell.execute_reply.started":"2025-11-28T11:46:07.325209Z","shell.execute_reply":"2025-11-28T11:46:07.336002Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# LOAD METADATA","metadata":{}},{"cell_type":"markdown","source":"__________________","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(TRAIN_CSV)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:07.338540Z","iopub.execute_input":"2025-11-28T11:46:07.339344Z","iopub.status.idle":"2025-11-28T11:46:07.377222Z","shell.execute_reply.started":"2025-11-28T11:46:07.339318Z","shell.execute_reply":"2025-11-28T11:46:07.376718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open(LABEL_JSON, \"r\") as f:\n    label_map = json.load(f)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:07.378024Z","iopub.execute_input":"2025-11-28T11:46:07.378321Z","iopub.status.idle":"2025-11-28T11:46:07.384907Z","shell.execute_reply.started":"2025-11-28T11:46:07.378302Z","shell.execute_reply":"2025-11-28T11:46:07.384485Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:07.385743Z","iopub.execute_input":"2025-11-28T11:46:07.385994Z","iopub.status.idle":"2025-11-28T11:46:07.423178Z","shell.execute_reply.started":"2025-11-28T11:46:07.385977Z","shell.execute_reply":"2025-11-28T11:46:07.422555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df[[\"label\"]].value_counts()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:07.424106Z","iopub.execute_input":"2025-11-28T11:46:07.424476Z","iopub.status.idle":"2025-11-28T11:46:07.437132Z","shell.execute_reply.started":"2025-11-28T11:46:07.424457Z","shell.execute_reply":"2025-11-28T11:46:07.436628Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ANALYTICAL ","metadata":{}},{"cell_type":"code","source":"# LABEL DISTRIBUTION \nplt.figure(figsize=(9,5))\nsns.countplot(x=train_df[\"label\"], palette=\"viridis\")\nplt.title(\"Label Distribution Across Training Set\")\nplt.xlabel(\"Label ID\")\nplt.ylabel(\"Number of Samples\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:07.438026Z","iopub.execute_input":"2025-11-28T11:46:07.438378Z","iopub.status.idle":"2025-11-28T11:46:07.720096Z","shell.execute_reply.started":"2025-11-28T11:46:07.438343Z","shell.execute_reply":"2025-11-28T11:46:07.719463Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Percentage distribution\nlabel_counts = train_df[\"label\"].value_counts().sort_index()\nlabel_percent = (label_counts / len(train_df)) * 100","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:07.721015Z","iopub.execute_input":"2025-11-28T11:46:07.721399Z","iopub.status.idle":"2025-11-28T11:46:07.731693Z","shell.execute_reply.started":"2025-11-28T11:46:07.721371Z","shell.execute_reply":"2025-11-28T11:46:07.731090Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(9,5))\nsns.barplot(x=label_percent.index, y=label_percent.values, palette=\"mako\")\nplt.title(\"Percentage Distribution (%)\")\nplt.xlabel(\"Label ID\")\nplt.ylabel(\"Percentage of Dataset (%)\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:07.732584Z","iopub.execute_input":"2025-11-28T11:46:07.732848Z","iopub.status.idle":"2025-11-28T11:46:07.944594Z","shell.execute_reply.started":"2025-11-28T11:46:07.732829Z","shell.execute_reply":"2025-11-28T11:46:07.944022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#  summary\nlabel_summary = pd.DataFrame({\n    \"Class ID\": label_counts.index,\n    \"Class Name\": [label_map[str(i)] for i in label_counts.index],\n    \"Count\": label_counts.values,\n    \"Percentage\": label_percent.values\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:07.945464Z","iopub.execute_input":"2025-11-28T11:46:07.945748Z","iopub.status.idle":"2025-11-28T11:46:07.950294Z","shell.execute_reply.started":"2025-11-28T11:46:07.945724Z","shell.execute_reply":"2025-11-28T11:46:07.949677Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Class-Level Statistical Summary:\")\nlabel_summary","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:07.951219Z","iopub.execute_input":"2025-11-28T11:46:07.951687Z","iopub.status.idle":"2025-11-28T11:46:07.971367Z","shell.execute_reply.started":"2025-11-28T11:46:07.951668Z","shell.execute_reply":"2025-11-28T11:46:07.970860Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## CLASS-BALANCE EVALUATION METRIC","metadata":{}},{"cell_type":"code","source":"max_class = label_counts.max()\nmin_class = label_counts.min()\nimbalance_ratio = max_class / min_class","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:07.972422Z","iopub.execute_input":"2025-11-28T11:46:07.972737Z","iopub.status.idle":"2025-11-28T11:46:07.976615Z","shell.execute_reply.started":"2025-11-28T11:46:07.972713Z","shell.execute_reply":"2025-11-28T11:46:07.976168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nprint(\"Class Balance Evaluation:\")\nprint(\"Maximum class count:\", max_class)\nprint(\"Minimum class count:\", min_class)\nprint(\"Imbalance ratio (max/min):\", round(imbalance_ratio, 3))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:07.977418Z","iopub.execute_input":"2025-11-28T11:46:07.977683Z","iopub.status.idle":"2025-11-28T11:46:07.992892Z","shell.execute_reply.started":"2025-11-28T11:46:07.977660Z","shell.execute_reply":"2025-11-28T11:46:07.992275Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## IMAGE RESOLUTION ANALYSIS","metadata":{"execution":{"iopub.status.busy":"2025-11-23T14:31:53.258326Z","iopub.execute_input":"2025-11-23T14:31:53.258615Z","iopub.status.idle":"2025-11-23T14:31:53.273204Z","shell.execute_reply.started":"2025-11-23T14:31:53.258594Z","shell.execute_reply":"2025-11-23T14:31:53.272645Z"}}},{"cell_type":"code","source":"def get_resolution(path):\n    img = cv2.imread(path)\n    if img is None:\n        return None\n    return img.shape[1], img.shape[0]  # width, height","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:07.993842Z","iopub.execute_input":"2025-11-28T11:46:07.994110Z","iopub.status.idle":"2025-11-28T11:46:08.010291Z","shell.execute_reply.started":"2025-11-28T11:46:07.994085Z","shell.execute_reply":"2025-11-28T11:46:08.009692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_list = train_df.sample(500, random_state=42).image_id.values\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:08.011034Z","iopub.execute_input":"2025-11-28T11:46:08.011589Z","iopub.status.idle":"2025-11-28T11:46:08.027952Z","shell.execute_reply.started":"2025-11-28T11:46:08.011562Z","shell.execute_reply":"2025-11-28T11:46:08.027448Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nresolutions = []\nfor img_id in sample_list:\n    img_path = os.path.join(TRAIN_IMG_DIR, img_id)\n    res = get_resolution(img_path)\n    if res:\n        resolutions.append(res)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:08.028790Z","iopub.execute_input":"2025-11-28T11:46:08.028984Z","iopub.status.idle":"2025-11-28T11:46:14.049938Z","shell.execute_reply.started":"2025-11-28T11:46:08.028969Z","shell.execute_reply":"2025-11-28T11:46:14.049434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"res_df = pd.DataFrame(resolutions, columns=[\"Width\", \"Height\"])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:14.050732Z","iopub.execute_input":"2025-11-28T11:46:14.051000Z","iopub.status.idle":"2025-11-28T11:46:14.055663Z","shell.execute_reply.started":"2025-11-28T11:46:14.050937Z","shell.execute_reply":"2025-11-28T11:46:14.055223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"res_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:14.056429Z","iopub.execute_input":"2025-11-28T11:46:14.056658Z","iopub.status.idle":"2025-11-28T11:46:14.075011Z","shell.execute_reply.started":"2025-11-28T11:46:14.056631Z","shell.execute_reply":"2025-11-28T11:46:14.074363Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## BRIGHTNESS AND CONTRAST ANALYSIS ","metadata":{}},{"cell_type":"code","source":"def brightness(img):\n    hsv = cv2.cvtColor(img, cv2.COLOR_BGR2HSV)\n    return hsv[:,:,2].mean()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:14.075889Z","iopub.execute_input":"2025-11-28T11:46:14.076132Z","iopub.status.idle":"2025-11-28T11:46:14.089421Z","shell.execute_reply.started":"2025-11-28T11:46:14.076098Z","shell.execute_reply":"2025-11-28T11:46:14.088870Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def contrast(img):\n    return img.std()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:14.095059Z","iopub.execute_input":"2025-11-28T11:46:14.095377Z","iopub.status.idle":"2025-11-28T11:46:14.106618Z","shell.execute_reply.started":"2025-11-28T11:46:14.095357Z","shell.execute_reply":"2025-11-28T11:46:14.106049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"brightness_vals = []\ncontrast_vals = []\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:14.107333Z","iopub.execute_input":"2025-11-28T11:46:14.107522Z","iopub.status.idle":"2025-11-28T11:46:14.121038Z","shell.execute_reply.started":"2025-11-28T11:46:14.107500Z","shell.execute_reply":"2025-11-28T11:46:14.120528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for img_id in sample_list[:200]:   # reduce compute cost\n    img_path = os.path.join(TRAIN_IMG_DIR, img_id)\n    im = cv2.imread(img_path)\n    if im is not None:\n        brightness_vals.append(brightness(im))\n        contrast_vals.append(contrast(im))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:14.121979Z","iopub.execute_input":"2025-11-28T11:46:14.122270Z","iopub.status.idle":"2025-11-28T11:46:16.557334Z","shell.execute_reply.started":"2025-11-28T11:46:14.122247Z","shell.execute_reply":"2025-11-28T11:46:16.556675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(7,5))\nsns.histplot(brightness_vals, bins=30, kde=True, color=\"orange\")\nplt.title(\"Brightness Distribution (Sampled Images)\")\nplt.xlabel(\"Brightness Level\")\nplt.ylabel(\"Frequency\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:16.558463Z","iopub.execute_input":"2025-11-28T11:46:16.558780Z","iopub.status.idle":"2025-11-28T11:46:16.912975Z","shell.execute_reply.started":"2025-11-28T11:46:16.558754Z","shell.execute_reply":"2025-11-28T11:46:16.912342Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(7,5))\nsns.histplot(contrast_vals, bins=30, kde=True, color=\"purple\")\nplt.title(\"Contrast Distribution (Sampled Images)\")\nplt.xlabel(\"Contrast Level\")\nplt.ylabel(\"Frequency\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:16.913750Z","iopub.execute_input":"2025-11-28T11:46:16.913942Z","iopub.status.idle":"2025-11-28T11:46:17.220829Z","shell.execute_reply.started":"2025-11-28T11:46:16.913927Z","shell.execute_reply":"2025-11-28T11:46:17.220223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Brightness Statistics:\\n\")\ndisplay( pd.Series(brightness_vals).describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:17.221698Z","iopub.execute_input":"2025-11-28T11:46:17.221948Z","iopub.status.idle":"2025-11-28T11:46:17.231900Z","shell.execute_reply.started":"2025-11-28T11:46:17.221920Z","shell.execute_reply":"2025-11-28T11:46:17.231280Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Contrast Statistics:\\n\")\ndisplay(pd.Series(contrast_vals).describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:17.232931Z","iopub.execute_input":"2025-11-28T11:46:17.233626Z","iopub.status.idle":"2025-11-28T11:46:17.251020Z","shell.execute_reply.started":"2025-11-28T11:46:17.233586Z","shell.execute_reply":"2025-11-28T11:46:17.250583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"****--\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:17.251688Z","iopub.execute_input":"2025-11-28T11:46:17.251861Z","iopub.status.idle":"2025-11-28T11:46:17.263078Z","shell.execute_reply.started":"2025-11-28T11:46:17.251847Z","shell.execute_reply":"2025-11-28T11:46:17.262435Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# VISUAL SAMPLE ANALYSIS","metadata":{}},{"cell_type":"code","source":"# # SHOW 3 SAMPLES PER CLASS\n\n\nplt.figure(figsize=(12, 18))\n\nunique_labels = sorted(train_df[\"label\"].astype(int).unique())\n\nfor row_idx, lbl in enumerate(unique_labels):\n    # Sample 3 images from this class\n    subset = train_df[train_df[\"label\"].astype(int) == lbl].sample(3, random_state=42)\n\n    for col_idx, img_id in enumerate(subset.image_id.values):\n        img_path = os.path.join(TRAIN_IMG_DIR, img_id)\n        img = plt.imread(img_path)\n\n        plt.subplot(len(unique_labels), 3, row_idx * 3 + col_idx + 1)\n        plt.imshow(img)\n        plt.axis(\"off\")\n\n        # Title for the FIRST column of each row\n        if col_idx == 0:\n            plt.title(f\"Class {lbl}: {label_map[str(lbl)]}\", fontsize=10)\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:17.264128Z","iopub.execute_input":"2025-11-28T11:46:17.264800Z","iopub.status.idle":"2025-11-28T11:46:21.186653Z","shell.execute_reply.started":"2025-11-28T11:46:17.264781Z","shell.execute_reply":"2025-11-28T11:46:21.185588Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  COLOR DISTRIBUTION ANALYSIS (RGB HISTOGRAMS)","metadata":{}},{"cell_type":"code","source":"def rgb_histogram(path):\n    img = cv2.imread(path)\n    if img is None:\n        return None\n    colors = (\"b\",\"g\",\"r\")\n    hist_data = {}\n    for i,col in enumerate(colors):\n        hist = cv2.calcHist([img],[i],None,[256],[0,256])\n        hist_data[col] = hist.flatten()\n    return hist_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:21.187553Z","iopub.execute_input":"2025-11-28T11:46:21.187781Z","iopub.status.idle":"2025-11-28T11:46:21.192771Z","shell.execute_reply.started":"2025-11-28T11:46:21.187766Z","shell.execute_reply":"2025-11-28T11:46:21.192055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img_sample = os.path.join(TRAIN_IMG_DIR, train_df.sample(1).image_id.values[0])\nhist = rgb_histogram(img_sample)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:21.194009Z","iopub.execute_input":"2025-11-28T11:46:21.194356Z","iopub.status.idle":"2025-11-28T11:46:21.224095Z","shell.execute_reply.started":"2025-11-28T11:46:21.194329Z","shell.execute_reply":"2025-11-28T11:46:21.223516Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(8,5))\nplt.plot(hist[\"r\"], color=\"red\")\nplt.plot(hist[\"g\"], color=\"green\")\nplt.plot(hist[\"b\"], color=\"blue\")\nplt.title(\"RGB Histogram Example\")\nplt.xlabel(\"Intensity\")\nplt.ylabel(\"Frequency\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:21.224859Z","iopub.execute_input":"2025-11-28T11:46:21.225066Z","iopub.status.idle":"2025-11-28T11:46:21.445449Z","shell.execute_reply.started":"2025-11-28T11:46:21.225050Z","shell.execute_reply":"2025-11-28T11:46:21.444862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDGE-DETECTION QUALITY CHECK (CANNY)\n","metadata":{}},{"cell_type":"code","source":"def edge_detection_visual(path):\n    img = cv2.imread(path, 0)\n    edges = cv2.Canny(img, 100, 200)\n    plt.figure(figsize=(6,4))\n    plt.imshow(edges, cmap=\"gray\")\n    plt.title(\"Edge Detection Visualization\")\n    plt.axis(\"off\")\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:21.446302Z","iopub.execute_input":"2025-11-28T11:46:21.446498Z","iopub.status.idle":"2025-11-28T11:46:21.450568Z","shell.execute_reply.started":"2025-11-28T11:46:21.446483Z","shell.execute_reply":"2025-11-28T11:46:21.450056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"**\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:21.451573Z","iopub.execute_input":"2025-11-28T11:46:21.452089Z","iopub.status.idle":"2025-11-28T11:46:21.470452Z","shell.execute_reply.started":"2025-11-28T11:46:21.452063Z","shell.execute_reply":"2025-11-28T11:46:21.469832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"edge_detection_visual(img_sample)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:21.471214Z","iopub.execute_input":"2025-11-28T11:46:21.471489Z","iopub.status.idle":"2025-11-28T11:46:21.667779Z","shell.execute_reply.started":"2025-11-28T11:46:21.471473Z","shell.execute_reply":"2025-11-28T11:46:21.667219Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## ROTATION, FLIP, AND NOISE VISUALIZATION \n","metadata":{}},{"cell_type":"code","source":"\nimg_id = train_df.sample(1).image_id.values[0]\nimg_path = os.path.join(TRAIN_IMG_DIR, img_id)\nimg = plt.imread(img_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:21.668578Z","iopub.execute_input":"2025-11-28T11:46:21.668876Z","iopub.status.idle":"2025-11-28T11:46:21.683907Z","shell.execute_reply.started":"2025-11-28T11:46:21.668856Z","shell.execute_reply":"2025-11-28T11:46:21.683478Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12,6))\n\nplt.subplot(1,4,1)\nplt.imshow(img)\nplt.title(\"Original\")\nplt.axis(\"off\")\n\nplt.subplot(1,4,2)\nplt.imshow(np.rot90(img))\nplt.title(\"Rotation Example\")\nplt.axis(\"off\")\n\nplt.subplot(1,4,3)\nplt.imshow(np.fliplr(img))\nplt.title(\"Horizontal Flip\")\nplt.axis(\"off\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:21.684692Z","iopub.execute_input":"2025-11-28T11:46:21.684922Z","iopub.status.idle":"2025-11-28T11:46:22.325759Z","shell.execute_reply.started":"2025-11-28T11:46:21.684896Z","shell.execute_reply":"2025-11-28T11:46:22.325019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"noise = img + np.random.normal(0, 10, img.shape)\nplt.subplot(1,4,4)\nplt.imshow(np.clip(noise, 0, 255).astype(np.uint8))\nplt.title(\"Noise Injection\")\nplt.axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:22.326587Z","iopub.execute_input":"2025-11-28T11:46:22.326836Z","iopub.status.idle":"2025-11-28T11:46:22.533851Z","shell.execute_reply.started":"2025-11-28T11:46:22.326812Z","shell.execute_reply":"2025-11-28T11:46:22.533403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Simple Leaf Segmentation Masking Green Regions","metadata":{}},{"cell_type":"code","source":"def green_segmentation(img):\n    \"\"\"Simple HSV green-mask segmentation to isolate leaf region\"\"\"\n    hsv = cv2.cvtColor(img, cv2.COLOR_BGR2HSV)\n    \n    # Green range in HSV (tuned for cassava leaves)\n    lower = np.array([25, 40, 40])\n    upper = np.array([85, 255, 255])\n    \n    mask = cv2.inRange(hsv, lower, upper)\n    segmented = cv2.bitwise_and(img, img, mask=mask)\n    return segmented","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:22.534581Z","iopub.execute_input":"2025-11-28T11:46:22.534875Z","iopub.status.idle":"2025-11-28T11:46:22.538579Z","shell.execute_reply.started":"2025-11-28T11:46:22.534859Z","shell.execute_reply":"2025-11-28T11:46:22.538012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Show example of segmentation\nexample_id = train_df.sample(1).image_id.values[0]\nexample_path = os.path.join(TRAIN_IMG_DIR, example_id)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:22.539295Z","iopub.execute_input":"2025-11-28T11:46:22.539576Z","iopub.status.idle":"2025-11-28T11:46:22.555103Z","shell.execute_reply.started":"2025-11-28T11:46:22.539554Z","shell.execute_reply":"2025-11-28T11:46:22.554608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img_original = cv2.imread(example_path)\nimg_segmented = green_segmentation(img_original)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:22.555881Z","iopub.execute_input":"2025-11-28T11:46:22.556284Z","iopub.status.idle":"2025-11-28T11:46:22.594764Z","shell.execute_reply.started":"2025-11-28T11:46:22.556261Z","shell.execute_reply":"2025-11-28T11:46:22.594258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10,4))\nplt.subplot(1,2,1)\nplt.imshow(cv2.cvtColor(img_original, cv2.COLOR_BGR2RGB))\nplt.title(\"Original Image\")\nplt.axis(\"off\")\n\nplt.subplot(1,2,2)\nplt.imshow(cv2.cvtColor(img_segmented, cv2.COLOR_BGR2RGB))\nplt.title(\"Leaf Segmentation (Green Mask)\")\nplt.axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:22.595695Z","iopub.execute_input":"2025-11-28T11:46:22.595948Z","iopub.status.idle":"2025-11-28T11:46:23.103518Z","shell.execute_reply.started":"2025-11-28T11:46:22.595929Z","shell.execute_reply":"2025-11-28T11:46:23.102908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#  STRATIFIED TRAIN/VALIDATION SPLIT ...","metadata":{}},{"cell_type":"code","source":"full_df = train_df.copy()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:23.104334Z","iopub.execute_input":"2025-11-28T11:46:23.104544Z","iopub.status.idle":"2025-11-28T11:46:23.108130Z","shell.execute_reply.started":"2025-11-28T11:46:23.104528Z","shell.execute_reply":"2025-11-28T11:46:23.107520Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df, val_df = train_test_split(\n    full_df,\n    test_size=0.2,\n    stratify=full_df[\"label\"],\n    random_state=42\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:23.109026Z","iopub.execute_input":"2025-11-28T11:46:23.109288Z","iopub.status.idle":"2025-11-28T11:46:23.134282Z","shell.execute_reply.started":"2025-11-28T11:46:23.109270Z","shell.execute_reply":"2025-11-28T11:46:23.133815Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#  CLASS WEIGHTS (to reduce dataset imbalance)","metadata":{}},{"cell_type":"markdown","source":"-------------","metadata":{}},{"cell_type":"code","source":"classes = np.unique(train_df[\"label\"].astype(int))\nclass_weights_raw = compute_class_weight(\n    class_weight=\"balanced\",\n    classes=classes,\n    y=train_df[\"label\"].astype(int)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:23.135967Z","iopub.execute_input":"2025-11-28T11:46:23.136264Z","iopub.status.idle":"2025-11-28T11:46:23.143669Z","shell.execute_reply.started":"2025-11-28T11:46:23.136246Z","shell.execute_reply":"2025-11-28T11:46:23.143005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_weights = dict(zip(classes, class_weights_raw))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:23.144628Z","iopub.execute_input":"2025-11-28T11:46:23.144912Z","iopub.status.idle":"2025-11-28T11:46:23.154379Z","shell.execute_reply.started":"2025-11-28T11:46:23.144892Z","shell.execute_reply":"2025-11-28T11:46:23.153853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Class weights:\", class_weights)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:23.155222Z","iopub.execute_input":"2025-11-28T11:46:23.155569Z","iopub.status.idle":"2025-11-28T11:46:23.169135Z","shell.execute_reply.started":"2025-11-28T11:46:23.155550Z","shell.execute_reply":"2025-11-28T11:46:23.168570Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train vs Validation Class Distribution","metadata":{}},{"cell_type":"markdown","source":"_____________","metadata":{}},{"cell_type":"code","source":"print(\"Train size:\", len(train_df))\nprint(\"Validation size:\", len(val_df))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:23.169954Z","iopub.execute_input":"2025-11-28T11:46:23.170183Z","iopub.status.idle":"2025-11-28T11:46:23.183285Z","shell.execute_reply.started":"2025-11-28T11:46:23.170144Z","shell.execute_reply":"2025-11-28T11:46:23.182707Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dist = train_df[\"label\"].value_counts().sort_index()\nval_dist = val_df[\"label\"].value_counts().sort_index()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:23.184067Z","iopub.execute_input":"2025-11-28T11:46:23.184399Z","iopub.status.idle":"2025-11-28T11:46:23.202917Z","shell.execute_reply.started":"2025-11-28T11:46:23.184378Z","shell.execute_reply":"2025-11-28T11:46:23.202292Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dist_df = pd.DataFrame({\n    \"Class ID\": train_dist.index,\n    \"Disease\": [label_map[str(i)] for i in train_dist.index],\n    \"Train Count\": train_dist.values,\n    \"Val Count\": val_dist.values,\n    \"Train %\": (train_dist / len(train_df) * 100).values,\n    \"Val %\": (val_dist / len(val_df) * 100).values\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:23.203704Z","iopub.execute_input":"2025-11-28T11:46:23.203910Z","iopub.status.idle":"2025-11-28T11:46:23.219465Z","shell.execute_reply.started":"2025-11-28T11:46:23.203894Z","shell.execute_reply":"2025-11-28T11:46:23.218879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Class distribution in Train vs Validation:\")\ndisplay(dist_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:23.220498Z","iopub.execute_input":"2025-11-28T11:46:23.220796Z","iopub.status.idle":"2025-11-28T11:46:23.240881Z","shell.execute_reply.started":"2025-11-28T11:46:23.220774Z","shell.execute_reply":"2025-11-28T11:46:23.240346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\nbar_width = 0.35\nidx = np.arange(len(train_dist))\n\nplt.bar(idx - bar_width/2, train_dist.values, width=bar_width, label=\"Train\")\nplt.bar(idx + bar_width/2, val_dist.values, width=bar_width, label=\"Validation\")\n\nplt.xticks(idx, train_dist.index)\nplt.xlabel(\"Class ID\")\nplt.ylabel(\"Sample Count\")\nplt.title(\"Class Distribution: Train vs Validation\")\nplt.legend()\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:23.241926Z","iopub.execute_input":"2025-11-28T11:46:23.242248Z","iopub.status.idle":"2025-11-28T11:46:23.513038Z","shell.execute_reply.started":"2025-11-28T11:46:23.242225Z","shell.execute_reply":"2025-11-28T11:46:23.512460Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#  DATA GENERATORS ","metadata":{}},{"cell_type":"code","source":"\nIMG_SIZE = 224\nBATCH_SIZE = 32\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:23.513875Z","iopub.execute_input":"2025-11-28T11:46:23.514127Z","iopub.status.idle":"2025-11-28T11:46:23.517244Z","shell.execute_reply.started":"2025-11-28T11:46:23.514110Z","shell.execute_reply":"2025-11-28T11:46:23.516694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert labels to strings \n# requires labels to be string when using sparse mode\ntrain_df[\"label\"] = train_df[\"label\"].astype(str)\nval_df[\"label\"] = val_df[\"label\"].astype(str)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:23.518171Z","iopub.execute_input":"2025-11-28T11:46:23.518461Z","iopub.status.idle":"2025-11-28T11:46:23.538654Z","shell.execute_reply.started":"2025-11-28T11:46:23.518433Z","shell.execute_reply":"2025-11-28T11:46:23.538086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_gen = ImageDataGenerator(\n    preprocessing_function=tf.keras.applications.efficientnet.preprocess_input,\n    rotation_range=20,\n    zoom_range=0.20,\n    shear_range=0.15,\n    horizontal_flip=True,\n    width_shift_range=0.05,\n    height_shift_range=0.05,\n\n    brightness_range=[0.75, 1.25]\n\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:23.539579Z","iopub.execute_input":"2025-11-28T11:46:23.539781Z","iopub.status.idle":"2025-11-28T11:46:23.553209Z","shell.execute_reply.started":"2025-11-28T11:46:23.539766Z","shell.execute_reply":"2025-11-28T11:46:23.552621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_gen = ImageDataGenerator(\n    preprocessing_function=tf.keras.applications.efficientnet.preprocess_input\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:23.554118Z","iopub.execute_input":"2025-11-28T11:46:23.554504Z","iopub.status.idle":"2025-11-28T11:46:23.568732Z","shell.execute_reply.started":"2025-11-28T11:46:23.554459Z","shell.execute_reply":"2025-11-28T11:46:23.568211Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_flow = train_gen.flow_from_dataframe(\n    dataframe=train_df,\n    directory=TRAIN_IMG_DIR,\n    x_col=\"image_id\",\n    y_col=\"label\",\n    class_mode=\"sparse\",       \n    target_size=(IMG_SIZE, IMG_SIZE),\n    batch_size=BATCH_SIZE,\n    shuffle=True\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:23.569625Z","iopub.execute_input":"2025-11-28T11:46:23.569832Z","iopub.status.idle":"2025-11-28T11:46:53.564068Z","shell.execute_reply.started":"2025-11-28T11:46:23.569815Z","shell.execute_reply":"2025-11-28T11:46:53.563655Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_flow = val_gen.flow_from_dataframe(\n    dataframe=val_df,\n    directory=TRAIN_IMG_DIR,\n    x_col=\"image_id\",\n    y_col=\"label\",\n    class_mode=\"sparse\",      \n    target_size=(IMG_SIZE, IMG_SIZE),\n    batch_size=BATCH_SIZE,\n    shuffle=False\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:46:53.564806Z","iopub.execute_input":"2025-11-28T11:46:53.564995Z","iopub.status.idle":"2025-11-28T11:47:01.599676Z","shell.execute_reply.started":"2025-11-28T11:46:53.564980Z","shell.execute_reply":"2025-11-28T11:47:01.599136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Modeling ","metadata":{}},{"cell_type":"code","source":"def build_model(img_size):\n    inp = Input(shape=(img_size, img_size, 3))\n    base = EfficientNetB0(include_top=False, weights=\"imagenet\", input_tensor=inp)\n    base.trainable = False\n\n    x = GlobalAveragePooling2D()(base.output)\n    x = LayerNormalization()(x)\n    x = Dropout(0.45)(x)\n    x = Dense(512, activation=\"swish\",kernel_regularizer=tf.keras.regularizers.l2(1e-5))(x)\n    x = LayerNormalization()(x)\n    x = Dropout(0.45)(x)\n\n    out = Dense(5, activation=\"softmax\")(x)\n    model = Model(inputs=inp, outputs=out)\n    return model\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T11:47:01.600411Z","iopub.execute_input":"2025-11-28T11:47:01.600605Z","iopub.status.idle":"2025-11-28T11:47:01.606824Z","shell.execute_reply.started":"2025-11-28T11:47:01.600589Z","shell.execute_reply":"2025-11-28T11:47:01.606307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = build_model(IMG_SIZE)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-11-28T13:27:04.888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lr_schedule = tf.keras.optimizers.schedules.CosineDecay(\n    initial_learning_rate=5e-4,\n    decay_steps=1000\n)\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-11-28T13:27:04.888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nmodel.compile(\n    optimizer = Adam(lr_schedule),\n    loss=sparse_focal_loss(gamma=2.0), \n    metrics=[\"accuracy\"]\n)\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-11-28T13:27:04.888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.summary()\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-11-28T13:27:04.888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# CALLBACKS\n","metadata":{}},{"cell_type":"code","source":"callbacks_stage1 = [\n    EarlyStopping(monitor=\"val_loss\", patience=5, restore_best_weights=True),\n    ReduceLROnPlateau(monitor=\"val_loss\", factor=0.3, patience=2),\n    ModelCheckpoint(\"stage1_best.h5\", save_best_only=True),\n    CSVLogger(\"stage1_log.csv\")\n]\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-11-28T13:27:04.888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# TRAINING ","metadata":{}},{"cell_type":"markdown","source":"## Frozen EfficientNetB0","metadata":{}},{"cell_type":"code","source":"history_1 = model.fit(\n    train_flow,\n    validation_data=val_flow,\n    epochs=70,\n    callbacks=callbacks_stage1,\n    class_weight=class_weights     \n)","metadata":{"trusted":true,"execution":{"iopub.status.idle":"2025-11-28T12:16:08.627947Z","shell.execute_reply.started":"2025-11-28T11:47:04.672289Z","shell.execute_reply":"2025-11-28T12:16:08.627452Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## UNFREEZE TOP HALF OF BACKBONE","metadata":{}},{"cell_type":"code","source":"for layer in model.layers:\n    layer.trainable = True\n\nfor layer in model.layers[:200]:\n    layer.trainable = False\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:16:08.629177Z","iopub.execute_input":"2025-11-28T12:16:08.629512Z","iopub.status.idle":"2025-11-28T12:16:08.637848Z","shell.execute_reply.started":"2025-11-28T12:16:08.629484Z","shell.execute_reply":"2025-11-28T12:16:08.637365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(\n    optimizer = Adam(lr_schedule),\n    loss=sparse_focal_loss(gamma=2.0),   # \n    metrics=[\"accuracy\"]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:16:08.638779Z","iopub.execute_input":"2025-11-28T12:16:08.639023Z","iopub.status.idle":"2025-11-28T12:16:08.676069Z","shell.execute_reply.started":"2025-11-28T12:16:08.639006Z","shell.execute_reply":"2025-11-28T12:16:08.675565Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"callbacks_stage2 = [\n    EarlyStopping(monitor=\"val_loss\", patience=5, restore_best_weights=True),\n    ReduceLROnPlateau(monitor=\"val_loss\", factor=0.3, patience=2),\n    ModelCheckpoint(\"stage2_best.h5\", save_best_only=True),\n    CSVLogger(\"stage2_log.csv\")\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:16:08.677865Z","iopub.execute_input":"2025-11-28T12:16:08.678149Z","iopub.status.idle":"2025-11-28T12:16:08.683237Z","shell.execute_reply.started":"2025-11-28T12:16:08.678131Z","shell.execute_reply":"2025-11-28T12:16:08.682613Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#from sklearn.utils.class_weight import compute_class_weight\n\n# Proper class weights\n#classes = np.unique(train_df[\"label\"].astype(int))\n#class_weights_raw = compute_class_weight(\"balanced\", classes=classes, y=train_df[\"label\"].astype(int))\n#class_weights = dict(zip(classes, class_weights_raw))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:16:08.683914Z","iopub.execute_input":"2025-11-28T12:16:08.684128Z","iopub.status.idle":"2025-11-28T12:16:08.698551Z","shell.execute_reply.started":"2025-11-28T12:16:08.684112Z","shell.execute_reply":"2025-11-28T12:16:08.697567Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history_2 = model.fit(\n    train_flow,\n    validation_data=val_flow,\n    class_weight=class_weights,\n    epochs=70,\n    callbacks=callbacks_stage2\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:16:08.699653Z","iopub.execute_input":"2025-11-28T12:16:08.699935Z","iopub.status.idle":"2025-11-28T12:45:55.660033Z","shell.execute_reply.started":"2025-11-28T12:16:08.699911Z","shell.execute_reply":"2025-11-28T12:45:55.659436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PLOTTING FUNCTIONS (ACCURACY + LOSS)","metadata":{}},{"cell_type":"code","source":"def plot_training(h1, h2):\n    acc = h1.history[\"accuracy\"] + h2.history[\"accuracy\"]\n    val_acc = h1.history[\"val_accuracy\"] + h2.history[\"val_accuracy\"]\n    loss = h1.history[\"loss\"] + h2.history[\"loss\"]\n    val_loss = h1.history[\"val_loss\"] + h2.history[\"val_loss\"]\n\n    epochs = range(1, len(acc) + 1)\n\n    plt.figure(figsize=(14,5))\n    plt.subplot(1,2,1)\n    plt.plot(epochs, acc, label=\"Train Accuracy\")\n    plt.plot(epochs, val_acc, label=\"Validation Accuracy\")\n    plt.legend()\n    plt.title(\"Accuracy\")\n\n    plt.subplot(1,2,2)\n    plt.plot(epochs, loss, label=\"Train Loss\")\n    plt.plot(epochs, val_loss, label=\"Validation Loss\")\n    plt.legend()\n    plt.title(\"Loss\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:45:55.661390Z","iopub.execute_input":"2025-11-28T12:45:55.661671Z","iopub.status.idle":"2025-11-28T12:45:55.667057Z","shell.execute_reply.started":"2025-11-28T12:45:55.661646Z","shell.execute_reply":"2025-11-28T12:45:55.666593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_training(history_1, history_2)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:45:55.667943Z","iopub.execute_input":"2025-11-28T12:45:55.668141Z","iopub.status.idle":"2025-11-28T12:45:56.096215Z","shell.execute_reply.started":"2025-11-28T12:45:55.668126Z","shell.execute_reply":"2025-11-28T12:45:56.095622Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"acc_s1 = history_1.history[\"val_accuracy\"][-1]\nacc_s2 = history_2.history[\"val_accuracy\"][-1]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:45:56.097043Z","iopub.execute_input":"2025-11-28T12:45:56.097449Z","iopub.status.idle":"2025-11-28T12:45:56.100789Z","shell.execute_reply.started":"2025-11-28T12:45:56.097419Z","shell.execute_reply":"2025-11-28T12:45:56.100267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Stage 1 Final Validation Accuracy: {acc_s1:.4f}\")\nprint(f\"Stage 2 Final Validation Accuracy: {acc_s2:.4f}\")\nprint(f\"Improvement: {(acc_s2 - acc_s1)*100:.2f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:45:56.101652Z","iopub.execute_input":"2025-11-28T12:45:56.101966Z","iopub.status.idle":"2025-11-28T12:45:56.115395Z","shell.execute_reply.started":"2025-11-28T12:45:56.101941Z","shell.execute_reply":"2025-11-28T12:45:56.114814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"___\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:45:56.123138Z","iopub.execute_input":"2025-11-28T12:45:56.123360Z","iopub.status.idle":"2025-11-28T12:45:56.130472Z","shell.execute_reply.started":"2025-11-28T12:45:56.123343Z","shell.execute_reply":"2025-11-28T12:45:56.129964Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"loss_s1 = history_1.history[\"val_loss\"][-1]\nloss_s2 = history_2.history[\"val_loss\"][-1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:45:56.131187Z","iopub.execute_input":"2025-11-28T12:45:56.131365Z","iopub.status.idle":"2025-11-28T12:45:56.143860Z","shell.execute_reply.started":"2025-11-28T12:45:56.131352Z","shell.execute_reply":"2025-11-28T12:45:56.143403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Loss Comparison\nprint(f\"Stage 1 Final Validation Loss: {loss_s1:.4f}\")\nprint(f\"Stage 2 Final Validation Loss: {loss_s2:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:45:56.144717Z","iopub.execute_input":"2025-11-28T12:45:56.144954Z","iopub.status.idle":"2025-11-28T12:45:56.158774Z","shell.execute_reply.started":"2025-11-28T12:45:56.144937Z","shell.execute_reply":"2025-11-28T12:45:56.158175Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Stage 1 vs Stage 2 Analysis","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Detailed Training Curves per Stage","metadata":{}},{"cell_type":"code","source":"def plot_stage_history(history, stage_name=\"Stage\"):\n    epochs = range(1, len(history.history[\"accuracy\"]) + 1)\n\n    plt.figure(figsize=(12,4))\n\n    plt.subplot(1,2,1)\n    plt.plot(epochs, history.history[\"accuracy\"], label=\"Train\")\n    plt.plot(epochs, history.history[\"val_accuracy\"], label=\"Validation\")\n    plt.title(f\"{stage_name} Accuracy\")\n    plt.xlabel(\"Epoch\")\n    plt.ylabel(\"Accuracy\")\n    plt.legend()\n\n    plt.subplot(1,2,2)\n    plt.plot(epochs, history.history[\"loss\"], label=\"Train\")\n    plt.plot(epochs, history.history[\"val_loss\"], label=\"Validation\")\n    plt.title(f\"{stage_name} Loss\")\n    plt.xlabel(\"Epoch\")\n    plt.ylabel(\"Loss\")\n    plt.legend()\n\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:45:56.159527Z","iopub.execute_input":"2025-11-28T12:45:56.159781Z","iopub.status.idle":"2025-11-28T12:45:56.176555Z","shell.execute_reply.started":"2025-11-28T12:45:56.159766Z","shell.execute_reply":"2025-11-28T12:45:56.176025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_stage_history(history_1, stage_name=\"Stage 1 (Frozen Backbone)\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:45:56.177272Z","iopub.execute_input":"2025-11-28T12:45:56.177503Z","iopub.status.idle":"2025-11-28T12:45:56.733684Z","shell.execute_reply.started":"2025-11-28T12:45:56.177482Z","shell.execute_reply":"2025-11-28T12:45:56.733018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_stage_history(history_2, stage_name=\"Stage 2 (Fine-tuning)\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:45:56.734550Z","iopub.execute_input":"2025-11-28T12:45:56.734848Z","iopub.status.idle":"2025-11-28T12:45:57.278559Z","shell.execute_reply.started":"2025-11-28T12:45:56.734817Z","shell.execute_reply":"2025-11-28T12:45:57.278014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EVALUATION: CONFUSION MATRIX + CLASSIFICATION REPORT","metadata":{}},{"cell_type":"code","source":"val_flow_eval = val_gen.flow_from_dataframe(\n    val_df,\n    directory=TRAIN_IMG_DIR,\n    x_col=\"image_id\",\n    y_col=\"label\",\n    class_mode=\"sparse\",\n    target_size=(IMG_SIZE, IMG_SIZE),\n    shuffle=False\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:45:57.279383Z","iopub.execute_input":"2025-11-28T12:45:57.279629Z","iopub.status.idle":"2025-11-28T12:45:57.325387Z","shell.execute_reply.started":"2025-11-28T12:45:57.279602Z","shell.execute_reply":"2025-11-28T12:45:57.324891Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_probs = model.predict(val_flow_eval)\ny_pred = pred_probs.argmax(axis=1)\ny_true = val_df[\"label\"].values.astype(int)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:45:57.326008Z","iopub.execute_input":"2025-11-28T12:45:57.326195Z","iopub.status.idle":"2025-11-28T12:46:29.246198Z","shell.execute_reply.started":"2025-11-28T12:45:57.326181Z","shell.execute_reply":"2025-11-28T12:46:29.245598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cm = confusion_matrix(y_true, y_pred)\n\nplt.figure(figsize=(7,6))\nsns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\nplt.title(\"Confusion Matrix\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:46:29.247123Z","iopub.execute_input":"2025-11-28T12:46:29.247362Z","iopub.status.idle":"2025-11-28T12:46:29.534968Z","shell.execute_reply.started":"2025-11-28T12:46:29.247345Z","shell.execute_reply":"2025-11-28T12:46:29.534514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(classification_report(y_true, y_pred, digits=3))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:46:29.535980Z","iopub.execute_input":"2025-11-28T12:46:29.536277Z","iopub.status.idle":"2025-11-28T12:46:29.553530Z","shell.execute_reply.started":"2025-11-28T12:46:29.536251Z","shell.execute_reply":"2025-11-28T12:46:29.552979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bal_acc = balanced_accuracy_score(y_true, y_pred)\nw_acc = np.average(y_true == y_pred, weights=None)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:46:29.554392Z","iopub.execute_input":"2025-11-28T12:46:29.554709Z","iopub.status.idle":"2025-11-28T12:46:29.560216Z","shell.execute_reply.started":"2025-11-28T12:46:29.554690Z","shell.execute_reply":"2025-11-28T12:46:29.559695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# gives you fair measurements despite the imbalance\nprint(f\"Balanced Accuracy: {bal_acc:.4f}\")\n\n\nprint(f\"Weighted Accuracy (macro-weighted): {np.average(y_true == y_pred):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:46:29.561392Z","iopub.execute_input":"2025-11-28T12:46:29.561647Z","iopub.status.idle":"2025-11-28T12:46:29.573554Z","shell.execute_reply.started":"2025-11-28T12:46:29.561630Z","shell.execute_reply":"2025-11-28T12:46:29.572900Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ANALYSIS: Per-Class Precision, Recall, F1","metadata":{}},{"cell_type":"markdown","source":"---------------","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import classification_report\n\n# get classification report \ntarget_names = [f\"{i} - {label_map[str(i)]}\" for i in sorted(np.unique(y_true))]\nreport_dict = classification_report(\n    y_true, y_pred, output_dict=True, target_names=target_names, digits=4\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:46:29.574445Z","iopub.execute_input":"2025-11-28T12:46:29.574719Z","iopub.status.idle":"2025-11-28T12:46:29.599040Z","shell.execute_reply.started":"2025-11-28T12:46:29.574695Z","shell.execute_reply":"2025-11-28T12:46:29.598649Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# convert to DataFrame (only class rows)\nper_class_metrics = pd.DataFrame(report_dict).T\nper_class_metrics = per_class_metrics.iloc[:len(target_names)]  # remove 'accuracy', 'macro avg', ...\nper_class_metrics = per_class_metrics[[\"precision\", \"recall\", \"f1-score\", \"support\"]]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:46:29.599977Z","iopub.execute_input":"2025-11-28T12:46:29.600257Z","iopub.status.idle":"2025-11-28T12:46:29.606298Z","shell.execute_reply.started":"2025-11-28T12:46:29.600239Z","shell.execute_reply":"2025-11-28T12:46:29.605746Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Per-class metrics:\")\ndisplay(per_class_metrics)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:46:29.607252Z","iopub.execute_input":"2025-11-28T12:46:29.607497Z","iopub.status.idle":"2025-11-28T12:46:29.624884Z","shell.execute_reply.started":"2025-11-28T12:46:29.607480Z","shell.execute_reply":"2025-11-28T12:46:29.624330Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot precision / recall / f1\nplt.figure(figsize=(12,5))\nx = np.arange(len(per_class_metrics))\nw = 0.25\n\nplt.bar(x - w, per_class_metrics[\"precision\"], width=w, label=\"Precision\")\nplt.bar(x,       per_class_metrics[\"recall\"],    width=w, label=\"Recall\")\nplt.bar(x + w, per_class_metrics[\"f1-score\"], width=w, label=\"F1-score\")\n\nplt.xticks(x, per_class_metrics.index, rotation=45, ha=\"right\")\nplt.ylim(0, 1.05)\nplt.ylabel(\"Score\")\nplt.title(\"Per-class Precision / Recall / F1\")\nplt.legend()\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:46:29.626055Z","iopub.execute_input":"2025-11-28T12:46:29.626451Z","iopub.status.idle":"2025-11-28T12:46:29.912349Z","shell.execute_reply.started":"2025-11-28T12:46:29.626427Z","shell.execute_reply":"2025-11-28T12:46:29.911731Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Normalized confusion matrix\ncm_norm = cm.astype(\"float\") / cm.sum(axis=1, keepdims=True)\n\nplt.figure(figsize=(7,6))\nsns.heatmap(cm_norm, annot=True, fmt=\".2f\", cmap=\"Blues\")\nplt.title(\"Normalized Confusion Matrix (Row-wise)\")\nplt.xlabel(\"Predicted\")\nplt.ylabel(\"True\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:46:29.913309Z","iopub.execute_input":"2025-11-28T12:46:29.913594Z","iopub.status.idle":"2025-11-28T12:46:30.217010Z","shell.execute_reply.started":"2025-11-28T12:46:29.913570Z","shell.execute_reply":"2025-11-28T12:46:30.216485Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ANALYSIS: Visualizing Misclassified Samples","metadata":{}},{"cell_type":"code","source":"val_df_eval = val_df.reset_index(drop=True)\n\nmis_idx = np.where(y_true != y_pred)[0]\nprint(\"Number of misclassified samples:\", len(mis_idx))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:46:30.217946Z","iopub.execute_input":"2025-11-28T12:46:30.218303Z","iopub.status.idle":"2025-11-28T12:46:30.223326Z","shell.execute_reply.started":"2025-11-28T12:46:30.218274Z","shell.execute_reply":"2025-11-28T12:46:30.222766Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if len(mis_idx) > 0:\n    n_show = min(16, len(mis_idx))\n    chosen = np.random.choice(mis_idx, size=n_show, replace=False)\n\n    plt.figure(figsize=(14,14))\n    for i, idx in enumerate(chosen):\n        row = val_df_eval.iloc[idx]\n        img_path = os.path.join(TRAIN_IMG_DIR, row.image_id)\n        img = plt.imread(img_path)\n\n        true_lbl = int(y_true[idx])\n        pred_lbl = int(y_pred[idx])\n\n        plt.subplot(4,4,i+1)\n        plt.imshow(img)\n        plt.axis(\"off\")\n        plt.title(\n            f\"True: {true_lbl} ({label_map[str(true_lbl)]})\\n\"\n            f\"Pred: {pred_lbl} ({label_map[str(pred_lbl)]})\",\n            fontsize=8\n        )\n\n    plt.tight_layout()\n    plt.show()\nelse:\n    print(\"No misclassified samples to display (perfect accuracy on validation).\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:46:30.224317Z","iopub.execute_input":"2025-11-28T12:46:30.224547Z","iopub.status.idle":"2025-11-28T12:46:34.221802Z","shell.execute_reply.started":"2025-11-28T12:46:30.224522Z","shell.execute_reply":"2025-11-28T12:46:34.220717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#  GRAD-CAM INTERPRETABILITY MODULE ","metadata":{}},{"cell_type":"code","source":"def generate_grad_cam(model, img_path, layer_name=\"top_conv\", alpha=0.4):\n\n    img = cv2.imread(img_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    rgb = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    arr = tf.keras.applications.efficientnet.preprocess_input(\n        np.expand_dims(rgb.astype(\"float32\"), axis=0)\n    )\n\n\n    grad_model = tf.keras.models.Model(\n        [model.inputs],\n        [model.get_layer(layer_name).output, model.output]\n    )\n\n    with tf.GradientTape() as tape:\n        conv_out, preds = grad_model(arr)\n        class_idx = tf.argmax(preds[0])\n        loss = preds[:, class_idx]\n\n    grads = tape.gradient(loss, conv_out)\n    weights = tf.reduce_mean(grads, axis=(0,1,2))\n    cam = np.zeros(conv_out.shape[1:3], dtype=np.float32)\n\n    for i, w in enumerate(weights):\n        cam += w * conv_out[0,:,:,i]\n\n    cam = np.maximum(cam, 0)\n    cam = cv2.resize(cam, (IMG_SIZE, IMG_SIZE))\n    cam = (cam - cam.min()) / (cam.max() + 1e-9)\n    cam = np.uint8(255 * cam)\n\n    heatmap = cv2.applyColorMap(cam, cv2.COLORMAP_JET)\n    heatmap = cv2.cvtColor(heatmap, cv2.COLOR_BGR2RGB)\n\n    overlay = cv2.addWeighted(rgb, 1-alpha, heatmap, alpha, 0)\n\n    plt.figure(figsize=(5,5))\n    plt.imshow(overlay)\n    plt.axis(\"off\")\n    plt.title(\"Grad-CAM\")\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:46:34.223604Z","iopub.execute_input":"2025-11-28T12:46:34.224235Z","iopub.status.idle":"2025-11-28T12:46:34.234752Z","shell.execute_reply.started":"2025-11-28T12:46:34.224198Z","shell.execute_reply":"2025-11-28T12:46:34.234190Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# example \nexample = val_df.sample(1).image_id.values[0]\ngenerate_grad_cam(model, os.path.join(TRAIN_IMG_DIR, example))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:46:34.235647Z","iopub.execute_input":"2025-11-28T12:46:34.235949Z","iopub.status.idle":"2025-11-28T12:46:37.554876Z","shell.execute_reply.started":"2025-11-28T12:46:34.235898Z","shell.execute_reply":"2025-11-28T12:46:37.554368Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#  Grad-CAM for Multiple Validation Samples\n\ndef generate_grad_cam_multi(model, df, num_images=12, last_conv=\"top_conv\", alpha=0.4):\n    \"\"\"\n    Generates Grad-CAM heatmaps for multiple images.\n    df: DataFrame containing image_id and label columns (use validation dataframe)\n    num_images: number of samples to visualize\n    \"\"\"\n\n    sample_df = df.sample(num_images, random_state=42).reset_index(drop=True)\n\n    plt.figure(figsize=(14, 14))\n\n    for i in range(num_images):\n        img_id = sample_df.loc[i, \"image_id\"]\n        true_lbl = sample_df.loc[i, \"label\"]\n\n        img_path = os.path.join(TRAIN_IMG_DIR, img_id)\n\n        # Load and preprocess image\n        img = cv2.imread(img_path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        rgb = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n       # arr = np.expand_dims(rgb.astype(\"float32\") / 255.0, axis=0)\n        arr = tf.keras.applications.efficientnet.preprocess_input(\n        np.expand_dims(rgb.astype(\"float32\"), axis=0)\n)\n\n        # Build Grad-CAM model\n        grad_model = tf.keras.models.Model(\n            [model.inputs],\n            [model.get_layer(last_conv).output, model.output]\n        )\n\n        with tf.GradientTape() as tape:\n            conv_out, preds = grad_model(arr)\n            pred_idx = tf.argmax(preds[0])\n            loss = preds[:, pred_idx]\n\n        grads = tape.gradient(loss, conv_out)\n        weights = tf.reduce_mean(grads, axis=(0, 1, 2))\n        cam = np.zeros(conv_out.shape[1:3], dtype=np.float32)\n\n        for j, w in enumerate(weights):\n            cam += w * conv_out[0, :, :, j]\n\n        cam = np.maximum(cam, 0)\n        cam = cv2.resize(cam, (IMG_SIZE, IMG_SIZE))\n        cam = cam / (cam.max() + 1e-9)\n        cam = np.uint8(255 * cam)\n\n        heatmap = cv2.applyColorMap(cam, cv2.COLORMAP_JET)\n        heatmap = cv2.cvtColor(heatmap, cv2.COLOR_BGR2RGB)\n        overlay = cv2.addWeighted(rgb, 1 - alpha, heatmap, alpha, 0)\n\n        # Plot\n        plt.subplot(4, 3, i + 1)\n        plt.imshow(overlay)\n        pred_lbl = int(pred_idx.numpy())\n        title_str = f\"True: {true_lbl} ({label_map[str(true_lbl)]})\\nPred: {pred_lbl} ({label_map[str(pred_lbl)]})\"\n        plt.title(title_str, fontsize=8)\n        plt.axis(\"off\")\n\n    plt.tight_layout()\n    plt.show()\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:46:37.555658Z","iopub.execute_input":"2025-11-28T12:46:37.555942Z","iopub.status.idle":"2025-11-28T12:46:37.565335Z","shell.execute_reply.started":"2025-11-28T12:46:37.555912Z","shell.execute_reply":"2025-11-28T12:46:37.564818Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Multi-GradCAM\ngenerate_grad_cam_multi(model, val_df, num_images=12, last_conv=\"top_conv\", alpha=0.4)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:46:37.566089Z","iopub.execute_input":"2025-11-28T12:46:37.566361Z","iopub.status.idle":"2025-11-28T12:47:00.481202Z","shell.execute_reply.started":"2025-11-28T12:46:37.566338Z","shell.execute_reply":"2025-11-28T12:47:00.480574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# SAVE  MODEL","metadata":{}},{"cell_type":"code","source":"# FINAL_MODEL_PATH = \"cassava_efficientnetb0_final.keras\"\nmodel.save(\"cassava_final_model.keras\")\nprint(\"is save model \")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:47:00.482104Z","iopub.execute_input":"2025-11-28T12:47:00.482452Z","iopub.status.idle":"2025-11-28T12:47:01.194711Z","shell.execute_reply.started":"2025-11-28T12:47:00.482421Z","shell.execute_reply":"2025-11-28T12:47:01.194049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# T EST PREDICTION &&&  SUBMISSION","metadata":{}},{"cell_type":"code","source":"sample_sub = pd.read_csv(SUBMISSION_TEMPLATE)\ntest_df = pd.DataFrame({\"image_id\": sample_sub[\"image_id\"].values})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:47:01.195341Z","iopub.execute_input":"2025-11-28T12:47:01.195517Z","iopub.status.idle":"2025-11-28T12:47:01.210851Z","shell.execute_reply.started":"2025-11-28T12:47:01.195503Z","shell.execute_reply":"2025-11-28T12:47:01.210350Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_gen = ImageDataGenerator(\n    preprocessing_function=tf.keras.applications.efficientnet.preprocess_input\n)\n\ntest_flow = test_gen.flow_from_dataframe(\n    test_df,\n    directory=TEST_IMG_DIR,\n    x_col=\"image_id\",\n    y_col=None,\n    class_mode=None,\n    target_size=(IMG_SIZE, IMG_SIZE),\n    shuffle=False,\n    batch_size=32\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:47:01.211716Z","iopub.execute_input":"2025-11-28T12:47:01.212031Z","iopub.status.idle":"2025-11-28T12:47:01.222576Z","shell.execute_reply.started":"2025-11-28T12:47:01.211986Z","shell.execute_reply":"2025-11-28T12:47:01.222107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_pred = model.predict(test_flow).argmax(axis=1)\nsubmission = test_df.copy()\nsubmission[\"label\"] = test_pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:47:01.223278Z","iopub.execute_input":"2025-11-28T12:47:01.223461Z","iopub.status.idle":"2025-11-28T12:47:07.835589Z","shell.execute_reply.started":"2025-11-28T12:47:01.223448Z","shell.execute_reply":"2025-11-28T12:47:07.835023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T12:47:07.836489Z","iopub.execute_input":"2025-11-28T12:47:07.836850Z","iopub.status.idle":"2025-11-28T12:47:07.845408Z","shell.execute_reply.started":"2025-11-28T12:47:07.836825Z","shell.execute_reply":"2025-11-28T12:47:07.844887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}