{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"sourceType":"competition"}],"dockerImageVersionId":31154,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ---- Imports ----\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom pathlib import Path\nimport json\nimport pandas as pd\n\nfrom tensorflow import keras\nfrom tensorflow.keras import layers, models, regularizers\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications.efficientnet import EfficientNetB0, preprocess_input\n\nfrom sklearn.metrics import classification_report, confusion_matrix\nimport seaborn as sns  # optional, for nicer confusion matrix plots","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-01T01:14:48.340640Z","iopub.execute_input":"2025-12-01T01:14:48.341165Z","iopub.status.idle":"2025-12-01T01:14:48.345589Z","shell.execute_reply.started":"2025-12-01T01:14:48.341142Z","shell.execute_reply":"2025-12-01T01:14:48.344885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA = Path('/kaggle/input/cassava-leaf-disease-classification')\ntrain_csv = pd.read_csv(DATA/'train.csv')         # columns: image_id, label\nwith open(DATA/'label_num_to_disease_map.json') as f:\n    label_map = json.load(f)\n\ntrain_csv['filepath'] = train_csv['image_id'].apply(lambda x: str(DATA/'train_images'/x))\nnum_classes = train_csv['label'].nunique()\ntrain_csv.head(), label_map, num_classes\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T01:14:48.349098Z","iopub.execute_input":"2025-12-01T01:14:48.349377Z","iopub.status.idle":"2025-12-01T01:14:48.507185Z","shell.execute_reply.started":"2025-12-01T01:14:48.349358Z","shell.execute_reply":"2025-12-01T01:14:48.506548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_csv['label'].value_counts().sort_index().plot(kind='bar')\nplt.title('Class counts'); plt.show()\n\ndef show_samples(df, n=10):\n    sample = df.sample(n)\n    plt.figure(figsize=(12,8))\n    for i,(fp,lab) in enumerate(zip(sample['filepath'], sample['label'])):\n        plt.subplot(2, n//2, i+1); plt.imshow(plt.imread(fp)); plt.axis('off')\n        plt.title(f\"{lab}: {label_map[str(lab)]}\")\n    plt.tight_layout(); plt.show()\n\nshow_samples(train_csv, n=8)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T01:14:48.508320Z","iopub.execute_input":"2025-12-01T01:14:48.508553Z","iopub.status.idle":"2025-12-01T01:14:50.187941Z","shell.execute_reply.started":"2025-12-01T01:14:48.508535Z","shell.execute_reply":"2025-12-01T01:14:50.183881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Create Data Generators\nIMG_SIZE = (150, 200)\nBATCH = 32\nSEED = 2025\n\ndatagen = ImageDataGenerator(rescale=1./255, validation_split=0.2)\n\ntrain_gen = datagen.flow_from_dataframe(\n    train_csv, x_col='filepath', y_col='label',\n    target_size=IMG_SIZE, class_mode='raw',   # ← raw returns y as-is\n    batch_size=BATCH, shuffle=True, seed=SEED, subset='training'\n)\n\nval_gen = datagen.flow_from_dataframe(\n    train_csv, x_col='filepath', y_col='label',\n    target_size=IMG_SIZE, class_mode='raw',\n    batch_size=BATCH, shuffle=False, seed=SEED, subset='validation'\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T01:14:50.189056Z","iopub.execute_input":"2025-12-01T01:14:50.189653Z","iopub.status.idle":"2025-12-01T01:16:34.594409Z","shell.execute_reply.started":"2025-12-01T01:14:50.189602Z","shell.execute_reply":"2025-12-01T01:16:34.593724Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"For my baseline model, I implemented a simple Convolutional Neural Network (CNN) consisting of three convolution–max pooling blocks followed by a fully connected layer and a softmax classifier. This lightweight architecture was designed to establish a benchmark for performance before applying more advanced techniques such as deeper CNNs and transfer learning. The model was trained on images resized to 150×200 and optimized using sparse categorical cross-entropy. It provides a clear reference point for comparing improvements in later models.","metadata":{}},{"cell_type":"markdown","source":"To avoid overfitting and ensure efficient training, I used two Keras callbacks: ModelCheckpoint to save the best model based on validation accuracy, and EarlyStopping with a patience of five epochs to stop training when improvement plateaued. I also enabled restore_best_weights=True, ensuring that the model reverted to its strongest parameter state. This improved both stability and performance compared to simple training without callbacks.","metadata":{}},{"cell_type":"markdown","source":"Evaluate & Document Baseline Model Performance The baseline CNN achieved around X% training accuracy and Y% validation accuracy. The training and validation curves (Figures X and Y) show that the model [describe: overfits / underfits / is fairly well balanced]. This performance serves as the benchmark for more complex models such as a deeper CNN and an EfficientNetB0 transfer learning model.\n   ","metadata":{}},{"cell_type":"code","source":"model_cnn2 = models.Sequential([\n    layers.Input(shape=(150, 200, 3)),\n\n    # Block 1\n    layers.Conv2D(32, (3, 3), padding='same', activation='relu'),\n    layers.BatchNormalization(),\n    layers.MaxPooling2D(),\n\n    # Block 2\n    layers.Conv2D(64, (3, 3), padding='same', activation='relu'),\n    layers.BatchNormalization(),\n    layers.MaxPooling2D(),\n\n    # Block 3\n    layers.Conv2D(128, (3, 3), padding='same', activation='relu'),\n    layers.BatchNormalization(),\n    layers.MaxPooling2D(),\n    layers.Dropout(0.3),\n\n    # Block 4\n    layers.Conv2D(256, (3, 3), padding='same', activation='relu'),\n    layers.BatchNormalization(),\n    layers.MaxPooling2D(),\n    layers.Dropout(0.4),\n\n    layers.Flatten(),\n    layers.Dense(\n        256,\n        activation='relu',\n        kernel_regularizer=regularizers.l2(1e-4)   # L2 regularization\n    ),\n    layers.Dropout(0.5),\n    layers.Dense(num_classes, activation='softmax')\n])\n\nmodel_cnn2.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T01:16:34.595935Z","iopub.execute_input":"2025-12-01T01:16:34.596169Z","iopub.status.idle":"2025-12-01T01:16:38.485246Z","shell.execute_reply.started":"2025-12-01T01:16:34.596151Z","shell.execute_reply":"2025-12-01T01:16:38.484517Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Compile Model 2 \nCallbacks and Training\nPlot results for Model 2\n","metadata":{}},{"cell_type":"code","source":"model_cnn2.compile(\n    optimizer='adam',\n    loss='sparse_categorical_crossentropy',\n    metrics=['accuracy']\n)\n\ncheckpoint_cnn2 = keras.callbacks.ModelCheckpoint(\n    \"stronger_cnn_cassava.h5\",\n    monitor=\"val_accuracy\",\n    save_best_only=True,\n    verbose=1\n)\n\nearly_cnn2 = keras.callbacks.EarlyStopping(\n    monitor=\"val_accuracy\",\n    patience=5,\n    restore_best_weights=True\n)\n\nreduce_lr_cnn2 = keras.callbacks.ReduceLROnPlateau(\n    monitor=\"val_loss\",\n    factor=0.5,\n    patience=3,\n    verbose=1\n)\n\nhistory_cnn2 = model_cnn2.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=15,                \n    callbacks=[checkpoint_cnn2, early_cnn2, reduce_lr_cnn2]\n)\n\nplt.plot(history_cnn2.history['accuracy'], label='train acc (cnn2)')\nplt.plot(history_cnn2.history['val_accuracy'], label='val acc (cnn2)')\nplt.legend()\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.title('Stronger CNN Accuracy')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T01:16:38.486338Z","iopub.execute_input":"2025-12-01T01:16:38.486623Z","iopub.status.idle":"2025-12-01T01:46:13.051603Z","shell.execute_reply.started":"2025-12-01T01:16:38.486598Z","shell.execute_reply":"2025-12-01T01:46:13.050796Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Generators for EfficientNetB0 (224×224 + preprocess_input)","metadata":{}},{"cell_type":"code","source":"IMG_TL = (224, 224)\nBATCH_TL = 32\nSEED = 2025  # reuse if you like\n\n# Train generator with augmentation + EfficientNet preprocessing\ntrain_datagen_tl = ImageDataGenerator(\n    preprocessing_function=preprocess_input,\n    validation_split=0.2,\n    rotation_range=20,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\n# Validation generator (no augmentation, only preprocessing)\nval_datagen_tl = ImageDataGenerator(\n    preprocessing_function=preprocess_input,\n    validation_split=0.2\n)\n\ntrain_gen_tl = train_datagen_tl.flow_from_dataframe(\n    train_csv,\n    x_col='filepath',\n    y_col='label',\n    target_size=IMG_TL,\n    class_mode='raw',          # integer labels\n    batch_size=BATCH_TL,\n    shuffle=True,\n    seed=SEED,\n    subset='training'\n)\n\nval_gen_tl = val_datagen_tl.flow_from_dataframe(\n    train_csv,\n    x_col='filepath',\n    y_col='label',\n    target_size=IMG_TL,\n    class_mode='raw',\n    batch_size=BATCH_TL,\n    shuffle=False,\n    seed=SEED,\n    subset='validation'\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T01:46:13.052463Z","iopub.execute_input":"2025-12-01T01:46:13.052678Z","iopub.status.idle":"2025-12-01T01:46:28.339960Z","shell.execute_reply.started":"2025-12-01T01:46:13.052661Z","shell.execute_reply":"2025-12-01T01:46:28.339065Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ---- Build EfficientNetB0 model ----\nbase_model = EfficientNetB0(\n    include_top=False,\n    weights='imagenet',\n    input_shape=(224, 224, 3),\n    pooling='avg'\n)\n\nbase_model.trainable = False   # Stage 1: freeze backbone\n\ninputs = layers.Input(shape=(224, 224, 3))\nx = base_model(inputs, training=False)\nx = layers.Dense(256, activation='relu')(x)\nx = layers.Dropout(0.5)(x)\noutputs = layers.Dense(num_classes, activation='softmax')(x)\n\nmodel_eff = keras.Model(inputs, outputs)\nmodel_eff.summary()\n\nmodel_eff.compile(\n    optimizer=keras.optimizers.Adam(1e-3),\n    loss='sparse_categorical_crossentropy',\n    metrics=['accuracy']\n)\n\ncheckpoint_eff = keras.callbacks.ModelCheckpoint(\n    \"efficientnetb0_cassava.h5\",\n    monitor=\"val_accuracy\",\n    save_best_only=True,\n    verbose=1\n)\n\nearly_eff = keras.callbacks.EarlyStopping(\n    monitor=\"val_accuracy\",\n    patience=5,\n    restore_best_weights=True\n)\n\nreduce_lr_eff = keras.callbacks.ReduceLROnPlateau(\n    monitor=\"val_loss\",\n    factor=0.5,\n    patience=3,\n    verbose=1\n)\n\nhistory_eff = model_eff.fit(\n    train_gen_tl,\n    validation_data=val_gen_tl,\n    epochs=15,\n    callbacks=[checkpoint_eff, early_eff, reduce_lr_eff]\n)\n\nplt.plot(history_eff.history['accuracy'], label='train acc (EffB0)')\nplt.plot(history_eff.history['val_accuracy'], label='val acc (EffB0)')\nplt.legend()\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.title('EfficientNetB0 Transfer Learning')\nplt.show()\n\n# ------------------------------------------------------------------\n# Optional: Fine-tune top EfficientNet layers (Stage 2)\n# ------------------------------------------------------------------\nbase_model.trainable = True\nfor layer in base_model.layers[:-20]:     # keep last ~20 layers trainable\n    layer.trainable = False\n\nmodel_eff.compile(\n    optimizer=keras.optimizers.Adam(1e-4),   # lower LR for fine-tuning\n    loss='sparse_categorical_crossentropy',\n    metrics=['accuracy']\n)\n\nhistory_eff_ft = model_eff.fit(\n    train_gen_tl,\n    validation_data=val_gen_tl,\n    epochs=10,\n    callbacks=[checkpoint_eff, early_eff, reduce_lr_eff]\n)\n\nplt.plot(history_eff_ft.history['accuracy'], label='train acc (EffB0 FT)')\nplt.plot(history_eff_ft.history['val_accuracy'], label='val acc (EffB0 FT)')\nplt.legend()\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.title('EfficientNetB0 Fine-tuned')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T01:46:28.340890Z","iopub.execute_input":"2025-12-01T01:46:28.341178Z","iopub.status.idle":"2025-12-01T03:43:49.133725Z","shell.execute_reply.started":"2025-12-01T01:46:28.341158Z","shell.execute_reply":"2025-12-01T03:43:49.132909Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Week 6 Evaluation & Error Analysis on Unseen/Validation Data\n# Use the best-performing model, e.g., model_eff after fine-tuning","metadata":{}},{"cell_type":"code","source":"val_gen_tl.reset()\ny_true = val_gen_tl.labels.astype(int)\ny_prob = model_eff.predict(val_gen_tl)\ny_pred = np.argmax(y_prob, axis=1)\n\nprint(\"Classification report (EfficientNetB0):\")\nprint(classification_report(y_true, y_pred, digits=4))\n\ncm = confusion_matrix(y_true, y_pred)\n\nplt.figure(figsize=(7, 6))\nsns.heatmap(cm, annot=False, fmt='d')\nplt.xlabel('Predicted label')\nplt.ylabel('True label')\nplt.title('Confusion Matrix – EfficientNetB0')\nplt.show()\n\n# Show a few misclassified examples for qualitative error analysis\nmis_idx = np.where(y_true != y_pred)[0][:16]   # first 16 mistakes\nmis_images, mis_true, mis_pred = [], [], []\nfor i in mis_idx:\n    img, label = val_gen_tl[i]\n    # val_gen_tl returns a batch; take the first item\n    mis_images.append(img[0])\n    mis_true.append(label[0])\n    mis_pred.append(y_pred[i])\n\nplt.figure(figsize=(12, 12))\nfor i, (img, t, p) in enumerate(zip(mis_images, mis_true, mis_pred)):\n    plt.subplot(4, 4, i + 1)\n    # undo preprocess_input roughly by rescaling to [0,1] for display\n    img_disp = (img - img.min()) / (img.max() - img.min() + 1e-8)\n    plt.imshow(img_disp)\n    plt.axis('off')\n    plt.title(f\"True: {t}\\nPred: {p}\")\nplt.suptitle('Misclassified Validation Samples – EfficientNetB0', y=0.92)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T03:43:49.134511Z","iopub.execute_input":"2025-12-01T03:43:49.134716Z","iopub.status.idle":"2025-12-01T03:44:26.957166Z","shell.execute_reply.started":"2025-12-01T03:43:49.134701Z","shell.execute_reply":"2025-12-01T03:44:26.956311Z"}},"outputs":[],"execution_count":null}]}