{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import necessary libraries\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import MobileNetV2\nfrom tensorflow.keras.layers import Dense, Dropout, BatchNormalization, GlobalAveragePooling2D\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\n\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.optimizers import Adam\nimport pandas as pd\nimport numpy as np\nimport os\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, confusion_matrix\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T13:21:52.229949Z","iopub.execute_input":"2025-11-14T13:21:52.230168Z","iopub.status.idle":"2025-11-14T13:21:52.244748Z","shell.execute_reply.started":"2025-11-14T13:21:52.230152Z","shell.execute_reply":"2025-11-14T13:21:52.244195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Step 1: Load the dataset metadata\n# The train.csv contains image_id and label\ntrain_df = pd.read_csv('/kaggle/input/cassava-leaf-disease-classification/train.csv')\n\n# Add the full path to the images\ntrain_df['image_path'] = '/kaggle/input/cassava-leaf-disease-classification/train_images/' + train_df['image_id']\n\n# Convert label to string for categorical classification\ntrain_df['label'] = train_df['label'].astype(str)\n\n# Split the data into training and validation sets (80-20 split)\ntrain_data, val_data = train_test_split(train_df, test_size=0.2, random_state=42, stratify=train_df['label'])\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T13:21:52.245919Z","iopub.execute_input":"2025-11-14T13:21:52.246098Z","iopub.status.idle":"2025-11-14T13:21:52.334631Z","shell.execute_reply.started":"2025-11-14T13:21:52.246077Z","shell.execute_reply":"2025-11-14T13:21:52.334010Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 2: Data Augmentation and Preprocessing\n# Use ImageDataGenerator for data augmentation to improve generalization\n# Rescale images, apply rotations, flips, etc.\nIMG_SIZE = 224  # MobileNetV2 input size\nBATCH_SIZE = 32\n\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\nval_datagen = ImageDataGenerator(rescale=1./255)\n\n# Create generators\ntrain_generator = train_datagen.flow_from_dataframe(\n    train_data,\n    x_col='image_path',\n    y_col='label',\n    target_size=(IMG_SIZE, IMG_SIZE),\n    batch_size=BATCH_SIZE,\n    class_mode='categorical'\n)\n\nval_generator = val_datagen.flow_from_dataframe(\n    val_data,\n    x_col='image_path',\n    y_col='label',\n    target_size=(IMG_SIZE, IMG_SIZE),\n    batch_size=BATCH_SIZE,\n    class_mode='categorical'\n)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T13:21:56.526267Z","iopub.execute_input":"2025-11-14T13:21:56.526644Z","iopub.status.idle":"2025-11-14T13:23:03.250680Z","shell.execute_reply.started":"2025-11-14T13:21:56.526616Z","shell.execute_reply":"2025-11-14T13:23:03.249953Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Step 3: Build the Model using Transfer Learning\n# Use MobileNetV2 as base model (pre-trained on ImageNet), suitable for mobile-quality images\nbase_model = MobileNetV2(weights='imagenet', include_top=False, input_shape=(IMG_SIZE, IMG_SIZE, 3))\n\n# Freeze the base model layers to prevent updating pre-trained weights initially\nbase_model.trainable = False\n\n# Add custom layers on top\nmodel = Sequential([\n    base_model,\n    GlobalAveragePooling2D(),\n    Dense(512, activation='relu'),\n    BatchNormalization(),  # Batch Normalization to stabilize training\n    Dropout(0.5),  # Dropout to prevent overfitting\n    Dense(5, activation='softmax')  # 5 classes: 4 diseases + healthy\n])\n\n# Compile the model\nmodel.compile(\n    optimizer=Adam(learning_rate=0.001),\n    loss='categorical_crossentropy',\n    metrics=['accuracy']\n)\n\n# Model summary\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T13:24:04.716919Z","iopub.execute_input":"2025-11-14T13:24:04.717634Z","iopub.status.idle":"2025-11-14T13:24:05.427497Z","shell.execute_reply.started":"2025-11-14T13:24:04.717607Z","shell.execute_reply":"2025-11-14T13:24:05.426768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 4: Train the Model\n# Use callbacks for early stopping and learning rate reduction\n\nearly_stopping = EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)\nreduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=3, min_lr=0.00001)\n\n# Train the model\nEPOCHS = 20  # You can adjust this based on performance\n\nhistory = model.fit(\n    train_generator,\n    steps_per_epoch=len(train_generator),\n    validation_data=val_generator,\n    validation_steps=len(val_generator),\n    epochs=EPOCHS,\n    callbacks=[early_stopping, reduce_lr]\n)\n\n# Optional: Unfreeze some layers for fine-tuning\n# After initial training, unfreeze the base model and fine-tune with lower learning rate\nbase_model.trainable = True\nmodel.compile(optimizer=Adam(learning_rate=0.0001), loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Fine-tune for a few more epochs\nhistory_fine = model.fit(\n    train_generator,\n    steps_per_epoch=len(train_generator),\n    validation_data=val_generator,\n    validation_steps=len(val_generator),\n    epochs=10,  # Fewer epochs for fine-tuning\n    callbacks=[early_stopping, reduce_lr]\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T13:46:39.427092Z","iopub.execute_input":"2025-11-14T13:46:39.427607Z","iopub.status.idle":"2025-11-14T13:46:42.347343Z","shell.execute_reply.started":"2025-11-14T13:46:39.427582Z","shell.execute_reply":"2025-11-14T13:46:42.346218Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 5: Evaluate the Model\n# Plot training history\nplt.plot(history.history['accuracy'] + history_fine.history['accuracy'])\nplt.plot(history.history['val_accuracy'] + history_fine.history['val_accuracy'])\nplt.title('Model Accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.legend(['Train', 'Validation'], loc='upper left')\nplt.show()\n\n# Generate predictions on validation set for confusion matrix\nval_predictions = model.predict(val_generator)\nval_pred_classes = np.argmax(val_predictions, axis=1)\nval_true_classes = val_generator.classes\n\nprint(classification_report(val_true_classes, val_pred_classes))\nprint(confusion_matrix(val_true_classes, val_pred_classes))\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T13:44:25.187761Z","iopub.execute_input":"2025-11-14T13:44:25.188387Z","iopub.status.idle":"2025-11-14T13:44:50.188173Z","shell.execute_reply.started":"2025-11-14T13:44:25.188357Z","shell.execute_reply":"2025-11-14T13:44:50.187344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 6: Prepare Submission\n# Load test images\ntest_dir = '/kaggle/input/cassava-leaf-disease-classification/test_images/'\ntest_images = os.listdir(test_dir)\n\n# Create a DataFrame for test images\ntest_df = pd.DataFrame(test_images, columns=['image_id'])\ntest_df['image_path'] = test_dir + test_df['image_id']\n\n# Test data generator (no augmentation, just rescale)\ntest_datagen = ImageDataGenerator(rescale=1./255)\ntest_generator = test_datagen.flow_from_dataframe(\n    test_df,\n    x_col='image_path',\n    y_col=None,\n    target_size=(IMG_SIZE, IMG_SIZE),\n    batch_size=BATCH_SIZE,\n    class_mode=None,\n    shuffle=False\n)\n\n# Predict on test set\ntest_predictions = model.predict(test_generator)\ntest_pred_classes = np.argmax(test_predictions, axis=1)\n\n# Create submission file\nsubmission = pd.DataFrame({\n    'image_id': test_df['image_id'],\n    'label': test_pred_classes\n})\n\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Submission file created successfully!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T13:45:13.333759Z","iopub.execute_input":"2025-11-14T13:45:13.334566Z","iopub.status.idle":"2025-11-14T13:45:16.926096Z","shell.execute_reply.started":"2025-11-14T13:45:13.334522Z","shell.execute_reply":"2025-11-14T13:45:16.925515Z"}},"outputs":[],"execution_count":null}]}