{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"colab":{"provenance":[],"machine_shape":"hm","gpuType":"T4"},"accelerator":"GPU","kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"sourceType":"competition"},{"sourceId":9976452,"sourceType":"datasetVersion","datasetId":6138405},{"sourceId":204128081,"sourceType":"kernelVersion"},{"sourceId":6128,"sourceType":"modelInstanceVersion","modelInstanceId":4596,"modelId":2797}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Pre-Processing","metadata":{"id":"uY8a29I7EUuO"}},{"cell_type":"markdown","source":"### Data Augmentation","metadata":{"id":"qeUtKpnyFszn"}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom tensorflow.keras.applications import EfficientNetV2B0\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, Dropout, Flatten, GlobalAveragePooling2D\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator, load_img, img_to_array\nfrom tensorflow.keras.optimizers import Adam\nfrom sklearn.metrics import f1_score","metadata":{"id":"9Qthffyhr1NR","trusted":true,"execution":{"iopub.status.busy":"2024-11-21T19:49:32.597968Z","iopub.execute_input":"2024-11-21T19:49:32.598636Z","iopub.status.idle":"2024-11-21T19:49:44.017166Z","shell.execute_reply.started":"2024-11-21T19:49:32.598598Z","shell.execute_reply":"2024-11-21T19:49:44.016252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/cassava-leaf-disease-classification/train.csv')\nprint(train_df.head())","metadata":{"id":"rRnvdUh3F5we","outputId":"95232ef4-b94d-4c39-d471-81e822691177","trusted":true,"execution":{"iopub.status.busy":"2024-11-21T16:44:30.512708Z","iopub.execute_input":"2024-11-21T16:44:30.513159Z","iopub.status.idle":"2024-11-21T16:44:30.545351Z","shell.execute_reply.started":"2024-11-21T16:44:30.513132Z","shell.execute_reply":"2024-11-21T16:44:30.544666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\n\nwith open('/kaggle/input/cassava-leaf-disease-classification/label_num_to_disease_map.json', 'r') as f:\n    label_map = json.load(f)\nprint(label_map)","metadata":{"id":"nJ9XyiVCF9en","outputId":"2fa32d08-d4c2-4f00-f36c-f3eadaadc114","trusted":true,"execution":{"iopub.status.busy":"2024-11-21T16:44:30.546292Z","iopub.execute_input":"2024-11-21T16:44:30.546644Z","iopub.status.idle":"2024-11-21T16:44:30.565464Z","shell.execute_reply.started":"2024-11-21T16:44:30.546607Z","shell.execute_reply":"2024-11-21T16:44:30.564725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Read the CSV file\ntrain_df = pd.read_csv('/kaggle/input/cassava-leaf-disease-classification/train.csv')\n\n# Convert the 'label' column to strings\ntrain_df['label'] = train_df['label'].astype(str)\n\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# Custom data augmentation for plant disease detection\ndatagen = ImageDataGenerator(\n    rescale=1.0 / 255.0,  # Normalize\n    rotation_range=40,    # Random rotations\n    width_shift_range=0.3,  # Horizontal shift\n    height_shift_range=0.3, # Vertical shift\n    brightness_range=[0.7, 1.5],  # Vary brightness\n    zoom_range=0.4,  # Random zoom\n    horizontal_flip=True,  # Random horizontal flips\n    vertical_flip=True,  # Random vertical flips\n    fill_mode=\"nearest\",  # Fill empty pixels\n    validation_split=0.1\n)\n\n# Define training data generator\ntrain_gen = datagen.flow_from_dataframe(\n    dataframe=train_df,\n    directory='/kaggle/input/cassava-leaf-disease-classification/train_images',\n    x_col='image_id',\n    y_col='label',\n    target_size=(128, 128),\n    batch_size=16,\n    class_mode='categorical',  # One-hot encode labels\n    subset='training'\n)\n\n# Define validation data generator\nval_gen = datagen.flow_from_dataframe(\n    dataframe=train_df,\n    directory='/kaggle/input/cassava-leaf-disease-classification/train_images',\n    x_col='image_id',\n    y_col='label',\n    target_size=(128, 128),\n    batch_size=16,\n    class_mode='categorical',\n    subset='validation'\n)\n\nprint(train_df['label'].unique())  # Should print string labels like ['0', '1', '2', '3', '4']","metadata":{"id":"rdws8hsoJMJi","outputId":"86a6b714-de14-470f-ea13-933f84805480","trusted":true,"execution":{"iopub.status.busy":"2024-11-21T16:44:30.567553Z","iopub.execute_input":"2024-11-21T16:44:30.568137Z","iopub.status.idle":"2024-11-21T16:45:22.009290Z","shell.execute_reply.started":"2024-11-21T16:44:30.568109Z","shell.execute_reply":"2024-11-21T16:45:22.008431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check a batch of data from the training generator\nx_batch, y_batch = next(train_gen)\nprint(f\"Training batch image shape: {x_batch.shape}\")\nprint(f\"Training batch label shape: {y_batch.shape}\")\nprint(f\"Sample labels: {y_batch[:5]}\")\n\n# Check a batch of data from the validation generator\nx_batch_val, y_batch_val = next(val_gen)\nprint(f\"Validation batch image shape: {x_batch_val.shape}\")\nprint(f\"Validation batch label shape: {y_batch_val.shape}\")\nprint(f\"Sample validation labels: {y_batch_val[:5]}\")","metadata":{"id":"h4y9i7DCSnlo","outputId":"400f8b5d-812d-495b-a1fc-262381aafb63","trusted":true,"execution":{"iopub.status.busy":"2024-11-21T16:45:22.010447Z","iopub.execute_input":"2024-11-21T16:45:22.010732Z","iopub.status.idle":"2024-11-21T16:45:23.221537Z","shell.execute_reply.started":"2024-11-21T16:45:22.010706Z","shell.execute_reply":"2024-11-21T16:45:23.220641Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Visualize sample images","metadata":{"id":"G32MJrjtGBj9"}},{"cell_type":"code","source":"import os\nimport cv2\nimport matplotlib.pyplot as plt\n\n# Path to the training images\ntrain_images_path = '/kaggle/input/cassava-leaf-disease-classification/train_images'\n\n# Display the first few images\nfor i in range(5):\n    img_path = os.path.join(train_images_path, train_df.loc[i, 'image_id'])\n    img = cv2.imread(img_path)\n    plt.imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\n    plt.title(f\"Label: {train_df.loc[i, 'label']} ({label_map[str(train_df.loc[i, 'label'])]})\")\n    plt.axis('off')\n    plt.show()","metadata":{"id":"25NGj4IyGD3J","outputId":"ad46c6c4-12dc-4ea7-b07d-bbc3f0610b0f","trusted":true,"execution":{"iopub.status.busy":"2024-11-21T16:45:23.222676Z","iopub.execute_input":"2024-11-21T16:45:23.222961Z","iopub.status.idle":"2024-11-21T16:45:24.747085Z","shell.execute_reply.started":"2024-11-21T16:45:23.222935Z","shell.execute_reply":"2024-11-21T16:45:24.746132Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training","metadata":{"id":"K4vIcF7VGiGo"}},{"cell_type":"code","source":"weights_path = '/kaggle/input/efficientnetv2-b0-notop-h5/efficientnetv2-b0_notop.h5'\nbase_model = EfficientNetV2B0(weights=None, include_top=False, input_shape=(128, 128, 3))\nbase_model.load_weights(weights_path)\n\n# Add custom classification layers\nx = GlobalAveragePooling2D()(base_model.output)\nx = Dense(256, activation='relu')(x)\nx = Dropout(0.5)(x)\noutput = Dense(5, activation='softmax')(x)  # 5 classes for cassava leaf diseases\n\n# Build the model\nmodel_effnet = Model(inputs=base_model.input, outputs=output)\n\n# Compile the model\nmodel_effnet.compile(optimizer=Adam(learning_rate=1e-4), loss='categorical_crossentropy', metrics=['accuracy'])","metadata":{"id":"naApql72NW4L","trusted":true,"execution":{"iopub.status.busy":"2024-11-21T20:18:19.498141Z","iopub.execute_input":"2024-11-21T20:18:19.499087Z","iopub.status.idle":"2024-11-21T20:18:21.144664Z","shell.execute_reply.started":"2024-11-21T20:18:19.499052Z","shell.execute_reply":"2024-11-21T20:18:21.143953Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\n\n# Early stopping to avoid overfitting\nearly_stopping = EarlyStopping(\n    monitor='val_accuracy',\n    patience=3,\n    restore_best_weights=True\n)\n\n# Reduce learning rate when validation loss plateaus\nreduce_lr = ReduceLROnPlateau(\n    monitor='val_loss',\n    factor=0.5,\n    patience=2,\n    min_lr=1e-6\n)\n\n# Train the model\nhistory_effnet = model_effnet.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=50,  # Allow more epochs for fine-tuning\n    callbacks=[early_stopping, reduce_lr]\n)","metadata":{"id":"xE6yvbEdNiT_","outputId":"ec454947-d64a-4244-8783-c79619c2305c","trusted":true,"execution":{"iopub.status.busy":"2024-11-21T16:45:27.190954Z","iopub.execute_input":"2024-11-21T16:45:27.191179Z","iopub.status.idle":"2024-11-21T19:03:04.347265Z","shell.execute_reply.started":"2024-11-21T16:45:27.191155Z","shell.execute_reply":"2024-11-21T19:03:04.345624Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Evaluation and Comparison\n","metadata":{"id":"X7kHeWGnNlQk"}},{"cell_type":"code","source":"# Evaluate validation accuracy\nval_loss, val_accuracy = model_effnet.evaluate(val_gen)\nprint(f\"Validation Accuracy: {val_accuracy}\")\n\n# Calculate F1-score\nval_preds = model_effnet.predict(val_gen)\nval_preds_classes = np.argmax(val_preds, axis=1)\nval_true_classes = val_gen.classes\n\nf1 = f1_score(val_true_classes, val_preds_classes, average='weighted')\nprint(f\"Validation F1-Score: {f1}\")","metadata":{"id":"0xksm2BgNorg","outputId":"f7e6e201-d4a4-4455-aa12-ab8811daa7fd","trusted":true,"execution":{"iopub.status.busy":"2024-11-21T19:03:04.348743Z","iopub.execute_input":"2024-11-21T19:03:04.349071Z","iopub.status.idle":"2024-11-21T19:04:27.723175Z","shell.execute_reply.started":"2024-11-21T19:03:04.349043Z","shell.execute_reply":"2024-11-21T19:04:27.722041Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Training Curves","metadata":{"id":"jWPao0OONv7X"}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Plot training and validation accuracy\nplt.figure(figsize=(10, 5))\nplt.plot(history_effnet.history['accuracy'], label='Train Accuracy')\nplt.plot(history_effnet.history['val_accuracy'], label='Validation Accuracy')\nplt.title('Training and Validation Accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()","metadata":{"id":"ZTUdVTvVN14G","outputId":"b5accb63-9518-4a7f-c5a4-41265cc7261e","trusted":true,"execution":{"iopub.status.busy":"2024-11-21T19:04:27.726576Z","iopub.execute_input":"2024-11-21T19:04:27.727198Z","iopub.status.idle":"2024-11-21T19:04:27.967108Z","shell.execute_reply.started":"2024-11-21T19:04:27.727167Z","shell.execute_reply":"2024-11-21T19:04:27.965855Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## TEST PREDICTIONS","metadata":{"id":"ZPtInnDKN6Td"}},{"cell_type":"code","source":"!mkdir -p test_images/dummy_class\n!mv test_images/dummy_class/2216849948.jpg","metadata":{"id":"YaI9YJ_8Fajj","trusted":true,"execution":{"iopub.status.busy":"2024-11-21T19:19:39.084041Z","iopub.execute_input":"2024-11-21T19:19:39.084939Z","iopub.status.idle":"2024-11-21T19:19:41.801101Z","shell.execute_reply.started":"2024-11-21T19:19:39.084901Z","shell.execute_reply":"2024-11-21T19:19:41.799700Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the path to test images\ntest_dir = '/kaggle/input/cassava-leaf-disease-classification/test_images/'\n\n# Preprocess test images\ntest_image_paths = [os.path.join(test_dir, fname) for fname in os.listdir(test_dir) if fname.endswith(('.jpg', '.png'))]\n\n# Load and preprocess each test image\ntest_images = np.array([\n    img_to_array(load_img(img_path, target_size=(128, 128))) / 255.0\n    for img_path in test_image_paths\n])\n\n# Predict labels for test images\ntest_preds_effnet = model_effnet.predict(test_images)\ntest_labels_effnet = np.argmax(test_preds_effnet, axis=1)\n\n# Create a submission DataFrame\nsubmission_df = pd.DataFrame({\n    'image_id': [os.path.basename(img_path) for img_path in test_image_paths],\n    'label': test_labels_effnet\n})\n\n# Save predictions to a CSV file\nsubmission_csv_path = '/kaggle/working/submission.csv'\nsubmission_df.to_csv(submission_csv_path, index=False)\n\nprint(f\"Submission file saved at: {submission_csv_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T19:23:35.664573Z","iopub.execute_input":"2024-11-21T19:23:35.664980Z","iopub.status.idle":"2024-11-21T19:23:42.586359Z","shell.execute_reply.started":"2024-11-21T19:23:35.664947Z","shell.execute_reply":"2024-11-21T19:23:42.585417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df = pd.read_csv('submission.csv')\nprint(submission_df.head())","metadata":{"id":"xfxtah-sCsbM","outputId":"c8d91260-652d-4377-d818-9c4e62d7c477","trusted":true,"execution":{"iopub.status.busy":"2024-11-21T19:24:44.710756Z","iopub.execute_input":"2024-11-21T19:24:44.711713Z","iopub.status.idle":"2024-11-21T19:24:44.721159Z","shell.execute_reply.started":"2024-11-21T19:24:44.711673Z","shell.execute_reply":"2024-11-21T19:24:44.720126Z"}},"outputs":[],"execution_count":null}]}