{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84209,"databundleVersionId":9414711,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Importing Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, BatchNormalization\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix\nimport cv2\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-20T11:50:33.696987Z","iopub.execute_input":"2024-11-20T11:50:33.697345Z","iopub.status.idle":"2024-11-20T11:50:53.752808Z","shell.execute_reply.started":"2024-11-20T11:50:33.697313Z","shell.execute_reply":"2024-11-20T11:50:53.752146Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load & Read Data","metadata":{}},{"cell_type":"code","source":"# Define paths\nimage_dir = '/kaggle/input/computer-vision-xm/images/kaggle/working/Reorganized_Data/images/'\nlabels_csv = '/kaggle/input/computer-vision-xm/train.csv'\n\n# Load the CSV file with labels\nlabels_df = pd.read_csv(labels_csv)\n\n# Display the first few rows\nlabels_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-20T11:50:53.754129Z","iopub.execute_input":"2024-11-20T11:50:53.754585Z","iopub.status.idle":"2024-11-20T11:50:53.793478Z","shell.execute_reply.started":"2024-11-20T11:50:53.754558Z","shell.execute_reply":"2024-11-20T11:50:53.792507Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"# Image size and batch size\nIMG_SIZE = 128\nBATCH_SIZE = 32\n\n# Load and resize images\ndef load_and_preprocess_image(image_path):\n    image = cv2.imread(image_path)\n    image = cv2.resize(image, (IMG_SIZE, IMG_SIZE))\n    image = image / 255.0  # Normalize\n    return image\n\n# Apply preprocessing to all images\nimages = []\nlabels = []\n\nfor i, row in labels_df.iterrows():\n    image_path = os.path.join(image_dir, row['Images'])\n    images.append(load_and_preprocess_image(image_path))\n    labels.append(row['Labels'])\n\n# Convert lists to numpy arrays\nX = np.array(images)\ny = np.array(labels)","metadata":{"execution":{"iopub.status.busy":"2024-11-20T11:50:53.794784Z","iopub.execute_input":"2024-11-20T11:50:53.795185Z","iopub.status.idle":"2024-11-20T12:00:25.557563Z","shell.execute_reply.started":"2024-11-20T11:50:53.795144Z","shell.execute_reply":"2024-11-20T12:00:25.556822Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Split The Data","metadata":{}},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n\nprint(f\"Training data: {X_train.shape}, Validation data: {X_val.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-11-20T12:00:25.558584Z","iopub.execute_input":"2024-11-20T12:00:25.558842Z","iopub.status.idle":"2024-11-20T12:00:25.947409Z","shell.execute_reply.started":"2024-11-20T12:00:25.558818Z","shell.execute_reply":"2024-11-20T12:00:25.946452Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Augmentation","metadata":{}},{"cell_type":"code","source":"# Image data generator for augmentation\ntrain_datagen = ImageDataGenerator(\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True\n)\n\nval_datagen = ImageDataGenerator()\n\n# Apply to training and validation data\ntrain_generator = train_datagen.flow(X_train, y_train, batch_size=BATCH_SIZE)\nval_generator = val_datagen.flow(X_val, y_val, batch_size=BATCH_SIZE)","metadata":{"execution":{"iopub.status.busy":"2024-11-20T12:00:25.949307Z","iopub.execute_input":"2024-11-20T12:00:25.949570Z","iopub.status.idle":"2024-11-20T12:00:26.198315Z","shell.execute_reply.started":"2024-11-20T12:00:25.949544Z","shell.execute_reply":"2024-11-20T12:00:26.197599Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Building The CNN Model","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.regularizers import l2\n\n# Build the model with regularization\nmodel = Sequential([\n    Conv2D(32, (3, 3), activation='relu', input_shape=(IMG_SIZE, IMG_SIZE, 3),\n           kernel_regularizer=l2(1e-4)),  # L2 regularization\n    MaxPooling2D(pool_size=(2, 2)),\n    BatchNormalization(),\n    \n    Conv2D(64, (3, 3), activation='relu',\n           kernel_regularizer=l2(1e-4)),  # L2 regularization\n    MaxPooling2D(pool_size=(2, 2)),\n    BatchNormalization(),\n    \n    Conv2D(128, (3, 3), activation='relu',\n           kernel_regularizer=l2(1e-4)),  # L2 regularization\n    MaxPooling2D(pool_size=(2, 2)),\n    BatchNormalization(),\n    \n    Flatten(),\n    Dense(256, activation='relu', kernel_regularizer=l2(1e-4)),  # L2 regularization\n    Dropout(0.5),\n    Dense(1, activation='sigmoid')  # Binary classification\n])\n\n# Compile the model\nadam_optimizer = Adam(learning_rate=0.0005, beta_1=0.9, beta_2=0.999, clipvalue=1.0)\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-20T15:41:17.667834Z","iopub.execute_input":"2024-11-20T15:41:17.668469Z","iopub.status.idle":"2024-11-20T15:41:17.757994Z","shell.execute_reply.started":"2024-11-20T15:41:17.668435Z","shell.execute_reply":"2024-11-20T15:41:17.757194Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training The Model","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping\n\n# Define EarlyStopping callback\nearly_stopping = EarlyStopping(\n    monitor='val_loss',  # Metric to monitor (validation loss is common)\n    patience=5,          # Number of epochs with no improvement before stopping\n    restore_best_weights=True  # Restore weights from the epoch with the best value of the monitored metric\n)\n\n# Train the model with EarlyStopping\nhistory = model.fit(\n    train_generator,\n    validation_data=val_generator,\n    epochs=50,  # Set the maximum number of epochs\n    callbacks=[early_stopping],  # Add EarlyStopping to the training process\n    verbose=1\n)\n\n# Save model\nmodel.save('leaf_disease_classifier.h5')","metadata":{"execution":{"iopub.status.busy":"2024-11-20T15:41:22.234845Z","iopub.execute_input":"2024-11-20T15:41:22.235683Z","iopub.status.idle":"2024-11-20T15:43:43.949775Z","shell.execute_reply.started":"2024-11-20T15:41:22.235632Z","shell.execute_reply":"2024-11-20T15:43:43.948799Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Evaluating The Model","metadata":{}},{"cell_type":"code","source":"# Predictions on validation data\ny_pred = model.predict(X_val)\ny_pred_classes = np.where(y_pred > 0.5, 1, 0)\n\n# Evaluate performance\nprint(f\"Accuracy: {accuracy_score(y_val, y_pred_classes)}\")\nprint(\"Classification Report:\")\nprint(classification_report(y_val, y_pred_classes))","metadata":{"execution":{"iopub.status.busy":"2024-11-20T15:43:53.196201Z","iopub.execute_input":"2024-11-20T15:43:53.196554Z","iopub.status.idle":"2024-11-20T15:43:54.253539Z","shell.execute_reply.started":"2024-11-20T15:43:53.196522Z","shell.execute_reply":"2024-11-20T15:43:54.252783Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Confusion Matrix","metadata":{}},{"cell_type":"code","source":"cm = confusion_matrix(y_val, y_pred_classes)\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues')\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-20T15:43:57.584255Z","iopub.execute_input":"2024-11-20T15:43:57.585109Z","iopub.status.idle":"2024-11-20T15:43:57.820897Z","shell.execute_reply.started":"2024-11-20T15:43:57.585062Z","shell.execute_reply":"2024-11-20T15:43:57.819846Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Loss-epoch Graph","metadata":{}},{"cell_type":"code","source":"# For example, increasing the depth of the network or using pretrained models like ResNet or EfficientNet\n# Fine-tune the opti# Plot training & validation accuracy and loss over epochs\nplt.plot(history.history['loss'], label='Training Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T15:44:02.868868Z","iopub.execute_input":"2024-11-20T15:44:02.869225Z","iopub.status.idle":"2024-11-20T15:44:03.034096Z","shell.execute_reply.started":"2024-11-20T15:44:02.869194Z","shell.execute_reply":"2024-11-20T15:44:03.033205Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom tensorflow.keras.models import load_model\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\n\n# Define paths\nimage_dir = '/kaggle/input/computer-vision-xm/images/kaggle/working/Reorganized_Data/images/'  # Path to the images directory\ntest_csv = '/kaggle/input/computer-vision-xm/test.csv'  # Path to the CSV file with image filenames\n\n# Load the CSV file with image filenames\ntest_df = pd.read_csv(test_csv)\n\n# Display the first few rows of the dataframe to check\nprint(test_df.head())\n\n# Extract the image filenames from the CSV and filter only JPG files\nimage_filenames = test_df['Images'].tolist()\n\n# Filter images to include only .JPG files (case insensitive)\nimage_filenames = [img for img in image_filenames if img.lower().endswith('.jpg')]\n\n# Generate full paths for each image\nimage_paths = [os.path.join(image_dir, img) for img in image_filenames]\n\n# Check if the images exist (optional, for debugging)\nfor image_path in image_paths:\n    if not os.path.exists(image_path):\n        print(f\"Image not found: {image_path}\")\n\n# Define the image size used for training\nIMG_SIZE = 128\n\n# Preprocessing function for images\ndef preprocess_image(image_path):\n    image = load_img(image_path, target_size=(IMG_SIZE, IMG_SIZE))\n    image = img_to_array(image) / 255.0  # Normalize the image\n    return np.expand_dims(image, axis=0)\n\n# Load the pre-trained model (update the model path if necessary)\nmodel_path = '/kaggle/working/leaf_disease_classifier.h5'  # Change to your model's path\nmodel = load_model(model_path)\n\n# Prepare for predictions\npredictions = []\nimage_names = []\n\n# Iterate over all images and make predictions\nfor image_path in image_paths:\n    image_names.append(os.path.basename(image_path))  # Extract image filename for submission\n    preprocessed_image = preprocess_image(image_path)\n    prediction = model.predict(preprocessed_image)  # Get model prediction\n    predictions.append(int(prediction > 0.5))  # Convert probabilities to binary predictions (0 or 1)\n\n# Create a DataFrame for the submission\nsubmission_df = pd.DataFrame({\n    'Images': image_names,\n    'Labels': predictions\n})\n\n# Save the submission file\nsubmission_file_path = '/kaggle/working/submission.csv'\nsubmission_df.to_csv(submission_file_path, index=False)\n\nprint(f'Submission file saved to {submission_file_path}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T15:56:18.945688Z","iopub.execute_input":"2024-11-20T15:56:18.946400Z","iopub.status.idle":"2024-11-20T15:59:14.514519Z","shell.execute_reply.started":"2024-11-20T15:56:18.946367Z","shell.execute_reply":"2024-11-20T15:59:14.513774Z"}},"outputs":[],"execution_count":null}]}