{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84209,"databundleVersionId":9414711,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.applications.inception_v3 import InceptionV3, preprocess_input\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, BatchNormalization\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.layers import GlobalAveragePooling2D, Flatten, Dense\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix\n\nimport cv2\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-12T19:10:32.972100Z","iopub.execute_input":"2024-11-12T19:10:32.972807Z","iopub.status.idle":"2024-11-12T19:10:48.640709Z","shell.execute_reply.started":"2024-11-12T19:10:32.972753Z","shell.execute_reply":"2024-11-12T19:10:48.639630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Load & Read Data","metadata":{}},{"cell_type":"code","source":"# Define paths\nimage_dir = '/kaggle/input/computer-vision-xm/images/kaggle/working/Reorganized_Data/images/'\nlabels_csv = '/kaggle/input/computer-vision-xm/train.csv'\n\n# Load the CSV file with labels\nlabels_df = pd.read_csv(labels_csv)\n\n# Display the first few rows\nlabels_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-12T19:10:59.068869Z","iopub.execute_input":"2024-11-12T19:10:59.069945Z","iopub.status.idle":"2024-11-12T19:10:59.112277Z","shell.execute_reply.started":"2024-11-12T19:10:59.069902Z","shell.execute_reply":"2024-11-12T19:10:59.111289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Preprocessing the loading to numpy arrays","metadata":{}},{"cell_type":"code","source":"# Image size and batch size\nIMG_SIZE = 128 #128, 224 \nBATCH_SIZE = 32\n\n# Load and resize images\ndef load_and_preprocess_image(image_path):\n    image = cv2.imread(image_path)\n    image = cv2.resize(image, (IMG_SIZE, IMG_SIZE))\n    image = image / 255.0  # Normalize\n    return image\n\n# Apply preprocessing to all images\nimages = []\nlabels = []\n\nfor i, row in labels_df.iterrows():\n    image_path = os.path.join(image_dir, row['Images'])\n    images.append(load_and_preprocess_image(image_path))\n    labels.append(row['Labels'])\n\n# Convert lists to numpy arrays\nX = np.array(images)\ny = np.array(labels)","metadata":{"execution":{"iopub.status.busy":"2024-11-12T19:11:04.123820Z","iopub.execute_input":"2024-11-12T19:11:04.124252Z","iopub.status.idle":"2024-11-12T19:20:07.259314Z","shell.execute_reply.started":"2024-11-12T19:11:04.124187Z","shell.execute_reply":"2024-11-12T19:20:07.258427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split The Data","metadata":{}},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n\nprint(f\"Training data: {X_train.shape}, Validation data: {X_val.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-11-12T19:20:12.687054Z","iopub.execute_input":"2024-11-12T19:20:12.687826Z","iopub.status.idle":"2024-11-12T19:20:13.079598Z","shell.execute_reply.started":"2024-11-12T19:20:12.687785Z","shell.execute_reply":"2024-11-12T19:20:13.078594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Augmentation","metadata":{}},{"cell_type":"code","source":"# Image data generator for augmentation\ntrain_datagen = ImageDataGenerator(\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True\n)\n\nval_datagen = ImageDataGenerator()\n\n# Apply to training and validation data\ntrain_generator = train_datagen.flow(X_train, y_train, batch_size=BATCH_SIZE)\nval_generator = val_datagen.flow(X_val, y_val, batch_size=BATCH_SIZE)","metadata":{"execution":{"iopub.status.busy":"2024-11-12T19:20:19.852382Z","iopub.execute_input":"2024-11-12T19:20:19.853156Z","iopub.status.idle":"2024-11-12T19:20:20.122041Z","shell.execute_reply.started":"2024-11-12T19:20:19.853105Z","shell.execute_reply":"2024-11-12T19:20:20.121225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load the InceptionV3 model","metadata":{}},{"cell_type":"code","source":"class CustomGlobalAveragePooling2D(tf.keras.layers.Layer):\n    def __init__(self, **kwargs):\n        super(CustomGlobalAveragePooling2D, self).__init__(**kwargs)\n\n    def call(self, inputs):\n        return tf.keras.backend.mean(inputs, axis=[1, 2])\n","metadata":{"execution":{"iopub.status.busy":"2024-11-12T19:20:24.371919Z","iopub.execute_input":"2024-11-12T19:20:24.372617Z","iopub.status.idle":"2024-11-12T19:20:24.378026Z","shell.execute_reply.started":"2024-11-12T19:20:24.372576Z","shell.execute_reply":"2024-11-12T19:20:24.377032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the InceptionV3 model\nmodel = InceptionV3(weights='imagenet', include_top=False, input_shape=(IMG_SIZE,IMG_SIZE, 3))\n\n# Freeze the base model layers\nfor layer in model.layers:\n    layer.trainable = False\n\nx = model.output\nx = CustomGlobalAveragePooling2D()(x)  # Global Average Pooling\nx = Flatten()(x)\nx = Dense(1, activation='sigmoid')(x) # 2 classes: healthy and diseased\n\nmodel = tf.keras.Model(inputs=model.input, outputs=x)\n\n# Compile the model\nmodel.compile(optimizer='adam',\n              loss='binary_crossentropy', #categorical_crossentropy binary_crossentropy\n              metrics=['accuracy'])\n\n# Add early stopping to monitor validation loss\nearly_stopping = EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)\n\n# Model summary\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-12T19:20:28.259982Z","iopub.execute_input":"2024-11-12T19:20:28.260889Z","iopub.status.idle":"2024-11-12T19:20:32.220080Z","shell.execute_reply.started":"2024-11-12T19:20:28.260844Z","shell.execute_reply":"2024-11-12T19:20:32.219063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training The Model","metadata":{}},{"cell_type":"code","source":"# Train the model\nsteps_per_epoch = len(X_train) // BATCH_SIZE\nvalidation_steps = len(X_val) // BATCH_SIZE\n\nhistory = model.fit(\n        train_generator,\n        steps_per_epoch=steps_per_epoch,\n        validation_data=val_generator,\n        validation_steps=validation_steps,\n        epochs=50, # High number of epochs to allow early stopping to decide when to stop\n        callbacks=[early_stopping]) \n\n  # Save model ./models/saved_leaf_category_model.keras\n# model.save('./models/saved_leaf_disease.keras')      \n","metadata":{"execution":{"iopub.status.busy":"2024-11-12T19:25:04.031024Z","iopub.execute_input":"2024-11-12T19:25:04.031877Z","iopub.status.idle":"2024-11-12T19:25:37.438329Z","shell.execute_reply.started":"2024-11-12T19:25:04.031827Z","shell.execute_reply":"2024-11-12T19:25:37.437443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Elbow Epoch Graph","metadata":{}},{"cell_type":"code","source":"# Plot training & validation accuracy and loss over epochs\nplt.plot(history.history['loss'], label='Training Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-12T19:27:53.673749Z","iopub.execute_input":"2024-11-12T19:27:53.674781Z","iopub.status.idle":"2024-11-12T19:27:54.010966Z","shell.execute_reply.started":"2024-11-12T19:27:53.674734Z","shell.execute_reply":"2024-11-12T19:27:54.009980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluating The Model","metadata":{}},{"cell_type":"code","source":"# Predictions on validation data\ny_pred = model.predict(X_val)\ny_pred_classes = np.where(y_pred > 0.5, 1, 0)\n\n# Evaluate performance\nprint(f\"Accuracy: {accuracy_score(y_val, y_pred_classes)}\")\nprint(\"Classification Report:\")\nprint(classification_report(y_val, y_pred_classes))","metadata":{"execution":{"iopub.status.busy":"2024-11-12T19:27:59.694504Z","iopub.execute_input":"2024-11-12T19:27:59.695501Z","iopub.status.idle":"2024-11-12T19:28:10.084939Z","shell.execute_reply.started":"2024-11-12T19:27:59.695456Z","shell.execute_reply":"2024-11-12T19:28:10.083890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Confusion Matrix","metadata":{}},{"cell_type":"code","source":"cm = confusion_matrix(y_val, y_pred_classes)\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues')\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-12T19:28:17.679299Z","iopub.execute_input":"2024-11-12T19:28:17.679977Z","iopub.status.idle":"2024-11-12T19:28:17.929968Z","shell.execute_reply.started":"2024-11-12T19:28:17.679936Z","shell.execute_reply":"2024-11-12T19:28:17.929088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fine Tuning","metadata":{}},{"cell_type":"code","source":"# For example, increasing the depth of the network or using pretrained models like ResNet or EfficientNet\n# Fine-tune the optimizer (learning rate, regularization techniques, etc.)\n# Fine-tuning the model\nfrom tensorflow.keras.optimizers import Adam\n\n# Lower the learning rate for fine-tuning\nmodel.compile(optimizer=Adam(learning_rate=1e-4), loss='binary_crossentropy', metrics=['accuracy'])\n\n# Optionally unfreeze some layers or keep them all trainable\n# For example, unfreeze the last two convolutional blocks:\nfor layer in model.layers[:-2]:  # Keep the last two layers trainable\n    layer.trainable = True\n\n# Train the model again with fine-tuning\nfine_tune_history = model.fit(train_generator, validation_data=val_generator, epochs=10, callbacks=[early_stopping])\n","metadata":{"execution":{"iopub.status.busy":"2024-11-12T19:28:21.681299Z","iopub.execute_input":"2024-11-12T19:28:21.681736Z","iopub.status.idle":"2024-11-12T19:31:53.777818Z","shell.execute_reply.started":"2024-11-12T19:28:21.681697Z","shell.execute_reply":"2024-11-12T19:31:53.776883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model on the validation data\nval_loss, val_accuracy = model.evaluate(val_generator)\n\n# Print the accuracy\nprint(f'Validation Accuracy: {val_accuracy:.2f}')\n","metadata":{"execution":{"iopub.status.busy":"2024-11-12T19:33:13.687092Z","iopub.execute_input":"2024-11-12T19:33:13.688038Z","iopub.status.idle":"2024-11-12T19:33:14.346203Z","shell.execute_reply.started":"2024-11-12T19:33:13.687994Z","shell.execute_reply":"2024-11-12T19:33:14.345285Z"},"trusted":true},"execution_count":null,"outputs":[]}]}