{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Code 1","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\nfrom sklearn.metrics import accuracy_score, classification_report\nimport cv2\nimport os\n\n# Load the CSV file\ndf = pd.read_csv('/kaggle/input/cassava-leaf-disease-classification/train.csv')\n\n# Sample a smaller subset for faster training\ndf_sample = df.sample(n=1000, random_state=42)\n\n# Define path to the images\nimage_path = '/kaggle/input/cassava-leaf-disease-classification/train_images'\n\n# Function to load and preprocess images with error handling\ndef load_images(image_ids, image_path, img_size=(64, 64)):\n    images = []\n    missing_files = []\n    for image_id in image_ids:\n        img_path = os.path.join(image_path, image_id)\n        img = cv2.imread(img_path)\n        if img is not None:\n            img = cv2.resize(img, img_size)\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n            images.append(img)\n        else:\n            missing_files.append(img_path)\n    if missing_files:\n        print(f\"Missing or unreadable files: {len(missing_files)}\")\n        for missing_file in missing_files[:10]:  # Print first 10 missing files as a sample\n            print(missing_file)\n    return np.array(images)\n\n# Load and preprocess images\nX = load_images(df_sample['image_id'], image_path)\ny = df_sample['label']\n\n# Normalize pixel values\nX = X / 255.0\n\n# Flatten images for logistic regression\nX_flattened = X.reshape(X.shape[0], -1)\n\n# Encode labels\nlabel_encoder = LabelEncoder()\ny_encoded = label_encoder.fit_transform(y)\n\n# Split data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X_flattened, y_encoded, test_size=0.2, random_state=42)\n\n# Standardize features\nscaler = StandardScaler()\nX_train = scaler.fit_transform(X_train)\nX_test = scaler.transform(X_test)\n\n# Train logistic regression model\nlog_reg = LogisticRegression(multi_class='ovr', max_iter=100, solver='saga', tol=1e-2)\nlog_reg.fit(X_train, y_train)\n\n# Evaluate model\ny_pred = log_reg.predict(X_test)\naccuracy = accuracy_score(y_test, y_pred)\nprint(f'Accuracy: {accuracy:.4f}')\n\n# Convert numerical labels to strings for classification report\ntarget_names = [str(i) for i in label_encoder.classes_]\nprint(classification_report(y_test, y_pred, target_names=target_names))","metadata":{"execution":{"iopub.status.busy":"2024-06-16T15:13:17.099279Z","iopub.execute_input":"2024-06-16T15:13:17.099704Z","iopub.status.idle":"2024-06-16T15:14:18.975128Z","shell.execute_reply.started":"2024-06-16T15:13:17.099671Z","shell.execute_reply":"2024-06-16T15:14:18.973325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Code 2","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Flatten\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nimport cv2\nimport os\n\n# Load the CSV file\ndf = pd.read_csv('/kaggle/input/cassava-leaf-disease-classification/train.csv')\n\n# Sample a smaller subset for faster training (optional, for initial testing)\ndf_sample = df.sample(n=1000, random_state=42)\n\n# Define path to the images\nimage_path = '/kaggle/input/cassava-leaf-disease-classification/train_images'\n\n# Function to load and preprocess images with error handling\ndef load_images(image_ids, image_path, img_size=(128, 128)):\n    images = []\n    labels = []\n    missing_files = []\n    for image_id in image_ids:\n        img_path = os.path.join(image_path, image_id)\n        img = cv2.imread(img_path)\n        if img is not None:\n            img = cv2.resize(img, img_size)\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n            images.append(img)\n            label = df[df['image_id'] == image_id]['label'].values[0]\n            labels.append(label)\n        else:\n            missing_files.append(img_path)\n    if missing_files:\n        print(f\"Missing or unreadable files: {len(missing_files)}\")\n        for missing_file in missing_files[:10]:  # Print first 10 missing files as a sample\n            print(missing_file)\n    return np.array(images), np.array(labels)\n\n# Load and preprocess images\nX, y = load_images(df_sample['image_id'], image_path)\ny = df_sample['label'].values\n\n# Normalize pixel values\nX = X / 255.0\n\n# Encode labels\nlabel_encoder = LabelEncoder()\ny_encoded = label_encoder.fit_transform(y)\n\n# Split data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y_encoded, test_size=0.2, random_state=42)\n\n# Build the feedforward neural network model\nmodel = Sequential([\n    Flatten(input_shape=(128, 128, 3)),  # Flattening the 128x128 RGB images\n    Dense(512, activation='relu'),\n    Dense(256, activation='relu'),\n    Dense(128, activation='relu'),\n    Dense(5, activation='softmax')  # Assuming 5 classes for cassava leaf disease\n])\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\n# Train the model\nhistory = model.fit(X_train, y_train, epochs=10, validation_data=(X_test, y_test), batch_size=32)\n\n# Evaluate the model\nloss, accuracy = model.evaluate(X_test, y_test)\nprint(f'Accuracy: {accuracy:.4f}')\n\n# Save the model\nmodel.save('cassava_fnn_model.h5')\n\n# Plot training & validation accuracy and loss values\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(12, 4))\nplt.subplot(1, 2, 1)\nplt.plot(history.history['accuracy'], label='Train')\nplt.plot(history.history['val_accuracy'], label='Validation')\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.legend(loc='upper left')\n\nplt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='Train')\nplt.plot(history.history['val_loss'], label='Validation')\nplt.title('Model loss')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.legend(loc='upper left')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-16T15:15:09.996697Z","iopub.execute_input":"2024-06-16T15:15:09.998000Z","iopub.status.idle":"2024-06-16T15:17:05.270030Z","shell.execute_reply.started":"2024-06-16T15:15:09.997953Z","shell.execute_reply":"2024-06-16T15:17:05.268853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Code 3","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nimport cv2\nimport os\n\n# Load the CSV file\ndf = pd.read_csv('/kaggle/input/cassava-leaf-disease-classification/train.csv')\n\n# Sample a smaller subset for faster training (optional, for initial testing)\ndf_sample = df.sample(n=1000, random_state=42)\n\n# Define path to the images\nimage_path = '/kaggle/input/cassava-leaf-disease-classification/train_images'\n\n# Function to load and preprocess images with error handling\ndef load_images(image_ids, image_path, img_size=(128, 128)):\n    images = []\n    labels = []\n    missing_files = []\n    for image_id in image_ids:\n        img_path = os.path.join(image_path, image_id)\n        img = cv2.imread(img_path)\n        if img is not None:\n            img = cv2.resize(img, img_size)\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n            images.append(img)\n            label = df[df['image_id'] == image_id]['label'].values[0]\n            labels.append(label)\n        else:\n            missing_files.append(img_path)\n    if missing_files:\n        print(f\"Missing or unreadable files: {len(missing_files)}\")\n        for missing_file in missing_files[:10]:  # Print first 10 missing files as a sample\n            print(missing_file)\n    return np.array(images), np.array(labels)\n\n# Load and preprocess images\nX, y = load_images(df_sample['image_id'], image_path)\ny = df_sample['label'].values\n\n# Normalize pixel values\nX = X / 255.0\n\n# Encode labels\nlabel_encoder = LabelEncoder()\ny_encoded = label_encoder.fit_transform(y)\n\n# Split data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y_encoded, test_size=0.2, random_state=42)\n\n# Build the CNN model\nmodel = Sequential([\n    Conv2D(32, (3, 3), activation='relu', input_shape=(128, 128, 3)),\n    MaxPooling2D((2, 2)),\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Conv2D(128, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Flatten(),\n    Dense(512, activation='relu'),\n    Dropout(0.5),\n    Dense(5, activation='softmax')  # Assuming 5 classes for cassava leaf disease\n])\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\n# Train the model\nhistory = model.fit(X_train, y_train, epochs=10, validation_data=(X_test, y_test), batch_size=32)\n\n# Evaluate the model\nloss, accuracy = model.evaluate(X_test, y_test)\nprint(f'Accuracy: {accuracy:.4f}')\n\n# Save the model\nmodel.save('cassava_cnn_model.h5')\n\n# Plot training & validation accuracy and loss values\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(12, 4))\nplt.subplot(1, 2, 1)\nplt.plot(history.history['accuracy'], label='Train')\nplt.plot(history.history['val_accuracy'], label='Validation')\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.legend(loc='upper left')\n\nplt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='Train')\nplt.plot(history.history['val_loss'], label='Validation')\nplt.title('Model loss')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.legend(loc='upper left')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-16T15:17:52.093480Z","iopub.execute_input":"2024-06-16T15:17:52.094030Z","iopub.status.idle":"2024-06-16T15:21:05.505395Z","shell.execute_reply.started":"2024-06-16T15:17:52.093987Z","shell.execute_reply":"2024-06-16T15:21:05.504236Z"},"trusted":true},"execution_count":null,"outputs":[]}]}