{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\nfrom sklearn.metrics import accuracy_score, classification_report\nimport cv2\nimport os","metadata":{"execution":{"iopub.status.busy":"2024-06-16T03:31:26.604299Z","iopub.execute_input":"2024-06-16T03:31:26.604798Z","iopub.status.idle":"2024-06-16T03:31:29.631886Z","shell.execute_reply.started":"2024-06-16T03:31:26.604756Z","shell.execute_reply":"2024-06-16T03:31:29.630224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the CSV file\ndf = pd.read_csv('/kaggle/input/cassava-leaf-disease-classification/train.csv')\n\n# Sample a smaller subset for faster training\ndf_sample = df.sample(n=1000, random_state=42)\n\n# Define path to the images\nimage_path = '/kaggle/input/cassava-leaf-disease-classification/train_images'","metadata":{"execution":{"iopub.status.busy":"2024-06-16T03:32:04.044470Z","iopub.execute_input":"2024-06-16T03:32:04.045156Z","iopub.status.idle":"2024-06-16T03:32:04.096451Z","shell.execute_reply.started":"2024-06-16T03:32:04.045111Z","shell.execute_reply":"2024-06-16T03:32:04.094962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to load and preprocess images with error handling\ndef load_images(image_ids, image_path, img_size=(64, 64)):\n    images = []\n    missing_files = []\n    for image_id in image_ids:\n        img_path = os.path.join(image_path, image_id)\n        img = cv2.imread(img_path)\n        if img is not None:\n            img = cv2.resize(img, img_size)\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n            images.append(img)\n        else:\n            missing_files.append(img_path)\n    if missing_files:\n        print(f\"Missing or unreadable files: {len(missing_files)}\")\n        for missing_file in missing_files[:10]:  # Print first 10 missing files as a sample\n            print(missing_file)\n    return np.array(images)","metadata":{"execution":{"iopub.status.busy":"2024-06-16T03:32:22.674378Z","iopub.execute_input":"2024-06-16T03:32:22.674870Z","iopub.status.idle":"2024-06-16T03:32:22.805149Z","shell.execute_reply.started":"2024-06-16T03:32:22.674834Z","shell.execute_reply":"2024-06-16T03:32:22.803839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load and preprocess images\nX = load_images(df_sample['image_id'], image_path)\ny = df_sample['label']\n\n# Normalize pixel values\nX = X / 255.0\n\n# Flatten images for logistic regression\nX_flattened = X.reshape(X.shape[0], -1)\n\n# Encode labels\nlabel_encoder = LabelEncoder()\ny_encoded = label_encoder.fit_transform(y)\n\n# Split data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X_flattened, y_encoded, test_size=0.2, random_state=42)\n\n# Standardize features\nscaler = StandardScaler()\nX_train = scaler.fit_transform(X_train)\nX_test = scaler.transform(X_test)\n\n# Train logistic regression model\nlog_reg = LogisticRegression(multi_class='ovr', max_iter=100, solver='saga', tol=1e-2)\nlog_reg.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-06-16T03:32:45.674712Z","iopub.execute_input":"2024-06-16T03:32:45.675202Z","iopub.status.idle":"2024-06-16T03:33:42.769562Z","shell.execute_reply.started":"2024-06-16T03:32:45.675162Z","shell.execute_reply":"2024-06-16T03:33:42.768264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate model\ny_pred = log_reg.predict(X_test)\naccuracy = accuracy_score(y_test, y_pred)\nprint(f'Accuracy: {accuracy:.4f}')\n\n# Convert numerical labels to strings for classification report\ntarget_names = [str(i) for i in label_encoder.classes_]\nprint(classification_report(y_test, y_pred, target_names=target_names))","metadata":{"execution":{"iopub.status.busy":"2024-06-16T03:34:26.874265Z","iopub.execute_input":"2024-06-16T03:34:26.874751Z","iopub.status.idle":"2024-06-16T03:34:26.910236Z","shell.execute_reply.started":"2024-06-16T03:34:26.874716Z","shell.execute_reply":"2024-06-16T03:34:26.908827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Flatten\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nimport cv2\nimport os","metadata":{"execution":{"iopub.status.busy":"2024-06-16T03:34:42.529458Z","iopub.execute_input":"2024-06-16T03:34:42.530567Z","iopub.status.idle":"2024-06-16T03:34:56.025309Z","shell.execute_reply.started":"2024-06-16T03:34:42.530522Z","shell.execute_reply":"2024-06-16T03:34:56.024378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the CSV file\ndf = pd.read_csv('/kaggle/input/cassava-leaf-disease-classification/train.csv')\n\n# Sample a smaller subset for faster training (optional, for initial testing)\ndf_sample = df.sample(n=1000, random_state=42)\n\n# Define path to the images\nimage_path = '/kaggle/input/cassava-leaf-disease-classification/train_images'","metadata":{"execution":{"iopub.status.busy":"2024-06-16T03:34:56.673552Z","iopub.execute_input":"2024-06-16T03:34:56.674222Z","iopub.status.idle":"2024-06-16T03:34:56.698808Z","shell.execute_reply.started":"2024-06-16T03:34:56.674189Z","shell.execute_reply":"2024-06-16T03:34:56.697577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to load and preprocess images with error handling\ndef load_images(image_ids, image_path, img_size=(128, 128)):\n    images = []\n    labels = []\n    missing_files = []\n    for image_id in image_ids:\n        img_path = os.path.join(image_path, image_id)\n        img = cv2.imread(img_path)\n        if img is not None:\n            img = cv2.resize(img, img_size)\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n            images.append(img)\n            label = df[df['image_id'] == image_id]['label'].values[0]\n            labels.append(label)\n        else:\n            missing_files.append(img_path)\n    if missing_files:\n        print(f\"Missing or unreadable files: {len(missing_files)}\")\n        for missing_file in missing_files[:10]:  # Print first 10 missing files as a sample\n            print(missing_file)\n    return np.array(images), np.array(labels)","metadata":{"execution":{"iopub.status.busy":"2024-06-16T03:35:13.875188Z","iopub.execute_input":"2024-06-16T03:35:13.876102Z","iopub.status.idle":"2024-06-16T03:35:13.885738Z","shell.execute_reply.started":"2024-06-16T03:35:13.876045Z","shell.execute_reply":"2024-06-16T03:35:13.884463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load and preprocess images\nX, y = load_images(df_sample['image_id'], image_path)\ny = df_sample['label'].values\n\n# Normalize pixel values\nX = X / 255.0\n\n# Encode labels\nlabel_encoder = LabelEncoder()\ny_encoded = label_encoder.fit_transform(y)\n\n# Split data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y_encoded, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-06-16T03:36:12.719630Z","iopub.execute_input":"2024-06-16T03:36:12.720107Z","iopub.status.idle":"2024-06-16T03:36:20.939553Z","shell.execute_reply.started":"2024-06-16T03:36:12.720071Z","shell.execute_reply":"2024-06-16T03:36:20.938345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build the feedforward neural network model\nmodel = Sequential([\n    Flatten(input_shape=(128, 128, 3)),  # Flattening the 128x128 RGB images\n    Dense(512, activation='relu'),\n    Dense(256, activation='relu'),\n    Dense(128, activation='relu'),\n    Dense(5, activation='softmax')  # Assuming 5 classes for cassava leaf disease\n])\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\n# Train the model\nhistory = model.fit(X_train, y_train, epochs=10, validation_data=(X_test, y_test), batch_size=32)","metadata":{"execution":{"iopub.status.busy":"2024-06-16T03:36:42.994682Z","iopub.execute_input":"2024-06-16T03:36:42.995145Z","iopub.status.idle":"2024-06-16T03:38:03.142704Z","shell.execute_reply.started":"2024-06-16T03:36:42.995108Z","shell.execute_reply":"2024-06-16T03:38:03.141458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model\nloss, accuracy = model.evaluate(X_test, y_test)\nprint(f'Accuracy: {accuracy:.4f}')\n\n# Save the model\nmodel.save('cassava_fnn_model.h5')","metadata":{"execution":{"iopub.status.busy":"2024-06-16T03:38:44.394776Z","iopub.execute_input":"2024-06-16T03:38:44.395224Z","iopub.status.idle":"2024-06-16T03:38:45.137469Z","shell.execute_reply.started":"2024-06-16T03:38:44.395190Z","shell.execute_reply":"2024-06-16T03:38:45.136432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot training & validation accuracy and loss values\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(12, 4))\nplt.subplot(1, 2, 1)\nplt.plot(history.history['accuracy'], label='Train')\nplt.plot(history.history['val_accuracy'], label='Validation')\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.legend(loc='upper left')\n\nplt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='Train')\nplt.plot(history.history['val_loss'], label='Validation')\nplt.title('Model loss')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.legend(loc='upper left')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-16T03:39:06.290209Z","iopub.execute_input":"2024-06-16T03:39:06.291211Z","iopub.status.idle":"2024-06-16T03:39:06.917837Z","shell.execute_reply.started":"2024-06-16T03:39:06.291159Z","shell.execute_reply":"2024-06-16T03:39:06.916221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nimport cv2\nimport os","metadata":{"execution":{"iopub.status.busy":"2024-06-16T03:39:25.673924Z","iopub.execute_input":"2024-06-16T03:39:25.674863Z","iopub.status.idle":"2024-06-16T03:39:25.684754Z","shell.execute_reply.started":"2024-06-16T03:39:25.674821Z","shell.execute_reply":"2024-06-16T03:39:25.683529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the CSV file\ndf = pd.read_csv('/kaggle/input/cassava-leaf-disease-classification/train.csv')\n\n# Sample a smaller subset for faster training (optional, for initial testing)\ndf_sample = df.sample(n=1000, random_state=42)\n\n# Define path to the images\nimage_path = '/kaggle/input/cassava-leaf-disease-classification/train_images'","metadata":{"execution":{"iopub.status.busy":"2024-06-16T03:39:40.495264Z","iopub.execute_input":"2024-06-16T03:39:40.495791Z","iopub.status.idle":"2024-06-16T03:39:40.520253Z","shell.execute_reply.started":"2024-06-16T03:39:40.495750Z","shell.execute_reply":"2024-06-16T03:39:40.519103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to load and preprocess images with error handling\ndef load_images(image_ids, image_path, img_size=(128, 128)):\n    images = []\n    labels = []\n    missing_files = []\n    for image_id in image_ids:\n        img_path = os.path.join(image_path, image_id)\n        img = cv2.imread(img_path)\n        if img is not None:\n            img = cv2.resize(img, img_size)\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n            images.append(img)\n            label = df[df['image_id'] == image_id]['label'].values[0]\n            labels.append(label)\n        else:\n            missing_files.append(img_path)\n    if missing_files:\n        print(f\"Missing or unreadable files: {len(missing_files)}\")\n        for missing_file in missing_files[:10]:  # Print first 10 missing files as a sample\n            print(missing_file)\n    return np.array(images), np.array(labels)","metadata":{"execution":{"iopub.status.busy":"2024-06-16T03:39:56.979065Z","iopub.execute_input":"2024-06-16T03:39:56.980921Z","iopub.status.idle":"2024-06-16T03:39:56.993782Z","shell.execute_reply.started":"2024-06-16T03:39:56.980855Z","shell.execute_reply":"2024-06-16T03:39:56.992127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Load and preprocess images\nX, y = load_images(df_sample['image_id'], image_path)\ny = df_sample['label'].values\n\n# Normalize pixel values\nX = X / 255.0\n\n# Encode labels\nlabel_encoder = LabelEncoder()\ny_encoded = label_encoder.fit_transform(y)\n\n# Split data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y_encoded, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-06-16T03:40:17.554772Z","iopub.execute_input":"2024-06-16T03:40:17.555253Z","iopub.status.idle":"2024-06-16T03:40:26.028594Z","shell.execute_reply.started":"2024-06-16T03:40:17.555216Z","shell.execute_reply":"2024-06-16T03:40:26.027437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build the CNN model\nmodel = Sequential([\n    Conv2D(32, (3, 3), activation='relu', input_shape=(128, 128, 3)),\n    MaxPooling2D((2, 2)),\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Conv2D(128, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Flatten(),\n    Dense(512, activation='relu'),\n    Dropout(0.5),\n    Dense(5, activation='softmax')  # Assuming 5 classes for cassava leaf disease\n])\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\n# Train the model\nhistory = model.fit(X_train, y_train, epochs=10, validation_data=(X_test, y_test), batch_size=32)","metadata":{"execution":{"iopub.status.busy":"2024-06-16T03:40:49.474858Z","iopub.execute_input":"2024-06-16T03:40:49.475307Z","iopub.status.idle":"2024-06-16T03:43:51.292114Z","shell.execute_reply.started":"2024-06-16T03:40:49.475274Z","shell.execute_reply":"2024-06-16T03:43:51.290783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model\nloss, accuracy = model.evaluate(X_test, y_test)\nprint(f'Accuracy: {accuracy:.4f}')\n\n# Save the model\nmodel.save('cassava_cnn_model.h5')","metadata":{"execution":{"iopub.status.busy":"2024-06-16T03:44:16.115079Z","iopub.execute_input":"2024-06-16T03:44:16.115602Z","iopub.status.idle":"2024-06-16T03:44:17.348875Z","shell.execute_reply.started":"2024-06-16T03:44:16.115559Z","shell.execute_reply":"2024-06-16T03:44:17.347278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot training & validation accuracy and loss values\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(12, 4))\nplt.subplot(1, 2, 1)\nplt.plot(history.history['accuracy'], label='Train')\nplt.plot(history.history['val_accuracy'], label='Validation')\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.legend(loc='upper left')\n\nplt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='Train')\nplt.plot(history.history['val_loss'], label='Validation')\nplt.title('Model loss')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.legend(loc='upper left')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-16T03:44:38.169671Z","iopub.execute_input":"2024-06-16T03:44:38.170124Z","iopub.status.idle":"2024-06-16T03:44:38.687589Z","shell.execute_reply.started":"2024-06-16T03:44:38.170088Z","shell.execute_reply":"2024-06-16T03:44:38.686172Z"},"trusted":true},"execution_count":null,"outputs":[]}]}