{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#      print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-31T17:33:38.143251Z","iopub.execute_input":"2024-10-31T17:33:38.143546Z","iopub.status.idle":"2024-10-31T17:33:39.019891Z","shell.execute_reply.started":"2024-10-31T17:33:38.143513Z","shell.execute_reply":"2024-10-31T17:33:39.019132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np  \nimport pandas as pd \nimport seaborn as sns \nimport matplotlib.pyplot as plt \nfrom keras.models import Model, Sequential  \nfrom keras.layers import Dense, Input, Dropout, GlobalAveragePooling2D, Flatten, Conv2D, BatchNormalization, Activation, MaxPooling2D \nfrom keras.optimizers import Adam, SGD, RMSprop \nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.utils import shuffle\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score, classification_report\nimport cv2 \nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import models, layers, optimizers\nfrom tensorflow.keras.applications.efficientnet import EfficientNetB3\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator, load_img, img_to_array\nfrom tqdm import tqdm\nimport os             \nimport json          \nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:33:39.021455Z","iopub.execute_input":"2024-10-31T17:33:39.021857Z","iopub.status.idle":"2024-10-31T17:33:52.084139Z","shell.execute_reply.started":"2024-10-31T17:33:39.021823Z","shell.execute_reply":"2024-10-31T17:33:52.083170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv('../input/cassava-leaf-disease-classification/train.csv')\nfrom collections import Counter\nprint (Counter(data['label']))\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:33:52.085416Z","iopub.execute_input":"2024-10-31T17:33:52.085966Z","iopub.status.idle":"2024-10-31T17:33:52.144289Z","shell.execute_reply.started":"2024-10-31T17:33:52.085931Z","shell.execute_reply":"2024-10-31T17:33:52.143286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['label'].hist()","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:33:52.146875Z","iopub.execute_input":"2024-10-31T17:33:52.147525Z","iopub.status.idle":"2024-10-31T17:33:52.482140Z","shell.execute_reply.started":"2024-10-31T17:33:52.147489Z","shell.execute_reply":"2024-10-31T17:33:52.481045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"work_dir = '../input/cassava-leaf-disease-classification/'\nos.listdir(work_dir)\ntrain_path = '/kaggle/input/cassava-leaf-disease-classification/train_images'","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:33:52.483402Z","iopub.execute_input":"2024-10-31T17:33:52.483768Z","iopub.status.idle":"2024-10-31T17:33:52.489078Z","shell.execute_reply.started":"2024-10-31T17:33:52.483733Z","shell.execute_reply":"2024-10-31T17:33:52.487804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_df = pd.read_csv(work_dir+\"train.csv\")\ntraining_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:33:52.490888Z","iopub.execute_input":"2024-10-31T17:33:52.491273Z","iopub.status.idle":"2024-10-31T17:33:52.518431Z","shell.execute_reply.started":"2024-10-31T17:33:52.491213Z","shell.execute_reply":"2024-10-31T17:33:52.517430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:33:52.519672Z","iopub.execute_input":"2024-10-31T17:33:52.520025Z","iopub.status.idle":"2024-10-31T17:33:52.542442Z","shell.execute_reply.started":"2024-10-31T17:33:52.519983Z","shell.execute_reply":"2024-10-31T17:33:52.541380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:33:52.544102Z","iopub.execute_input":"2024-10-31T17:33:52.544617Z","iopub.status.idle":"2024-10-31T17:33:52.557666Z","shell.execute_reply.started":"2024-10-31T17:33:52.544566Z","shell.execute_reply":"2024-10-31T17:33:52.556602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file = open(work_dir + 'label_num_to_disease_map.json')\nreal_labels = json.load(file)\nreal_labels = {int(k):v for k,v in real_labels.items()}\n\ntraining_df['class name'] = training_df.label.map(real_labels)\nprint(training_df.head(10))\nprint(training_df['class name'].unique())","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:33:52.558832Z","iopub.execute_input":"2024-10-31T17:33:52.559118Z","iopub.status.idle":"2024-10-31T17:33:52.575485Z","shell.execute_reply.started":"2024-10-31T17:33:52.559086Z","shell.execute_reply":"2024-10-31T17:33:52.574599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"real_labels_list = list(real_labels.values())\nreal_labels","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:33:52.578880Z","iopub.execute_input":"2024-10-31T17:33:52.579142Z","iopub.status.idle":"2024-10-31T17:33:52.584944Z","shell.execute_reply.started":"2024-10-31T17:33:52.579112Z","shell.execute_reply":"2024-10-31T17:33:52.583970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count_of_classes = training_df['class name'].value_counts()\nprint(count_of_classes)","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:33:52.586277Z","iopub.execute_input":"2024-10-31T17:33:52.586668Z","iopub.status.idle":"2024-10-31T17:33:52.596094Z","shell.execute_reply.started":"2024-10-31T17:33:52.586626Z","shell.execute_reply":"2024-10-31T17:33:52.595218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(Counter(training_df['label']))\ntraining_df['label'].hist()","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:33:52.597088Z","iopub.execute_input":"2024-10-31T17:33:52.597428Z","iopub.status.idle":"2024-10-31T17:33:52.894445Z","shell.execute_reply.started":"2024-10-31T17:33:52.597387Z","shell.execute_reply":"2024-10-31T17:33:52.893498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = training_df[training_df.label == 0].sample(12) \nplt.figure(figsize=(16,12))\nfor ind, (image_id, label) in enumerate(zip(sample.image_id, sample.label)):\n    plt.subplot(4, 4, ind + 1)\n    image = cv2.imread(os.path.join(work_dir + \"train_images\", image_id))\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    plt.imshow(image)\n    plt.axis(\"off\")\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:33:52.895816Z","iopub.execute_input":"2024-10-31T17:33:52.896597Z","iopub.status.idle":"2024-10-31T17:33:54.549702Z","shell.execute_reply.started":"2024-10-31T17:33:52.896549Z","shell.execute_reply":"2024-10-31T17:33:54.548320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = training_df[training_df.label == 1].sample(12) \nplt.figure(figsize=(16,12))\nfor ind, (image_id, label) in enumerate(zip(sample.image_id, sample.label)):\n    plt.subplot(4, 4, ind + 1)\n    image = cv2.imread(os.path.join(work_dir + \"train_images\", image_id))\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    plt.imshow(image)\n    plt.axis(\"off\")\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:33:54.551057Z","iopub.execute_input":"2024-10-31T17:33:54.551369Z","iopub.status.idle":"2024-10-31T17:33:56.369843Z","shell.execute_reply.started":"2024-10-31T17:33:54.551336Z","shell.execute_reply":"2024-10-31T17:33:56.368469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = training_df[training_df.label == 2].sample(12) \nplt.figure(figsize=(16,12))\nfor ind, (image_id, label) in enumerate(zip(sample.image_id, sample.label)):\n    plt.subplot(4, 4, ind + 1)\n    image = cv2.imread(os.path.join(work_dir + \"train_images\", image_id))\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    plt.imshow(image)\n    plt.axis(\"off\")\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:33:56.371306Z","iopub.execute_input":"2024-10-31T17:33:56.371635Z","iopub.status.idle":"2024-10-31T17:33:57.958044Z","shell.execute_reply.started":"2024-10-31T17:33:56.371597Z","shell.execute_reply":"2024-10-31T17:33:57.956628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = training_df[training_df.label == 3].sample(12)\nplt.figure(figsize=(16,12))\nfor ind, (image_id, label) in enumerate(zip(sample.image_id, sample.label)):\n    plt.subplot(4, 4, ind + 1)\n    image = cv2.imread(os.path.join(work_dir + \"train_images\", image_id))\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    plt.imshow(image)\n    plt.axis(\"off\")\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:33:57.959405Z","iopub.execute_input":"2024-10-31T17:33:57.959750Z","iopub.status.idle":"2024-10-31T17:33:59.546443Z","shell.execute_reply.started":"2024-10-31T17:33:57.959716Z","shell.execute_reply":"2024-10-31T17:33:59.545291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = training_df[training_df.label == 4].sample(12)\nplt.figure(figsize=(16,12))\nfor ind, (image_id, label) in enumerate(zip(sample.image_id, sample.label)):\n    plt.subplot(4, 4, ind + 1)\n    image = cv2.imread(os.path.join(work_dir + \"train_images\", image_id))\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    plt.imshow(image)\n    plt.axis(\"off\")\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:33:59.547687Z","iopub.execute_input":"2024-10-31T17:33:59.548010Z","iopub.status.idle":"2024-10-31T17:34:01.188835Z","shell.execute_reply.started":"2024-10-31T17:33:59.547976Z","shell.execute_reply":"2024-10-31T17:34:01.187389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_new  = training_df","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:34:01.190169Z","iopub.execute_input":"2024-10-31T17:34:01.190531Z","iopub.status.idle":"2024-10-31T17:34:01.195062Z","shell.execute_reply.started":"2024-10-31T17:34:01.190494Z","shell.execute_reply":"2024-10-31T17:34:01.194145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_size = (64, 64)\ndef load_and_preprocess_image(img_path):\n    img = load_img(img_path, target_size = img_size)\n    img_array = img_to_array(img)\n    img_array = img_array / 255.0 \n    return img_array\nX = []\ny = []\nfor idx, row in training_df.iterrows():\n    img_path = os.path.join(work_dir, 'train_images', row['image_id'])\n    img_array = load_and_preprocess_image(img_path)\n    X.append(img_array)\n    y.append(row['label'])\nX = np.array(X)\ny = np.array(y)\nX = X.reshape(X.shape[0], -1)\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:34:01.196327Z","iopub.execute_input":"2024-10-31T17:34:01.196638Z","iopub.status.idle":"2024-10-31T17:37:50.013628Z","shell.execute_reply.started":"2024-10-31T17:34:01.196603Z","shell.execute_reply":"2024-10-31T17:37:50.012583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = LogisticRegression(n_jobs = -1,max_iter = 150)\nmodel.fit(X_train, y_train)\ny_pred = model.predict(X_test)\naccuracy = accuracy_score(y_test, y_pred)\nreport = classification_report(y_test, y_pred)\n\nprint(f'Accuracy: {accuracy}')\nprint('Classification Report:')\nprint(report)","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:37:50.014952Z","iopub.execute_input":"2024-10-31T17:37:50.015321Z","iopub.status.idle":"2024-10-31T17:39:40.978970Z","shell.execute_reply.started":"2024-10-31T17:37:50.015281Z","shell.execute_reply":"2024-10-31T17:39:40.977740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Batch_size = 64\nimg_height, img_width = 256, 256\ntraining_df['label'] = training_df['label'].astype('str') ","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:39:40.980957Z","iopub.execute_input":"2024-10-31T17:39:40.981617Z","iopub.status.idle":"2024-10-31T17:39:41.007022Z","shell.execute_reply.started":"2024-10-31T17:39:40.981565Z","shell.execute_reply":"2024-10-31T17:39:41.005752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_gen = ImageDataGenerator(\n    horizontal_flip = True,\n    vertical_flip = True,\n    validation_split = 0.2,\n)\n\ntrain_datagen = train_gen.flow_from_dataframe(\n    training_df,\n    directory = os.path.join(work_dir, \"train_images\"),\n    batch_size = Batch_size,\n    target_size = (img_height, img_width),\n    subset = \"training\",\n    seed = 42,\n    x_col = \"image_id\",\n    y_col = \"label\",\n    class_mode = \"categorical\"\n)\n\nval_gen = ImageDataGenerator(\n    validation_split = 0.2\n)\n\nval_datagen = val_gen.flow_from_dataframe(\n    training_df,\n    directory = os.path.join(work_dir, \"train_images\"),\n    batch_size = Batch_size,\n    target_size = (img_height, img_width),\n    subset = \"validation\",\n    seed = 42,\n    x_col = \"image_id\",\n    y_col = \"label\",\n    class_mode = \"categorical\"\n)","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:39:41.008635Z","iopub.execute_input":"2024-10-31T17:39:41.009354Z","iopub.status.idle":"2024-10-31T17:39:59.896804Z","shell.execute_reply.started":"2024-10-31T17:39:41.009304Z","shell.execute_reply":"2024-10-31T17:39:59.896034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_datagen),len(val_datagen)","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:39:59.897978Z","iopub.execute_input":"2024-10-31T17:39:59.898337Z","iopub.status.idle":"2024-10-31T17:39:59.905252Z","shell.execute_reply.started":"2024-10-31T17:39:59.898300Z","shell.execute_reply":"2024-10-31T17:39:59.904218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img, label = next(train_datagen)\n\nSteps_per_train = float(train_datagen.n) / train_datagen.batch_size\nSteps_per_val = float(val_datagen.n) / val_datagen.batch_size\n\nSteps_per_train, Steps_per_val","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:39:59.906471Z","iopub.execute_input":"2024-10-31T17:39:59.906775Z","iopub.status.idle":"2024-10-31T17:40:00.249126Z","shell.execute_reply.started":"2024-10-31T17:39:59.906743Z","shell.execute_reply":"2024-10-31T17:40:00.248212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model():\n    model = models.Sequential()\n    model.add(EfficientNetB3(include_top=False, weights='imagenet',\n                             input_shape=(img_height, img_width, 3), drop_connect_rate=0.3))\n    model.add(layers.GlobalAveragePooling2D())\n    model.add(layers.Flatten())\n    model.add(layers.Dense(256, activation=\"relu\"))\n    model.add(layers.Dropout(0.3))\n    model.add(layers.Dense(5, activation='softmax'))\n    \n    loss = tf.keras.losses.CategoricalCrossentropy(\n        label_smoothing=0.0001,\n        name='categorical_crossentropy'\n    )\n    optimizer = optimizers.Adam(learning_rate=1e-4)\n    \n    model.compile(optimizer=optimizer,\n                  loss=loss,\n                  metrics=[\"categorical_accuracy\"])\n    return model\n\nmodel = create_model()\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:40:00.250161Z","iopub.execute_input":"2024-10-31T17:40:00.250513Z","iopub.status.idle":"2024-10-31T17:40:03.614426Z","shell.execute_reply.started":"2024-10-31T17:40:00.250478Z","shell.execute_reply":"2024-10-31T17:40:03.613530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.build((None))\nprint('Our EfficientNet CNN has %d layers' %len(model.layers))","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:40:03.615627Z","iopub.execute_input":"2024-10-31T17:40:03.615931Z","iopub.status.idle":"2024-10-31T17:40:03.621005Z","shell.execute_reply.started":"2024-10-31T17:40:03.615898Z","shell.execute_reply":"2024-10-31T17:40:03.620160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rlronp=tf.keras.callbacks.ReduceLROnPlateau(monitor=\"val_loss\",\n                                            factor=0.2,\n                                            mode = \"min\",\n                                            min_lr=1e-6,\n                                            patience=2, \n                                            verbose=1)\n\nestop=tf.keras.callbacks.EarlyStopping(monitor=\"val_loss\", \n                                       mode= \"min\",\n                                       patience=3, \n                                       verbose=1,\n                                       restore_best_weights=True)\n\nhistory = model.fit(\n    train_datagen,\n    steps_per_epoch=int(Steps_per_train),\n    epochs=5,\n    verbose =1,\n    validation_data=val_datagen,\n    validation_steps=int(Steps_per_val),\n    callbacks=[rlronp, estop]\n)","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:40:03.622316Z","iopub.execute_input":"2024-10-31T17:40:03.622620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_acc = history.history[\"categorical_accuracy\"]\nval_acc = history.history[\"val_categorical_accuracy\"]\nepochs = range(1, len(train_acc)+1)\nplt.plot(epochs, train_acc, \"bo\", label = \"Training Accuracy\")\nplt.plot(epochs, val_acc, \"b\", label = \"Validation Accuracy\")\nplt.title(\"Training and Validation Accuracy\")\nplt.legend()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8,6))\ntrain_loss = history.history[\"loss\"]\nval_loss = history.history[\"val_loss\"]\nepochs = range(1, len(train_loss)+1)\nplt.plot(epochs, train_loss, \"bo\", label = \"Training Loss\")\nplt.plot(epochs, val_loss, \"b\", label = \"Validation Loss\")\nplt.title(\"Training and Validation Loss\")\nplt.legend()","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}