{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport os\nimport csv\nimport json\nimport matplotlib.pyplot as plt\nfrom keras.models import Model, Sequential\nfrom keras.layers import Dense, Input, Dropout, GlobalAveragePooling2D, Flatten, Conv2D, BatchNormalization, Activation, MaxPooling2D\nfrom keras.optimizers import Adam, SGD, RMSprop\nfrom tensorflow.keras import models, layers, optimizers\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator, load_img, img_to_array\nfrom tensorflow.keras.applications.efficientnet import EfficientNetB3\nfrom tqdm import tqdm\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.utils import shuffle                      \nimport cv2                                \nimport tensorflow as tf\nimport warnings\nwarnings.simplefilter (\"ignore\")\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2024-06-22T03:22:55.713811Z","iopub.execute_input":"2024-06-22T03:22:55.714567Z","iopub.status.idle":"2024-06-22T03:22:55.726263Z","shell.execute_reply.started":"2024-06-22T03:22:55.714529Z","shell.execute_reply":"2024-06-22T03:22:55.725072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv('/kaggle/input/cassava-leaf-disease-classification/train.csv')\ndata","metadata":{"execution":{"iopub.status.busy":"2024-06-22T03:14:49.114313Z","iopub.execute_input":"2024-06-22T03:14:49.114986Z","iopub.status.idle":"2024-06-22T03:14:49.168713Z","shell.execute_reply.started":"2024-06-22T03:14:49.114947Z","shell.execute_reply":"2024-06-22T03:14:49.167673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#creating the main directory\nDir='/kaggle/input/cassava-leaf-disease-classification'\nos.listdir(Dir)","metadata":{"execution":{"iopub.status.busy":"2024-06-22T03:15:15.868619Z","iopub.execute_input":"2024-06-22T03:15:15.869396Z","iopub.status.idle":"2024-06-22T03:15:15.877111Z","shell.execute_reply.started":"2024-06-22T03:15:15.869358Z","shell.execute_reply":"2024-06-22T03:15:15.875947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#target label\nwith open('/kaggle/input/cassava-leaf-disease-classification/label_num_to_disease_map.json') as f:\n    json_data=json.loads(f.read())\n    json_data={int(i):v for i,v in json_data.items()}\njson_data","metadata":{"execution":{"iopub.status.busy":"2024-06-22T03:17:26.429956Z","iopub.execute_input":"2024-06-22T03:17:26.430371Z","iopub.status.idle":"2024-06-22T03:17:26.439686Z","shell.execute_reply.started":"2024-06-22T03:17:26.430338Z","shell.execute_reply":"2024-06-22T03:17:26.438607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#map the target label name with the data\ndata['label_name']=data['label'].map(json_data)\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-22T03:17:34.020188Z","iopub.execute_input":"2024-06-22T03:17:34.020570Z","iopub.status.idle":"2024-06-22T03:17:34.037219Z","shell.execute_reply.started":"2024-06-22T03:17:34.020541Z","shell.execute_reply":"2024-06-22T03:17:34.035952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Analysis","metadata":{}},{"cell_type":"code","source":"def plot_img(class_id,label,total_imgs):\n    #get image id\n    plot_list=data[data['label']==class_id].sample(total_imgs)['image_id'].tolist()\n    labels=[label for i in range(total_imgs)]\n    size=int(np.sqrt(total_imgs))\n    if size*size<total_imgs:\n        size+=1\n        \n    #set fig size\n    plt.figure(figsize=(30,30))\n    \n    #plot img\n    for index,(image_id,label) in enumerate(zip(plot_list,labels)):\n        plt.subplot(size,size,index+1)\n        img=Image.open(str('/kaggle/input/cassava-leaf-disease-classification/train_images/'+image_id))\n        plt.imshow(img)\n        plt.title(label,fontsize=20)\n        plt.axis('off')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T13:50:00.914177Z","iopub.execute_input":"2024-06-21T13:50:00.914489Z","iopub.status.idle":"2024-06-21T13:50:00.923187Z","shell.execute_reply.started":"2024-06-21T13:50:00.914461Z","shell.execute_reply":"2024-06-21T13:50:00.922116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#for class 0,similarly can be done for other classes\nplot_img(0,json_data[0],6)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T13:50:00.924466Z","iopub.execute_input":"2024-06-21T13:50:00.925121Z","iopub.status.idle":"2024-06-21T13:50:04.159844Z","shell.execute_reply.started":"2024-06-21T13:50:00.925086Z","shell.execute_reply":"2024-06-21T13:50:04.158126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#storing a copy in temp to be used in cnn model\ntemp=data\ntemp.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T13:50:04.161540Z","iopub.execute_input":"2024-06-21T13:50:04.162185Z","iopub.status.idle":"2024-06-21T13:50:04.182858Z","shell.execute_reply.started":"2024-06-21T13:50:04.162106Z","shell.execute_reply":"2024-06-21T13:50:04.181515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#count of each class label\ncount = data['label_name'].value_counts()\nprint(count)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T13:50:04.187227Z","iopub.execute_input":"2024-06-21T13:50:04.187683Z","iopub.status.idle":"2024-06-21T13:50:04.211171Z","shell.execute_reply.started":"2024-06-21T13:50:04.187644Z","shell.execute_reply":"2024-06-21T13:50:04.208834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(y=\"label_name\",data=data)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T13:50:04.212998Z","iopub.execute_input":"2024-06-21T13:50:04.213429Z","iopub.status.idle":"2024-06-21T13:50:04.517154Z","shell.execute_reply.started":"2024-06-21T13:50:04.213357Z","shell.execute_reply":"2024-06-21T13:50:04.515905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Logistic Regression","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nfrom sklearn.utils import shuffle\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score, classification_report","metadata":{"execution":{"iopub.status.busy":"2024-06-21T13:50:04.518798Z","iopub.execute_input":"2024-06-21T13:50:04.519158Z","iopub.status.idle":"2024-06-21T13:50:04.665572Z","shell.execute_reply.started":"2024-06-21T13:50:04.519130Z","shell.execute_reply":"2024-06-21T13:50:04.664403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_size = (50,50)\ndef process_img(img_path):\n    img = load_img(img_path, target_size = img_size)\n    img_array = img_to_array(img)\n    # Normalize pixel values\n    img_array = img_array / 255.0  \n    return img_array\nprint('function created')","metadata":{"execution":{"iopub.status.busy":"2024-06-21T13:50:04.667122Z","iopub.execute_input":"2024-06-21T13:50:04.667753Z","iopub.status.idle":"2024-06-21T13:50:04.677114Z","shell.execute_reply.started":"2024-06-21T13:50:04.667703Z","shell.execute_reply":"2024-06-21T13:50:04.675369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = []\ny = []\nprint('started')\nfor ind, row in data.iterrows():\n    img_path = os.path.join('/kaggle/input/cassava-leaf-disease-classification/train_images', row['image_id'])\n    img_array = process_img(img_path)\n    x.append(img_array)\n    y.append(row['label'])\nprint('done')","metadata":{"execution":{"iopub.status.busy":"2024-06-21T13:55:21.516444Z","iopub.execute_input":"2024-06-21T13:55:21.517468Z","iopub.status.idle":"2024-06-21T13:58:44.905393Z","shell.execute_reply.started":"2024-06-21T13:55:21.517415Z","shell.execute_reply":"2024-06-21T13:58:44.904126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#splitting of dataset\nx = np.array(x)\ny = np.array(y)\nx = x.reshape(x.shape[0], -1)\nx_train, x_test, y_train, y_test = train_test_split(x, y, test_size=0.2, random_state=42, stratify=y)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T13:59:48.063695Z","iopub.execute_input":"2024-06-21T13:59:48.064873Z","iopub.status.idle":"2024-06-21T13:59:48.673452Z","shell.execute_reply.started":"2024-06-21T13:59:48.064833Z","shell.execute_reply":"2024-06-21T13:59:48.672307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(x_train))\nprint(len(x_test))\nprint(len(y_train))\nprint(len(y_test))","metadata":{"execution":{"iopub.status.busy":"2024-06-21T13:59:53.254764Z","iopub.execute_input":"2024-06-21T13:59:53.255269Z","iopub.status.idle":"2024-06-21T13:59:53.263715Z","shell.execute_reply.started":"2024-06-21T13:59:53.255230Z","shell.execute_reply":"2024-06-21T13:59:53.261997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#fitting in the model\nmodel = LogisticRegression(max_iter = 100)\nmodel.fit(x_train, y_train)\ny_pred = model.predict(x_test)\naccuracy = accuracy_score(y_test, y_pred)\nreport = classification_report(y_test, y_pred)\nprint(f'Accuracy: {accuracy}')\nprint(report)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T13:59:58.484179Z","iopub.execute_input":"2024-06-21T13:59:58.484661Z","iopub.status.idle":"2024-06-21T14:00:41.310079Z","shell.execute_reply.started":"2024-06-21T13:59:58.484625Z","shell.execute_reply":"2024-06-21T14:00:41.308691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CNN","metadata":{}},{"cell_type":"code","source":"#train_images\nimage_size=100\ntrain_img='/kaggle/input/cassava-leaf-disease-classification/train_images'\nfor image in tqdm(os.listdir(train_img)): \n    path = os.path.join(train_img, image)\n    img = cv2.imread(path, cv2.IMREAD_GRAYSCALE) \n    img = cv2.resize(img, (image_size, image_size)).flatten()   \n    np_img_1=np.asarray(img)\n\nplt.figure(figsize=(10,10))\nplt.subplot(1, 2, 1)\nplt.imshow(np_img_1.reshape(image_size, image_size))\nplt.axis('off')","metadata":{"execution":{"iopub.status.busy":"2024-06-21T14:07:23.391910Z","iopub.execute_input":"2024-06-21T14:07:23.392654Z","iopub.status.idle":"2024-06-21T14:08:49.770642Z","shell.execute_reply.started":"2024-06-21T14:07:23.392576Z","shell.execute_reply":"2024-06-21T14:08:49.769327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test images\nimage_size=100\ntest_img='/kaggle/input/cassava-leaf-disease-classification/test_images'\nfor image in tqdm(os.listdir(train_img)): \n    path = os.path.join(train_img, image)\n    img = cv2.imread(path, cv2.IMREAD_GRAYSCALE) \n    img = cv2.resize(img, (image_size, image_size)).flatten()   \n    np_img_2=np.asarray(img)\n\nplt.figure(figsize=(10,10))\nplt.subplot(1, 2, 1)\nplt.imshow(np_img_2.reshape(image_size, image_size))\nplt.axis('off')","metadata":{"execution":{"iopub.status.busy":"2024-06-21T14:09:01.458124Z","iopub.execute_input":"2024-06-21T14:09:01.458727Z","iopub.status.idle":"2024-06-21T14:10:39.535703Z","shell.execute_reply.started":"2024-06-21T14:09:01.458694Z","shell.execute_reply":"2024-06-21T14:10:39.533658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_train = data[\"label\"]\nX_train = data.drop(labels = [\"label\"],axis = 1)","metadata":{"execution":{"iopub.status.busy":"2024-06-22T03:18:07.302225Z","iopub.execute_input":"2024-06-22T03:18:07.302627Z","iopub.status.idle":"2024-06-22T03:18:07.312453Z","shell.execute_reply.started":"2024-06-22T03:18:07.302596Z","shell.execute_reply":"2024-06-22T03:18:07.311218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_val, Y_train, Y_val = train_test_split(X_train, Y_train, test_size = 0.1, random_state=2)\nprint(\"x_train shape\",X_train.shape)\nprint(\"x_test shape\",X_val.shape)\nprint(\"y_train shape\",Y_train.shape)\nprint(\"y_test shape\",Y_val.shape)","metadata":{"execution":{"iopub.status.busy":"2024-06-22T03:18:11.397354Z","iopub.execute_input":"2024-06-22T03:18:11.397725Z","iopub.status.idle":"2024-06-22T03:18:11.424322Z","shell.execute_reply.started":"2024-06-22T03:18:11.397694Z","shell.execute_reply":"2024-06-22T03:18:11.423153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train","metadata":{"execution":{"iopub.status.busy":"2024-06-22T03:18:15.675860Z","iopub.execute_input":"2024-06-22T03:18:15.676255Z","iopub.status.idle":"2024-06-22T03:18:15.688438Z","shell.execute_reply.started":"2024-06-22T03:18:15.676223Z","shell.execute_reply":"2024-06-22T03:18:15.687302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_train.unique()","metadata":{"execution":{"iopub.status.busy":"2024-06-22T03:18:23.618688Z","iopub.execute_input":"2024-06-22T03:18:23.619043Z","iopub.status.idle":"2024-06-22T03:18:23.628133Z","shell.execute_reply.started":"2024-06-22T03:18:23.619015Z","shell.execute_reply":"2024-06-22T03:18:23.626983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['label'] = data['label'].astype('str')","metadata":{"execution":{"iopub.status.busy":"2024-06-22T03:18:27.759853Z","iopub.execute_input":"2024-06-22T03:18:27.760266Z","iopub.status.idle":"2024-06-22T03:18:27.772417Z","shell.execute_reply.started":"2024-06-22T03:18:27.760232Z","shell.execute_reply":"2024-06-22T03:18:27.771155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(Y_train)","metadata":{"execution":{"iopub.status.busy":"2024-06-22T03:18:31.022683Z","iopub.execute_input":"2024-06-22T03:18:31.023047Z","iopub.status.idle":"2024-06-22T03:18:31.030044Z","shell.execute_reply.started":"2024-06-22T03:18:31.023018Z","shell.execute_reply":"2024-06-22T03:18:31.028820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = 10 \nbatch_size = 100\nimg_height, img_width = 256, 256\ntrain_gen = ImageDataGenerator(horizontal_flip = True,vertical_flip = True,validation_split = 0.2,)\ntrain_datagen = train_gen.flow_from_dataframe(data,directory = os.path.join(Dir, \"train_images\"),batch_size = batch_size,target_size = (img_height, img_width),subset = \"training\",seed = 42,x_col = \"image_id\",y_col = \"label\",class_mode = \"categorical\")","metadata":{"execution":{"iopub.status.busy":"2024-06-22T03:19:23.731576Z","iopub.execute_input":"2024-06-22T03:19:23.731977Z","iopub.status.idle":"2024-06-22T03:20:08.562362Z","shell.execute_reply.started":"2024-06-22T03:19:23.731946Z","shell.execute_reply":"2024-06-22T03:20:08.561000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_gen = ImageDataGenerator(validation_split = 0.2)\nval_datagen = val_gen.flow_from_dataframe(data,directory = os.path.join(Dir, \"train_images\"),batch_size = batch_size,target_size = (img_height, img_width),subset = \"validation\",seed = 42,x_col = \"image_id\",y_col = \"label\",class_mode = \"categorical\")","metadata":{"execution":{"iopub.status.busy":"2024-06-22T03:20:12.599221Z","iopub.execute_input":"2024-06-22T03:20:12.599611Z","iopub.status.idle":"2024-06-22T03:20:21.501420Z","shell.execute_reply.started":"2024-06-22T03:20:12.599572Z","shell.execute_reply":"2024-06-22T03:20:21.500183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img, label = next(train_datagen)","metadata":{"execution":{"iopub.status.busy":"2024-06-22T03:20:26.335813Z","iopub.execute_input":"2024-06-22T03:20:26.336189Z","iopub.status.idle":"2024-06-22T03:20:27.482010Z","shell.execute_reply.started":"2024-06-22T03:20:26.336159Z","shell.execute_reply":"2024-06-22T03:20:27.481021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Steps_per_train = float(train_datagen.n) / train_datagen.batch_size\nSteps_per_val = float(val_datagen.n) / val_datagen.batch_size","metadata":{"execution":{"iopub.status.busy":"2024-06-22T03:20:32.629976Z","iopub.execute_input":"2024-06-22T03:20:32.631118Z","iopub.status.idle":"2024-06-22T03:20:32.635729Z","shell.execute_reply.started":"2024-06-22T03:20:32.631076Z","shell.execute_reply":"2024-06-22T03:20:32.634472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model():\n    model = models.Sequential()\n    model.add(EfficientNetB3(include_top=False, weights='imagenet',\n                             input_shape=(img_height, img_width, 3), drop_connect_rate=0.3))\n    model.add(layers.GlobalAveragePooling2D())\n    model.add(layers.Flatten())\n    model.add(layers.Dense(256, activation=\"relu\"))\n    model.add(layers.Dropout(0.3))\n    model.add(layers.Dense(5, activation='softmax'))\n    \n    loss = tf.keras.losses.CategoricalCrossentropy(\n        label_smoothing=0.0001,\n        name='categorical_crossentropy'\n    )\n    optimizer = optimizers.Adam(learning_rate=1e-4)\n    \n    model.compile(optimizer=optimizer,\n                  loss=loss,\n                  metrics=[\"categorical_accuracy\"])\n    return model\n\nmodel = create_model()\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-06-22T03:23:08.485925Z","iopub.execute_input":"2024-06-22T03:23:08.486327Z","iopub.status.idle":"2024-06-22T03:23:13.799707Z","shell.execute_reply.started":"2024-06-22T03:23:08.486293Z","shell.execute_reply":"2024-06-22T03:23:13.798597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.build((None))","metadata":{"execution":{"iopub.status.busy":"2024-06-22T03:23:19.210103Z","iopub.execute_input":"2024-06-22T03:23:19.210489Z","iopub.status.idle":"2024-06-22T03:23:19.215773Z","shell.execute_reply.started":"2024-06-22T03:23:19.210459Z","shell.execute_reply":"2024-06-22T03:23:19.214671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rlronp=tf.keras.callbacks.ReduceLROnPlateau(monitor=\"val_loss\",\n                                            factor=0.2,\n                                            mode = \"min\",\n                                            min_lr=1e-6,\n                                            patience=2, \n                                            verbose=1)\n\nestop=tf.keras.callbacks.EarlyStopping(monitor=\"val_loss\", \n                                       mode= \"min\",\n                                       patience=3, \n                                       verbose=1,\n                                       restore_best_weights=True)\n\nhistory = model.fit(\n    train_datagen,\n    steps_per_epoch=int(Steps_per_train),\n    epochs=5,\n    verbose =1,\n    validation_data=val_datagen,\n    validation_steps=int(Steps_per_val),\n    callbacks=[rlronp, estop]\n)","metadata":{"execution":{"iopub.status.busy":"2024-06-22T03:23:22.237559Z","iopub.execute_input":"2024-06-22T03:23:22.237925Z","iopub.status.idle":"2024-06-22T11:09:05.958823Z","shell.execute_reply.started":"2024-06-22T03:23:22.237896Z","shell.execute_reply":"2024-06-22T11:09:05.954346Z"},"trusted":true},"execution_count":null,"outputs":[]}]}