{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Importing Libraries","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm\nimport pandas as pd\nimport random as rn\nimport numpy as np\nimport json\nimport cv2\nimport os","metadata":{"execution":{"iopub.status.busy":"2023-05-21T06:02:36.038769Z","iopub.execute_input":"2023-05-21T06:02:36.039172Z","iopub.status.idle":"2023-05-21T06:02:36.958155Z","shell.execute_reply.started":"2023-05-21T06:02:36.039132Z","shell.execute_reply":"2023-05-21T06:02:36.957237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.options.mode.chained_assignment = None\n\nfrom IPython.core.interactiveshell import InteractiveShell   \nInteractiveShell.ast_node_interactivity = \"all\"\n\nimport tensorflow as tf\n","metadata":{"execution":{"iopub.status.busy":"2023-05-21T06:02:41.669853Z","iopub.execute_input":"2023-05-21T06:02:41.670483Z","iopub.status.idle":"2023-05-21T06:02:49.404236Z","shell.execute_reply.started":"2023-05-21T06:02:41.670444Z","shell.execute_reply":"2023-05-21T06:02:49.402704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!nvidia-smi\n","metadata":{"execution":{"iopub.status.busy":"2023-05-21T04:07:27.669571Z","iopub.execute_input":"2023-05-21T04:07:27.670286Z","iopub.status.idle":"2023-05-21T04:07:28.765346Z","shell.execute_reply.started":"2023-05-21T04:07:27.670241Z","shell.execute_reply":"2023-05-21T04:07:28.764022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir('/kaggle/input/cassava-leaf-disease-classification')\n\n# tfrecords is a tensorflow file format for storing the images\n# json files are mainly used for data transfer (mostly text)\n# csv files contains image file names and their corresponding labels","metadata":{"execution":{"iopub.status.busy":"2023-05-21T06:02:54.250009Z","iopub.execute_input":"2023-05-21T06:02:54.250678Z","iopub.status.idle":"2023-05-21T06:02:54.261615Z","shell.execute_reply.started":"2023-05-21T06:02:54.250642Z","shell.execute_reply":"2023-05-21T06:02:54.260613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reading json files to know different classes of possible leaf disease\n\nwith open('/kaggle/input/cassava-leaf-disease-classification/label_num_to_disease_map.json') as f:\n    print(json.loads(f.read()))","metadata":{"execution":{"iopub.status.busy":"2023-05-21T06:02:59.339960Z","iopub.execute_input":"2023-05-21T06:02:59.340333Z","iopub.status.idle":"2023-05-21T06:02:59.356419Z","shell.execute_reply.started":"2023-05-21T06:02:59.340302Z","shell.execute_reply":"2023-05-21T06:02:59.355248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_lbl = pd.read_csv('/kaggle/input/cassava-leaf-disease-classification/train.csv',nrows=100)\nimg_lbl.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-21T06:03:02.816869Z","iopub.execute_input":"2023-05-21T06:03:02.817702Z","iopub.status.idle":"2023-05-21T06:03:02.846376Z","shell.execute_reply.started":"2023-05-21T06:03:02.817667Z","shell.execute_reply":"2023-05-21T06:03:02.845319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Removing Duplicate images as mentioned in the discussion ('1562043567.jpg', '3551135685.jpg', '2252529694.jpg' are duplicate)\n\nimg_lbl=img_lbl[~img_lbl['image_id'].isin(['1562043567.jpg', '3551135685.jpg', '2252529694.jpg'])]","metadata":{"execution":{"iopub.status.busy":"2023-05-21T06:03:06.120248Z","iopub.execute_input":"2023-05-21T06:03:06.120599Z","iopub.status.idle":"2023-05-21T06:03:06.133840Z","shell.execute_reply.started":"2023-05-21T06:03:06.120570Z","shell.execute_reply":"2023-05-21T06:03:06.132935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_lbl['label'].value_counts()\n\n# Cassava Mosaic Disease (CMD) is the most spread leaf disease.\n# Cassava Bacterial Blight (CBB) is the least spread leaf disease.","metadata":{"execution":{"iopub.status.busy":"2023-05-21T06:03:09.570819Z","iopub.execute_input":"2023-05-21T06:03:09.571314Z","iopub.status.idle":"2023-05-21T06:03:09.586794Z","shell.execute_reply.started":"2023-05-21T06:03:09.571274Z","shell.execute_reply":"2023-05-21T06:03:09.585809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# importing some random images\n\nX=[]   # variable to store leaf images\nZ=[]   # variable to store leaf diseases\n\nfor img, dseas in tqdm(img_lbl.sample(9).values):\n    image=cv2.imread('/kaggle/input/cassava-leaf-disease-classification/train_images/{}'.format(img),\n                     cv2.IMREAD_COLOR)\n    image=cv2.resize(image,(600,600))\n    X.append(image)    # Appending the images into X\n    Z.append(dseas)    # Appending the image labels into Z","metadata":{"execution":{"iopub.status.busy":"2023-05-21T06:03:13.243698Z","iopub.execute_input":"2023-05-21T06:03:13.244173Z","iopub.status.idle":"2023-05-21T06:03:13.548303Z","shell.execute_reply.started":"2023-05-21T06:03:13.244123Z","shell.execute_reply":"2023-05-21T06:03:13.547319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax=plt.subplots(3,3)\nfig.set_size_inches(20,20)\nl=0\nfor row in range(3):    \n    for col in range(3):\n        ax[row,col].imshow(X[l])\n        ax[row,col].set_title('Disease Class : '+str(Z[l]))\n        l=l+1\n\nplt.tight_layout\nsns.set(font_scale=1.5)","metadata":{"execution":{"iopub.status.busy":"2023-05-21T06:03:17.596683Z","iopub.execute_input":"2023-05-21T06:03:17.597044Z","iopub.status.idle":"2023-05-21T06:03:21.067027Z","shell.execute_reply.started":"2023-05-21T06:03:17.597012Z","shell.execute_reply":"2023-05-21T06:03:21.065922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Splitting data into train andkerasdation\n\nfrom sklearn.model_selection import train_test_split\ntrain,validation = train_test_split(img_lbl,test_size=0.2,shuffle=True,stratify=img_lbl['label'])","metadata":{"execution":{"iopub.status.busy":"2023-05-21T06:03:32.963155Z","iopub.execute_input":"2023-05-21T06:03:32.963822Z","iopub.status.idle":"2023-05-21T06:03:32.970600Z","shell.execute_reply.started":"2023-05-21T06:03:32.963788Z","shell.execute_reply":"2023-05-21T06:03:32.969623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D,GlobalAveragePooling2D,Dense,Flatten,Dropout\nfrom tensorflow.keras.applications import EfficientNetB3","metadata":{"execution":{"iopub.status.busy":"2023-05-21T06:03:36.733007Z","iopub.execute_input":"2023-05-21T06:03:36.733373Z","iopub.status.idle":"2023-05-21T06:03:36.742270Z","shell.execute_reply.started":"2023-05-21T06:03:36.733343Z","shell.execute_reply":"2023-05-21T06:03:36.741325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# Imagedatagenerator for training\ndatagen_trng = ImageDataGenerator(preprocessing_function=tf.keras.applications.efficientnet.preprocess_input,\n                                  rotation_range=30,\n                                  width_shift_range=0.3,\n                                  height_shift_range=0.3,\n                                  shear_range=0.2,\n                                  zoom_range=0.3,\n                                  horizontal_flip=True,\n                                  fill_mode='nearest')\n\n# label should be converted to string to be used\ntrain['label']=train['label'].astype('str')                        \n\n# Augmenting Images for training\ntrain_datagen=datagen_trng.flow_from_dataframe(dataframe=train,\n                                        directory='/kaggle/input/cassava-leaf-disease-classification/train_images',\n                                        x_col=\"image_id\",\n                                        y_col=\"label\",\n                                        color_mode=\"rgb\",\n                                        target_size=(300,300),\n                                        batch_size=4,\n                                        seed=42,\n                                        class_mode=\"categorical\")","metadata":{"execution":{"iopub.status.busy":"2023-05-21T06:03:40.391316Z","iopub.execute_input":"2023-05-21T06:03:40.391682Z","iopub.status.idle":"2023-05-21T06:03:40.699417Z","shell.execute_reply.started":"2023-05-21T06:03:40.391644Z","shell.execute_reply":"2023-05-21T06:03:40.698553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Imagedatagenerator for validation\ndatagen_valid = ImageDataGenerator(preprocessing_function=tf.keras.applications.efficientnet.preprocess_input)\n\n# label should be converted to string to be used\nvalidation['label']=validation['label'].astype('str')\n\n# Augmenting Images for validating\nvalid_datagen=datagen_valid.flow_from_dataframe(dataframe=validation,\n                                        directory='/kaggle/input/cassava-leaf-disease-classification/train_images',\n                                        x_col='image_id',\n                                        y_col=\"label\",\n                                        color_mode=\"rgb\",\n                                        target_size=(300,300),\n                                        batch_size=4,\n                                        seed=42,\n                                        class_mode=\"categorical\")","metadata":{"execution":{"iopub.status.busy":"2023-05-21T06:03:44.174197Z","iopub.execute_input":"2023-05-21T06:03:44.174566Z","iopub.status.idle":"2023-05-21T06:03:44.255233Z","shell.execute_reply.started":"2023-05-21T06:03:44.174535Z","shell.execute_reply":"2023-05-21T06:03:44.254157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining model\n\nmodel=Sequential()\nmodel.add(EfficientNetB3(include_top=False,weights='imagenet',input_shape=(300,300,3)))\nmodel.add(GlobalAveragePooling2D())\nmodel.add(Flatten())\nmodel.add(Dense(128,activation='relu'))\nmodel.add(Dropout(0.4))\nmodel.add(Dense(256,activation='relu'))\nmodel.add(Dropout(0.4))\nmodel.add(Dense(5,activation='softmax'))\n\nmodel.compile(optimizer=tf.keras.optimizers.Adam(),loss=tf.keras.losses.CategoricalCrossentropy(),\n              metrics=tf.keras.metrics.CategoricalAccuracy())\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-05-21T06:03:48.497802Z","iopub.execute_input":"2023-05-21T06:03:48.498188Z","iopub.status.idle":"2023-05-21T06:03:57.849844Z","shell.execute_reply.started":"2023-05-21T06:03:48.498155Z","shell.execute_reply":"2023-05-21T06:03:57.848666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining callbacks\n\nfrom tensorflow.keras.callbacks import EarlyStopping,ReduceLROnPlateau\n\nearly_stop=EarlyStopping(monitor='val_categorical_accuracy',\n                         min_delta=0.002,\n                         patience=3,\n                         mode='max',\n                         verbose=1,\n                         restore_best_weights=True)\n\nreduce_lr=ReduceLROnPlateau(monitor='val_categorical_accuracy',\n                            patience=2,\n                            factor=0.1,\n                            mode='max',\n                            min_lr=1e-6,\n                            verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-21T06:04:02.413953Z","iopub.execute_input":"2023-05-21T06:04:02.414345Z","iopub.status.idle":"2023-05-21T06:04:02.423223Z","shell.execute_reply.started":"2023-05-21T06:04:02.414312Z","shell.execute_reply":"2023-05-21T06:04:02.421048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history= model.fit(train_datagen,\n                   batch_size=train_datagen.n//train_datagen.batch_size,\n                   epochs=25,verbose=1,shuffle=True,\n                   validation_data=valid_datagen,\n                   callbacks=[early_stop,reduce_lr])","metadata":{"execution":{"iopub.status.busy":"2023-05-21T06:04:08.965096Z","iopub.execute_input":"2023-05-21T06:04:08.965450Z","iopub.status.idle":"2023-05-21T06:05:42.363232Z","shell.execute_reply.started":"2023-05-21T06:04:08.965420Z","shell.execute_reply":"2023-05-21T06:05:42.362349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluating our model","metadata":{}},{"cell_type":"code","source":"model_eval=model.evaluate(valid_datagen,verbose=1)\nprint('Validation loss: ',model_eval[0])\nprint('Validation accuracy: ',model_eval[1])","metadata":{"execution":{"iopub.status.busy":"2023-05-21T06:05:56.660882Z","iopub.execute_input":"2023-05-21T06:05:56.661579Z","iopub.status.idle":"2023-05-21T06:05:57.021583Z","shell.execute_reply.started":"2023-05-21T06:05:56.661544Z","shell.execute_reply":"2023-05-21T06:05:57.020642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_dict=history.history\n\nloss_value=history_dict['loss']\nval_loss_value=history_dict['val_loss']\nepoch=range(1,len(loss_value)+1)\n\nlin1=plt.plot(epoch,val_loss_value,label='Validation loss')\nlin2=plt.plot(epoch,loss_value,label='Training loss')\nplt.setp(lin1,linewidth=2.0,marker='+',markersize=10.0)\nplt.setp(lin2,linewidth=2.0,marker='4',markersize=10.0)\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.grid(True)\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2023-05-21T06:06:01.510991Z","iopub.execute_input":"2023-05-21T06:06:01.511566Z","iopub.status.idle":"2023-05-21T06:06:01.880756Z","shell.execute_reply.started":"2023-05-21T06:06:01.511533Z","shell.execute_reply":"2023-05-21T06:06:01.879884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc_value=history_dict['categorical_accuracy']\nval_acc_value=history_dict['val_categorical_accuracy']\nepoch=range(1,len(loss_value)+1)\n\nlin1=plt.plot(epoch,val_acc_value,label='Validation CategoricalAccuracy')\nlin2=plt.plot(epoch,acc_value,label='Training CategoricalAccuracy')\nplt.setp(lin1,linewidth=2.0,marker='+',markersize=10.0)\nplt.setp(lin2,linewidth=2.0,marker='4',markersize=10.0)\nplt.xlabel('Epochs')\nplt.ylabel('Categorical Accuracy')\nplt.grid(True)\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2023-05-21T06:06:11.715481Z","iopub.execute_input":"2023-05-21T06:06:11.715842Z","iopub.status.idle":"2023-05-21T06:06:12.114834Z","shell.execute_reply.started":"2023-05-21T06:06:11.715811Z","shell.execute_reply":"2023-05-21T06:06:12.113926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"As we can see, the best validation loass obtained after RandomSearch is 1.069 which is greater than the validation loss obtained in our first model training. So, we'll not use the model after RandomSearch. Saving our first model for deployment.","metadata":{}}]}