{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Cassava Disease Classification"},{"metadata":{},"cell_type":"markdown","source":"## Introduction\n1. The notebook is beginners approach for Cassava Disease Classification.\n2. The current notebook uses EfficientNetB3 for transfer learning.\n3. It achieve accuracy of more than 87% on validation data and 96% on training data. "},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n%config Completer.use_jedi = False\n\nimport warnings\nwarnings.filterwarnings('ignore')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Data Loading and Splitting"},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv('../input/cassava-leaf-disease-classification/train.csv')\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Importing the json file with labels\nimport json\n\nwith open('../input/cassava-leaf-disease-classification/label_num_to_disease_map.json') as f:\n    real_labels = json.load(f)\n    real_labels = {int(k):v for k,v in real_labels.items()}\n\nreal_labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.countplot(df['label'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain,test = train_test_split(df,test_size = 0.1,stratify = df['label'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Data Generator"},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.applications.efficientnet import EfficientNetB3, preprocess_input\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"datagen = ImageDataGenerator(\n    preprocessing_function = preprocess_input)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['label'] = train['label'].astype(str)\ntrain_gen = datagen.flow_from_dataframe(\n    train,\n    directory = '../input/cassava-leaf-disease-classification/train_images',\n    x_col = 'image_id',\n    y_col = 'label',\n    target_size = (512,512),\n    batch_size = 8\n)\n\ntest['label'] = test['label'].astype(str)\nval_gen = datagen.flow_from_dataframe(\n    test,\n    directory = '../input/cassava-leaf-disease-classification/train_images',\n    x_col = 'image_id',\n    y_col = 'label',\n    target_size = (512,512),\n    batch_size = 8\n)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Modeling"},{"metadata":{"trusted":true},"cell_type":"code","source":"base_model = EfficientNetB3(include_top=False, weights='imagenet',input_shape=(512,512,3))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras import layers,models\n\nmodel = models.Sequential()\nmodel.add(base_model)\nmodel.add(layers.GlobalAveragePooling2D())\nmodel.add(layers.Dense(256,activation='relu'))\nmodel.add(layers.Dropout(0.4))\nmodel.add(layers.Dense(5,activation='softmax'))\n\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.compile(\n    optimizer='adam',\n    loss = 'categorical_crossentropy',\n    metrics = ['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.callbacks import ModelCheckpoint,ReduceLROnPlateau\ncheckpoint = ModelCheckpoint(filepath='./best_model.h5', monitor='val_loss', save_best_only=True,verbose=1)\n\nreducelr = ReduceLROnPlateau( \n    monitor='val_loss',\n    factor=0.2,\n    patience=2,\n    min_lr=1e-6,\n    mode='min',\n    verbose=1\n)\n\nmy_callbacks = [checkpoint,reducelr]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_steps = np.ceil(train_gen.n/train_gen.batch_size)\nval_steps = np.ceil(val_gen.n/val_gen.batch_size)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Training"},{"metadata":{"trusted":true},"cell_type":"code","source":"history = model.fit(\n    train_gen,\n    batch_size = 8,\n    epochs = 10,\n    steps_per_epoch = train_steps,\n    validation_data = val_gen,\n    validation_steps = val_steps,\n    callbacks = my_callbacks)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.save('./best_model2.h5')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Plotting results"},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\n# summarize history for accuracy\nplt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()\n# summarize history for loss\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Submission\n### Note: Separate notebook has to be created for submission"},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.models import load_model\nmodel = load_model('./best_model.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ss = pd.read_csv('../input/cassava-leaf-disease-classification/sample_submission.csv')\nss.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_path = '../input/cassava-leaf-disease-classification/test_images/'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = []\n\nfor image_name in ss.image_id:\n    img = image.load_img(test_path+image_name,target_size = (512,512))\n    x = image.img_to_array(img)\n    x = np.expand_dims(x,axis=0)\n    x = preprocess_input(x)\n    \n    prediction = model.predict(x)\n    prediction = np.argmax(prediction)\n    \n    preds.append(prediction)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"my_submission = pd.DataFrame({'image_id': ss.image_id, 'label': preds})\nmy_submission.to_csv('submission.csv', index=False)\nmy_submission.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Conclusion\n1. The current model was able to get accuracy of 87% on validation data and 96% on training data.\n2. On test dataset it gives accuracy of 88%.\n3. Furthers techniques can be applied to prevent the model from overfiiting to get better accuracy.\n\n\n![](https://st3.depositphotos.com/1998651/13850/v/1600/depositphotos_138506364-stock-illustration-cup-of-coffee-with-have.jpg)\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}