{"cells":[{"metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"cell_type":"code","source":"!pip install pydot\n!pip install pydotplus\n!pip install graphviz","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 🌿 Cassava Disease Classification 🌿"},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nimport glob\nimport random\nimport shutil\nimport warnings\nimport json\nimport itertools\nimport numpy as np\nimport pandas as pd\nfrom collections import Counter\n\nimport plotly.express as px\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport keras\nfrom keras.preprocessing.image import ImageDataGenerator\nimport tensorflow as tf\nfrom PIL import Image\n\nfrom sklearn.model_selection import train_test_split\n\n# Defining the working directories\nwork_dir = '../input/cassava-leaf-disease-classification/'\nos.listdir(work_dir) \ntrain_path = '/kaggle/input/cassava-leaf-disease-classification/train_images'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Num GPUs Available: \", len(tf.config.experimental.list_physical_devices('GPU')))\nwith tf.device('/GPU:0'):\n    print('Yes, there is GPU')\n    \ntf.debugging.set_log_device_placement(True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Lets set all random seeds\n\ndef seed_everything(seed=0):\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    os.environ['TF_DETERMINISTIC_OPS'] = '1'\n\nseed = 21\nseed_everything(seed)\nwarnings.filterwarnings('ignore')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 1. Data loading 📚"},{"metadata":{"trusted":true},"cell_type":"code","source":"data = pd.read_csv(work_dir + 'train.csv')\nprint(data['label'].value_counts()) # Checking the frequencies of the labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Importing the json file with labels\nwith open(work_dir + 'label_num_to_disease_map.json') as f:\n    real_labels = json.load(f)\n    real_labels = {int(k):v for k,v in real_labels.items()}\n    \n# Defining the working dataset\ndata['class_name'] = data['label'].map(real_labels)\n\nreal_labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# generate train and test sets\ntrain, test = train_test_split(data, test_size = 0.05, random_state = 42, stratify = data['class_name'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"IMG_SIZE = 456\nsize = (IMG_SIZE,IMG_SIZE)\nn_CLASS = 5\nBATCH_SIZE = 15","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 2. Data generator ⚙️"},{"metadata":{"trusted":true},"cell_type":"code","source":"datagen_train = ImageDataGenerator(\n    preprocessing_function = tf.keras.applications.efficientnet.preprocess_input,\n    rotation_range = 40,\n    width_shift_range = 0.2,\n    height_shift_range = 0.2,\n    shear_range = 0.2,\n    zoom_range = 0.2,\n    horizontal_flip = True,\n    vertical_flip = True,\n    fill_mode = 'nearest',\n)\n\ndatagen_val = ImageDataGenerator(\n    preprocessing_function = tf.keras.applications.efficientnet.preprocess_input,\n)\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Now, generate the new sets "},{"metadata":{"trusted":true},"cell_type":"code","source":"train_set = datagen_train.flow_from_dataframe(\n    train,\n    directory=train_path,\n    seed=42,\n    x_col='image_id',\n    y_col='class_name',\n    target_size = size,\n    class_mode='categorical',\n    interpolation='nearest',\n    shuffle = True,\n    batch_size = BATCH_SIZE,\n)\n\ntest_set = datagen_val.flow_from_dataframe(\n    test,\n    directory=train_path,\n    seed=42,\n    x_col='image_id',\n    y_col='class_name',\n    target_size = size,\n    class_mode='categorical',\n    interpolation='nearest',\n    shuffle=True,\n    batch_size=BATCH_SIZE,    \n)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 3. Modeling part 🍻"},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import GlobalAveragePooling2D, Flatten, Dense, Dropout, BatchNormalization\nfrom keras.optimizers import RMSprop, Adam\nfrom keras.callbacks import EarlyStopping, ModelCheckpoint, ReduceLROnPlateau\nfrom tensorflow.keras.applications import EfficientNetB3","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def create_model():\n    \n    model = Sequential()\n    # initialize the model with input shape\n    model.add(\n        EfficientNetB3(\n            input_shape = (IMG_SIZE, IMG_SIZE, 3), \n            include_top = False,\n            weights='imagenet',\n            drop_connect_rate=0.6,\n        )\n    )\n    model.add(GlobalAveragePooling2D())\n    model.add(Flatten())\n    model.add(Dense(\n        256, \n        activation='relu', \n        bias_regularizer=tf.keras.regularizers.L1L2(l1=0.01, l2=0.001)\n    ))\n    model.add(Dropout(0.5))\n    model.add(Dense(n_CLASS, activation = 'softmax'))\n    \n    return model\n\nleaf_model = create_model()\nleaf_model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"keras.utils.plot_model(leaf_model)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"EPOCHS = 15\nSTEP_SIZE_TRAIN = train_set.n // train_set.batch_size\nSTEP_SIZE_TEST = test_set.n // test_set.batch_size","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Fit the model"},{"metadata":{"trusted":true},"cell_type":"code","source":"def model_fit():\n    leaf_model = create_model()\n    \n    # Loss function \n    # https://www.tensorflow.org/api_docs/python/tf/keras/losses/CategoricalCrossentropy\n    loss = tf.keras.losses.CategoricalCrossentropy(\n        from_logits = False,\n        label_smoothing=0.0001,\n        name='categorical_crossentropy'\n    )\n    \n    # Compile the model\n    leaf_model.compile(\n        optimizer = Adam(learning_rate = 1e-3),\n        loss = loss, #'categorical_crossentropy'\n        metrics = ['categorical_accuracy']\n    )\n    \n    # Stop training when the val_loss has stopped decreasing for 3 epochs.\n    # https://www.tensorflow.org/api_docs/python/tf/keras/callbacks/EarlyStopping\n    es = EarlyStopping(\n        monitor='val_loss', \n        mode='min', \n        patience=3,\n        restore_best_weights=True, \n        verbose=1,\n    )\n    \n    # Save the model with the minimum validation loss\n    # https://www.tensorflow.org/api_docs/python/tf/keras/callbacks/ModelCheckpoint\n    checkpoint_cb = ModelCheckpoint(\n        \"Cassava_best_model.h5\",\n        save_best_only=True,\n        monitor='val_loss',\n        mode='min',\n    )\n    \n    # Reduce learning rate once learning stagnates\n    # https://www.tensorflow.org/api_docs/python/tf/keras/callbacks/ReduceLROnPlateau\n    reduce_lr = ReduceLROnPlateau(\n        monitor='val_loss',\n        factor=0.2,\n        patience=2,\n        min_lr=1e-6,\n        mode='min',\n        verbose=1,\n    )\n    \n    # Fit the model\n    history = leaf_model.fit(\n        train_set,\n        validation_data=test_set,\n        epochs=EPOCHS,\n        batch_size=BATCH_SIZE,\n        steps_per_epoch=STEP_SIZE_TRAIN,\n        validation_steps=STEP_SIZE_TEST,\n        callbacks=[es, checkpoint_cb, reduce_lr],\n    )\n    \n    # Save the model\n    leaf_model.save('Cassava_model'+'.h5')  \n    \n    return history","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sess = tf.compat.v1.Session(config=tf.compat.v1.ConfigProto(log_device_placement=True))\n\nfrom tensorflow.compat.v1.keras import backend as K\nK.set_session(sess)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"try:\n    final_model = keras.models.load_model('Cassava_model.h5')\nexcept Exception as e:\n    with tf.device('/GPU:0'):\n        results = model_fit()\n    print('Train Categorical Accuracy: ', max(results.history['categorical_accuracy']))\n    print('Test Categorical Accuracy: ', max(results.history['val_categorical_accuracy']))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 4. Plot results 📊"},{"metadata":{"trusted":true},"cell_type":"code","source":"def trai_test_plot(acc, test_acc, loss, test_loss):\n    \n    fig, (ax1, ax2) = plt.subplots(1,2, figsize= (15,10))\n    fig.suptitle(\"Model's metrics comparisson\", fontsize=20)\n\n    ax1.plot(range(1, len(acc) + 1), acc)\n    ax1.plot(range(1, len(test_acc) + 1), test_acc)\n    ax1.set_title('History of Accuracy', fontsize=15)\n    ax1.set_xlabel('Epochs', fontsize=15)\n    ax1.set_ylabel('Accuracy', fontsize=15)\n    ax1.legend(['training', 'validation'])\n\n\n    ax2.plot(range(1, len(loss) + 1), loss)\n    ax2.plot(range(1, len(test_loss) + 1), test_loss)\n    ax2.set_title('History of Loss', fontsize=15)\n    ax2.set_xlabel('Epochs', fontsize=15)\n    ax2.set_ylabel('Loss', fontsize=15)\n    ax2.legend(['training', 'validation'])\n    plt.show()\n    \n\ntrai_test_plot(\n    results.history['categorical_accuracy'],\n    results.history['val_categorical_accuracy'],\n    results.history['loss'],\n    results.history['val_loss']\n)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 5. Submission 📝"},{"metadata":{"trusted":true},"cell_type":"code","source":"final_model = keras.models.load_model('Cassava_model.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TEST_DIR = '../input/cassava-leaf-disease-classification/test_images/'\ntest_images = os.listdir(TEST_DIR)\npredictions = []\n\nfor image in test_images:\n    img = Image.open(TEST_DIR + image)\n    img = img.resize(size)\n    img = np.expand_dims(img, axis=0)\n    predictions.extend(final_model.predict(img).argmax(axis = 1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub = pd.DataFrame({'image_id': test_images, 'label': predictions})\ndisplay(sub)\nsub.to_csv('submission.csv', index = False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}