{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"!nvidia-smi","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 1: IMPORTING LIBRARIES AND DATA"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport json\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport plotly.graph_objects as go\nimport cv2\n\nimport tensorflow as tf\nimport tensorflow.keras.backend as k\nfrom keras_preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras import layers, optimizers\nfrom tensorflow.keras.initializers import glorot_uniform\nfrom tensorflow.keras.utils import plot_model\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, EarlyStopping, ModelCheckpoint\nfrom sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"base_path = \"../input/cassava-leaf-disease-classification/\"\n\nwith open(os.path.join(base_path, 'label_num_to_disease_map.json'), 'r') as f:\n    class_map = json.load(f)\n    class_map = {int(k):v for k,v in class_map.items()}\nprint(class_map)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"so out of **5 classes**, we are having **4 classes** with disease and **1 Healthy**"},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Number of Images in Train Set {} & in Test Set {}\".format(len(os.listdir(os.path.join(base_path, 'train_images'))), \n                                                                 len(os.listdir(os.path.join(base_path, 'test_images')))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv(os.path.join(base_path, 'train.csv'))\ndf['class_name'] = df.label.map(class_map)\ndf","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 2: DATA VISULAZIATION"},{"metadata":{"trusted":true},"cell_type":"code","source":"df.class_name.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.histogram(df, x=\"class_name\", color=\"class_name\")\n\nfig.update_layout(\n    yaxis=dict(title_text='Count', titlefont=dict(size=20)),\n    xaxis=dict(title_text='Class Label Name (Healthy or Disease)', titlefont=dict(size=20)),\n    title_text='Class Label Name Count Plot'\n)\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"def plot_batch(data=df):\n    plt.figure(figsize=(16,12))\n    for i in range(9):\n        k = np.random.randint(0, len(data)) #for plotting random images from dataset\n        image = cv2.imread(os.path.join(base_path, 'train_images/', data.image_id[k]))\n        \n        plt.subplot(3,3,i+1)\n        plt.imshow(image)\n        plt.axis(\"off\")\n        plt.title(\"Class Label:{}\\nClass Name:{}\".format(data.label[k], data.class_name[k]))\n    \n    plt.tight_layout()\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"plot_batch()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**now let's have a look at batch images of each class one by one**\n\n<h2 align=center> Label:0 - Cassava Bacterial Blight (CBB)</h2>\n\n**Main characteristics to leverage: angular spots, brown spots with yellow borders, yellow leaves, leaves wilting**\n\n<img style=\"height:300px\" src=\"https://www.googleapis.com/download/storage/v1/b/kaggle-forum-message-attachments/o/inbox%2F1865449%2Fbe9cdd94efb9b1660066ad10b55c8626%2Fbact_bright.jpeg?generation=1605827469211692&alt=media\">\n\nall the images and characterstic are takes from [discussion](https://www.kaggle.com/c/cassava-leaf-disease-classification/discussion/198143)"},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"temp_df = df.loc[df['label']==0]\ntemp_df.reset_index(inplace=True)\nplot_batch(temp_df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"- black leaf spots and blights, angular leaf spots, and premature drying and shedding of leaves due to the wilting of young leaves and severe attack.\n\n- At first, angular, water-soaked spots occur on the leaves which are restricted by the veins; the spots are more clearly seen on the lower leaf surface. The spots expand rapidly, join together, especially along the margins of the leaves, and turn brown with yellow borders\n\n- Droplets of a creamy-white ooze occur at the centre of the spots; later, they turn yellow."},{"metadata":{},"cell_type":"markdown","source":"<h2 align=center> Label:1 - Cassava Brown Streak Disease (CBSD) </h2>\n\n**Main characteristics to leverage: yellow spots**\n<img style=\"height:300px\" src=\"https://www.googleapis.com/download/storage/v1/b/kaggle-forum-message-attachments/o/inbox%2F1865449%2Ffeba3dafc914d04517659650d137b77a%2Fbrown_st.jpeg?generation=1605830407530983&alt=media\">"},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"temp_df = df.loc[df['label']==1]\ntemp_df.reset_index(inplace=True)\nplot_batch(temp_df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"- CBSD leaf symptoms consist of a characteristic yellow or necrotic vein banding which may enlarge and coalesce to form comparatively large yellow patches.\n\n- Tuberous root symptoms consist of dark-brown necrotic areas within the tuber and reduction in root size"},{"metadata":{},"cell_type":"markdown","source":"<h2 align=center> Label:2 Cassava Green Mottle (CGM) </h2>\n\n**Main characteristics to leverage: yellow patterns, irregular patches of yellow and green, leaf margins distortion, stunted**\n\n<img style=\"height:300px\" src=\"https://www.googleapis.com/download/storage/v1/b/kaggle-forum-message-attachments/o/inbox%2F1865449%2F4f2975866feb2a1d4ef4111c2d57db29%2Fgreen_mottle.jpeg?generation=1605829101431013&alt=media\">"},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"temp_df = df.loc[df['label']==2]\ntemp_df.reset_index(inplace=True)\nplot_batch(temp_df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"- This disease causes white spotting of leaves, which increase from the initial small spots to cover the entire leaf causing loss of chlorophyll. Young leaves are puckered with faint to distinct yellow spots\n\n- Leaves with this disease show mottled symptoms which can be confused with symptoms of cassava mosaic disease (CMD). Severely damaged leaves shrink, dry out and fall off, which can cause a characteristic candle-stick appearance"},{"metadata":{},"cell_type":"markdown","source":"<h2 align=center> Label:3 - Cassava Mosaic Disease (CMD) </h2>\n\n**Main characteristics to leverage: severe shape distortion, mosaic patterns**\n\n<img style=\"height:300px\" src=\"https://www.googleapis.com/download/storage/v1/b/kaggle-forum-message-attachments/o/inbox%2F1865449%2F36990f77ded6667e5c30d19b5405d4d3%2Fmosaic_disease.jpeg?generation=1605829705010773&alt=media\">"},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"temp_df = df.loc[df['label']==3]\ntemp_df.reset_index(inplace=True)\nplot_batch(temp_df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"- CMD produces a variety of foliar symptoms that include mosaic, mottling, misshapen and twisted leaflets, and an overall reduction in size of leaves and plants\n\n- Leaves affected by this disease have patches of normal green color mixed with different proportions of yellow and white depending on the severity"},{"metadata":{},"cell_type":"markdown","source":"<h2 align=center> Label:4 - Healthy </h2>"},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"temp_df = df.loc[df['label']==4]\ntemp_df.reset_index(inplace=True)\nplot_batch(temp_df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"test set image"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.imshow(cv2.imread(\"../input/cassava-leaf-disease-classification/test_images/2216849948.jpg\"));\nplt.axis('off');","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 3: CREATING TRAIN & VALIDATION DATASET"},{"metadata":{"trusted":true},"cell_type":"code","source":"# giving image path to image_id column\ndf['path'] = df['image_id'].apply(lambda x: base_path + 'train_images/' + x)\n\n# Convert the data in mask column to string format, to use categorical mode in flow_from_dataframe\n#df['label'] = df['label'].apply(lambda x: str(x))\n\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"seed = 42\nX_train, X_val = train_test_split(df, test_size = 0.1, random_state=seed, shuffle=True)\nprint(\"Training Set: {} \\t Validation Set: {}\".format(len(X_train), len(X_val)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"del df # to free up space\ndel temp_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"with tf.device('/cpu:0'):\n    train_data = tf.data.Dataset.from_tensor_slices((X_train.path.values, X_train.label.values))\n    valid_data = tf.data.Dataset.from_tensor_slices((X_val.path.values, X_val.label.values))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for path, label in train_data.take(3):\n    print ('Path: {}, Label: {}'.format(path, label))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def process_data_train(image_path, label):\n    # load the raw data from the file as a string\n    img = tf.io.read_file(image_path)\n    img = tf.image.decode_jpeg(img, channels=3)\n    img = tf.image.random_brightness(img, 0.3)\n    img = tf.image.random_flip_left_right(img, seed=None)\n    img = tf.image.random_flip_up_down(img)\n    img = tf.image.random_crop(img, size=[row,col, 3])\n    return img, label\n\ndef process_data_valid(image_path, label):\n    # load the raw data from the file as a string\n    img = tf.io.read_file(image_path)\n    img = tf.image.decode_jpeg(img, channels=3)\n    img = tf.image.resize(img, [row,col])\n    return img, label","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Set `num_parallel_calls` so multiple images are loaded/processed in parallel.\nrow,col = 380, 380\n\nwith tf.device('/cpu:0'):\n    train_data = train_data.map(process_data_train, num_parallel_calls=tf.data.experimental.AUTOTUNE)\n    valid_data = valid_data.map(process_data_valid, num_parallel_calls=tf.data.experimental.AUTOTUNE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def configure_for_performance(ds, batch_size = 32):\n    ds = ds.cache('/kaggle/dump.tfcache') \n    \n    ds = ds.shuffle(buffer_size=1024)\n    ds = ds.batch(batch_size)\n    ds = ds.prefetch(buffer_size=tf.data.experimental.AUTOTUNE)\n    return ds\n\nbs = 16\n\nwith tf.device('/cpu:0'):\n    train_data_batch = configure_for_performance(train_data, bs)\n    valid_data_batch = valid_data.batch(bs)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#plotting 1st batch\n\ndef plot_df_batch():\n    plt.figure(figsize=(10, 10))\n    image_batch, label_batch = next(iter(train_data_batch)) #loading batch\n    for i in range(8):\n        ax = plt.subplot(4, 4, i + 1)\n        plt.imshow(image_batch[i].numpy().astype(\"uint8\"))\n        label = label_batch[i].numpy()\n        plt.title(\"Class Label :\" + str(label))\n        plt.axis(\"off\")\n\nplot_df_batch()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_augmentation = tf.keras.Sequential([\n    tf.keras.layers.experimental.preprocessing.RandomRotation(0.2, interpolation='nearest'),\n    tf.keras.layers.experimental.preprocessing.RandomContrast((0.2))\n])\n\ndef plot_arg():\n    plt.figure(figsize=(10, 10))\n    image_batch, label_batch = next(iter(train_data_batch)) #loading batch\n    for i in range(8):\n        augmented_images = data_augmentation(image_batch)\n        ax = plt.subplot(4, 4, i + 1)\n        plt.imshow(augmented_images[i].numpy().astype(\"uint8\"))\n        label = label_batch[i].numpy()\n        plt.title(label)\n        plt.axis(\"off\")\n        \nplot_arg()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# row, col = 512, 512\n# bs = 8 # batch size\n\n# datagen = ImageDataGenerator(\n#     rescale=1./255.,\n#     validation_split=0.1,\n#     zoom_range=0.2,\n#     rotation_range=0.15,\n#     horizontal_flip=True,\n#     vertical_flip=True,\n#     fill_mode='nearest',\n#     shear_range=0.2,\n#     height_shift_range=0.1,\n#     width_shift_range=0.1\n# )\n\n# train_generator = datagen.flow_from_dataframe(\n#     df,\n#     x_col='path',\n#     y_col='label',\n#     class_mode='categorical',\n#     batch_size=bs,\n#     shuffle=True,\n#     target_size=(row,col),\n#     subset='training'\n# )\n# val_generator = datagen.flow_from_dataframe(\n#     df,\n#     x_col='path',\n#     y_col='label',\n#     class_mode='categorical',\n#     batch_size=bs,\n#     shuffle=True,\n#     target_size=(row,col),\n#     subset='validation'\n# )","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 4: CALLING PRE-TRAINED MODEL"},{"metadata":{"trusted":true},"cell_type":"code","source":"def build_model(opt=tf.keras.optimizers.Adam(lr = 1e-4), loss='sparse_categorical_crossentropy', metrics =['sparse_categorical_accuracy']):\n    \n    base = tf.keras.applications.EfficientNetB4(\n        weights='../input/tfkerasefficientnetimagenetnotop/efficientnetb4_notop.h5', \n        include_top=False, \n        input_shape=(row,col,3),\n        drop_connect_rate=0.4\n    )\n    base.trainable = True\n    \n    inputs = tf.keras.layers.Input(shape=(row,col, 3))\n    X = data_augmentation(inputs)\n    X = tf.keras.layers.experimental.preprocessing.Rescaling(1./255.)(X)\n    X = base(inputs)\n    X = GlobalAveragePooling2D()(X)\n    X = Dropout(0.3)(X)\n    X = Dense(256, activation='relu', kernel_initializer='he_normal')(X)\n    X = BatchNormalization()(X)\n    X = Dropout(0.3)(X)\n    output = Dense(5, kernel_initializer='he_normal', activation='softmax')(X)\n\n    model = tf.keras.Model(inputs, output)\n    \n    model.compile(\n        optimizer=opt, loss=loss, metrics=metrics\n    )\n    \n    model.summary()\n    return model\n\n\nmodel = build_model()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 5: MODEL TRAINING AND DEFINING CALLBACKS"},{"metadata":{"trusted":true},"cell_type":"code","source":"checkpointer = ModelCheckpoint(\n    filepath='leaf-doctor-wieghts.hdf5', \n    verbose=1, \n    save_best_only=True\n)\nreduce_lr = ReduceLROnPlateau(\n    monitor='val_loss', \n    factor=0.2, \n    patience=2, \n    verbose=1, \n    mode='min'\n)\nearlystopping = EarlyStopping(\n    monitor='val_loss', \n    mode='min', \n    verbose=1,\n    patience=5\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"h = model.fit(\n    train_data_batch,\n    epochs =15, \n    validation_data=valid_data_batch,  \n    callbacks=[checkpointer, earlystopping, reduce_lr]\n) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"i = 1\nplt.figure(figsize=(14,5))\nfor metric in ['loss', 'sparse_categorical_accuracy']:\n    plt.subplot(1,2,i)\n    plt.plot(h.history[metric], marker='o', linestyle='--', label=metric)\n    plt.plot(h.history['val_' + metric], marker='o', linestyle='--', label='val_' + metric)\n    plt.xlabel('EPOCH')\n    plt.ylabel(metric.upper())\n    plt.legend()\n    plt.title(metric.upper() +' Vs EPOCH')\n    i+=1\n    \nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#storing model as json\nmodel_json = model.to_json()\nwith open('leaf-doctor-model.json', 'w') as json_file:\n    json_file.write(model_json)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 5: Testing & Submitting Prediction"},{"metadata":{"trusted":true},"cell_type":"code","source":"import glob","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_images = glob.glob('../input/cassava-leaf-disease-classification/test_images/*.jpg')\nprint(test_images)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.DataFrame(np.array(test_images), columns=['Path'])\ndf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = tf.data.Dataset.from_tensor_slices((df.Path.values))\n\ndef process_test(image_path):\n    # load the raw data from the file as a string\n    img = tf.io.read_file(image_path)\n    img = tf.image.decode_jpeg(img, channels=3)\n    img = tf.image.random_brightness(img, 0.3)\n    img = tf.image.random_flip_left_right(img, seed=None)\n    img = tf.image.random_flip_up_down(img)\n    img = tf.image.random_crop(img, size=[row,col, 3])\n    return img\n    \ndf = df.map(process_test, num_parallel_calls=tf.data.experimental.AUTOTUNE).batch(bs)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred = model.predict(df, workers=16, verbose=1)\npred","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred = np.argmax(pred, axis=-1)\n\nsub = pd.read_csv('../input/cassava-leaf-disease-classification/sample_submission.csv')\nsub['label'] = pred\nsub","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<h1 style=\"text-align:justify\"> If my notebook was helpful then plese upvote, this will keep me motivated :)\n- Drop comment for any doubts</h1>"}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}