{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport glob\nimport shutil\nimport json\nimport keras\nimport itertools\n\nimport numpy as np\nimport pandas as pd\n\nimport seaborn as sns\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nimport plotly.figure_factory as ff\n\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nfrom scipy.signal import convolve2d\n%matplotlib inline\n\nimport imageio\nfrom collections import Counter\n\nfrom keras.preprocessing.image import ImageDataGenerator\nimport tensorflow as tf\nfrom keras.models import Sequential\nfrom keras.layers import GlobalAveragePooling2D, Flatten, Dense, Dropout\nfrom keras.optimizers import RMSprop, Adam\nfrom keras.callbacks import EarlyStopping, ModelCheckpoint, ReduceLROnPlateau\nfrom tensorflow.keras.applications.resnet50 import ResNet50\nfrom tensorflow.keras.applications import EfficientNetB3\n\n\n# Defining the working directories\n\nwork_dir = '../input/cassava-leaf-disease-classification/'\nos.listdir(work_dir) \ntrain_path = '/kaggle/input/cassava-leaf-disease-classification/train_images'","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Loading the data"},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv(work_dir + 'train.csv')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Importing the json file with label and mapping the class labels"},{"metadata":{"trusted":true},"cell_type":"code","source":"disease_names = open('../input/cassava-leaf-disease-classification/label_num_to_disease_map.json')\ndisease_names = json.load(disease_names)\ndf['disease_name'] = df['label'].apply(lambda x: disease_names[str(x)])\n#visualize the top five rows from table\ndf.head()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.info()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Calculating the number of label's per disease"},{"metadata":{"trusted":true},"cell_type":"code","source":"df.disease_name.value_counts()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Visualizing the Distribution of Class Labels"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = make_subplots(rows=1, cols=2,\n            specs=[[{\"type\": \"xy\"}, {\"type\": \"domain\"}]],)\n# value_counts: to count number of images in each class with respect to disease_name column\n# Bar plot \nt1 = go.Bar(x=df['disease_name'].value_counts().index, \n            y=df['disease_name'].value_counts().values,\n            text=df['disease_name'].value_counts().values,\n            textposition='auto',name='Count',\n           marker_color='indianred')\n#Pie chart with labels and counts\nt2 = go.Pie(labels=df['disease_name'].value_counts().index,\n           values=df['disease_name'].value_counts().values,\n           hole=0.3)\nfig.add_trace(t1,row=1, col=1)\nfig.add_trace(t2,row=1, col=2)\nfig.update_layout(title='Distribution of Class Labels')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Now let’s load an image and observe its various properties in general."},{"metadata":{"trusted":true},"cell_type":"code","source":"pic = imageio.imread('../input/cassava-leaf-disease-classification/train_images/1004105566.jpg')\nplt.figure(figsize = (5,5))\nplt.imshow(pic)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Observing the basic properties of the image"},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Type of the image : ' , type(pic))\n\nprint('Shape of the image : {}'.format(pic.shape))\nprint('Image Height : {}'.format(pic.shape[0]))\nprint('Image Width : {}'.format(pic.shape[1]))\nprint('Dimension of Image : {}'.format(pic.ndim))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* The shape of the ndarray shows that it is a three-layered matrix. The first two numbers here are length and width, and the third number (i.e. 3) is for three layers: Red, Green, Blue. So, if we calculate the size of an RGB image, the total size will be counted as height x width x 3"},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Image size {}'.format(pic.size))\nprint('Maximum RGB value in this image {}'.format(pic.max()))\nprint('Minimum RGB value in this image {}'.format(pic.min()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# A specific pixel located at Row : 100 ; Column : 50 \n\n# Each channel's value of it, gradually R , G , B\n\nprint('Value of only R channel : {}'.format(pic[ 100, 50, 0]))\nprint('Value of only G channel : {}'.format(pic[ 100, 50, 1]))\nprint('Value of only B channel : {}'.format(pic[ 100, 50, 2]))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Okay, now let’s take a quick view of each channel in the whole image."},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.title('R-CHANNEL')\nplt.ylabel('Height {}'.format(pic.shape[0]))\nplt.xlabel('Width {}'.format(pic.shape[1]))\nplt.imshow(pic[ : , : , 0])\nplt.show()\n\nplt.title('G-CHANNEL')\nplt.ylabel('Height {}'.format(pic.shape[0]))\nplt.xlabel('Width {}'.format(pic.shape[1]))\nplt.imshow(pic[ : , : , 1])\nplt.show()\n\nplt.title('B-CHANNEL')\nplt.ylabel('Height {}'.format(pic.shape[0]))\nplt.xlabel('Width {}'.format(pic.shape[1]))\nplt.imshow(pic[ : , : , 2])\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Splitting Layers\n\n* Now, we know that each pixel of the image is represented by three integers. Splitting the image into separate color components is just a matter of pulling out the correct slice of the image array."},{"metadata":{"trusted":true},"cell_type":"code","source":"pic = imageio.imread('../input/cassava-leaf-disease-classification/train_images/1004105566.jpg')\nfig, ax = plt.subplots(nrows = 1, ncols=3, figsize=(15,5))\n\nfor c, ax in zip(range(3), ax):\n    # create zero matrix\n    split_img = np.zeros(pic.shape, dtype=\"uint8\") # 'dtype' by default: 'numpy.float64'\n    \n    # assing each channel \n    split_img[ :, :, c] = pic[ :, :, c]\n    \n    # display each channel\n    ax.imshow(split_img)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Image Processing\n\n\nOne of MOOC course on edX, we’ve introduced with some satellite images and its processing system. It’s very informative of course. However, let’s do a few analysis tasks on it\n\nThere’s something interesting about this image. Like many other visualizations, the colors in each RGB layer mean something. For example, the intensity of the red will be an indication of altitude of the geographical data point in the pixel. The intensity of blue will indicate a measure of aspect, and the green will indicate slope. These colors will help communicate this information in a quicker and more effective way rather than showing numbers.\n\n* Red pixel indicates: Altitude\n* Blue pixel indicates: Aspect\n* Green pixel indicates: Slope\n\nThere is, by just looking at this colorful image, a trained eye that can tell already what the altitude is, what the slope is, and what the aspect is. So, that’s the idea of loading some more meaning to these colors to indicate something more scientific\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"\n# Only Red Pixel value , higher than 180\npic = imageio.imread('../input/cassava-leaf-disease-classification/train_images/1004105566.jpg')\nred_mask = pic[:, :, 0] < 180\npic[red_mask] = 0\nplt.figure(figsize=(5,5))\nplt.imshow(pic)\n\n# Only Green Pixel value , higher than 180\npic = imageio.imread('../input/cassava-leaf-disease-classification/train_images/1004105566.jpg')\ngreen_mask = pic[:, :, 1] < 180\npic[green_mask] = 0\nplt.figure(figsize=(5,5))\nplt.imshow(pic)\n\n# Only Blue Pixel value , higher than 180\npic = imageio.imread('../input/cassava-leaf-disease-classification/train_images/1004105566.jpg')\nblue_mask = pic[:, :, 2] < 180\npic[blue_mask] = 0\nplt.figure(figsize=(5,5))\nplt.imshow(pic)\n\n# Composite mask using logical_and\npic = imageio.imread('../input/cassava-leaf-disease-classification/train_images/1004105566.jpg')\nfinal_mask = np.logical_and(red_mask, green_mask, blue_mask)\npic[final_mask] = 40\nplt.figure(figsize=(5,5))\nplt.imshow(pic)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Splitting the data"},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train, val = train_test_split(df, test_size=0.1, random_state=42, stratify=df['disease_name'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Data Pre-Processing\n\n* Now we need to read all the images using Image Data Generator\n* When we'll be reading the data from the folders we also need to make sure that we need to do some Data Augmentation.\n* The Data Augmentation can be done by using ImageDataGenerator library.\n* The ImageDataGenerator what is does that it applies the Data Augmentation techniques like zooming, scaling, horizontal flipping, vertical flipping, etc.\n* **IN THE TEST DATA WE SHOULD NEVER PERFORM DATA AUGMENTATION WE SHOULD ONLY PERFORM SCALING.**\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"Image_Width = 224\nImage_Height = 224\nImage_Size = (Image_Width, Image_Height)\nImage_Channel = 3\nn_CLASS = 5","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"datagen = ImageDataGenerator(\n                    rotation_range = 40,\n                    width_shift_range = 0.2,\n                    height_shift_range = 0.2,\n                    shear_range = 0.2,\n                    zoom_range = 0.2,\n                    horizontal_flip = True,\n                    vertical_flip = True,\n                    fill_mode = 'nearest')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_set = datagen.flow_from_dataframe(train,\n                         directory = train_path,\n                         seed=42,\n                         x_col = 'image_id',\n                         y_col = 'disease_name',\n                         target_size = Image_Size,\n                         #color_mode=\"rgb\",\n                         class_mode = 'categorical',\n                         interpolation = 'nearest',\n                         shuffle = True,\n                         batch_size = 32)\n\nval_set = datagen.flow_from_dataframe(val,\n                         directory = train_path,\n                         seed=42,\n                         x_col = 'image_id',\n                         y_col = 'disease_name',\n                         target_size = Image_Size,\n                         #color_mode=\"rgb\",\n                         class_mode = 'categorical',\n                         interpolation = 'nearest',\n                         shuffle = True,\n                         batch_size = 32)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Creating Our Model"},{"metadata":{"trusted":true},"cell_type":"code","source":"def create_model():\n    \n    model = Sequential()\n    # initialize the model with input shape as (224,224,3)\n    model.add(EfficientNetB3(input_shape = (Image_Width, Image_Height, 3), include_top = False, weights = 'imagenet'))\n    #for layer in model.layers[:-40]:  # Training just part of the architecture do not optimize the performance\n    #    layer.trainable = False\n    model.add(GlobalAveragePooling2D())\n    model.add(Flatten())\n    model.add(Dense(512, activation = 'relu', bias_regularizer=tf.keras.regularizers.L1L2(l1=0.01, l2=0.001)))\n    model.add(Dropout(0.7))\n    model.add(Dense(n_CLASS, activation = 'softmax'))\n    \n    return model\n\nleaf_model = create_model()\nleaf_model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"leaf_model = create_model()\nleaf_model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"EPOCHS = 10\nSTEP_SIZE_TRAIN = train_set.n//train_set.batch_size\nSTEP_SIZE_VALID = val_set.n//val_set.batch_size","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Training our Model"},{"metadata":{"trusted":true},"cell_type":"code","source":"def Model_fit():\n    \n    #leaf_model = None\n    \n    leaf_model = create_model()\n    \n    '''Compiling the model'''\n    \n    loss = tf.keras.losses.CategoricalCrossentropy(from_logits = False,\n                                                   label_smoothing=0.0001,\n                                                   name='categorical_crossentropy' )\n    \n    leaf_model.compile(optimizer = Adam(learning_rate = 1e-3),\n                        loss = loss, #'categorical_crossentropy'\n                        metrics = ['categorical_accuracy']) #'acc'\n    \n    # Stop training when the val_loss has stopped decreasing for 5 epochs.\n    es = EarlyStopping(monitor='val_loss', mode='min', patience=5,\n                       restore_best_weights=True, verbose=1)\n    \n    # Save the model with the minimum validation loss\n    checkpoint_cb = ModelCheckpoint(\"Cassava_best_model.h5\",\n                                    save_best_only=True,\n                                    monitor = 'val_loss',\n                                    mode='min')\n    \n    # reduce learning rate\n    reduce_lr = ReduceLROnPlateau(monitor = 'val_loss',\n                                  factor = 0.3,\n                                  patience = 3,\n                                  min_lr = 1e-5,\n                                  mode = 'min',\n                                  verbose = 1)\n    \n    history = leaf_model.fit(train_set,\n                             validation_data = val_set,\n                             epochs= EPOCHS,\n                             batch_size = 32,\n                             #class_weight = d_class_weights,\n                             steps_per_epoch = STEP_SIZE_TRAIN,\n                             validation_steps = STEP_SIZE_VALID,\n                             callbacks=[es, checkpoint_cb, reduce_lr])\n    \n    leaf_model.save('Cassava_model'+'.h5')  \n    \n    return history","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"results = Model_fit()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#%% CHECKING THE METRIC\n\nprint('Train_Cat-Acc: ', max(results.history['categorical_accuracy']))\nprint('Val_Cat-Acc: ', max(results.history['val_categorical_accuracy']))\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Plotting The Results"},{"metadata":{"trusted":true},"cell_type":"code","source":"def Train_Val_Plot(acc,val_acc,loss,val_loss):\n    \n    fig, (ax1, ax2) = plt.subplots(1,2, figsize= (15,10))\n    fig.suptitle(\" MODEL'S METRICS VISUALIZATION \", fontsize=20)\n\n    ax1.plot(range(1, len(acc) + 1), acc)\n    ax1.plot(range(1, len(val_acc) + 1), val_acc)\n    ax1.set_title('History of Accuracy', fontsize=15)\n    ax1.set_xlabel('Epochs', fontsize=15)\n    ax1.set_ylabel('Accuracy', fontsize=15)\n    ax1.legend(['training', 'validation'])\n\n\n    ax2.plot(range(1, len(loss) + 1), loss)\n    ax2.plot(range(1, len(val_loss) + 1), val_loss)\n    ax2.set_title('History of Loss', fontsize=15)\n    ax2.set_xlabel('Epochs', fontsize=15)\n    ax2.set_ylabel('Loss', fontsize=15)\n    ax2.legend(['training', 'validation'])\n    plt.show()\n    \n\nTrain_Val_Plot(results.history['categorical_accuracy'],results.history['val_categorical_accuracy'],\n               results.history['loss'],results.history['val_loss'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Evaluating our model"},{"metadata":{"trusted":true},"cell_type":"code","source":"import keras\n\nfinal_model = keras.models.load_model('Cassava_best_model.h5')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Making predictions on the test data"},{"metadata":{"trusted":true},"cell_type":"code","source":"\n\nTEST_DIR = '../input/cassava-leaf-disease-classification/test_images/'\ntest_images = os.listdir(TEST_DIR)\npredictions = []\n\nfor image in test_images:\n    img = Image.open(TEST_DIR + image)\n    img = img.resize((Image_Width, Image_Height))\n    img = np.expand_dims(img, axis=0)\n    predictions.extend(final_model.predict(img).argmax(axis = 1))\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Creating the csv for final submission"},{"metadata":{"trusted":true},"cell_type":"code","source":"\nsub = pd.DataFrame({'image_id': test_images, 'label': predictions})\ndisplay(sub)\nsub.to_csv('submission.csv', index = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}