{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Cassava Leaf Disease Classification\n\n* Multi class classification\n* The input dataset is provided with 21,000 images with labels\n* Expected outcome is classifying the images into 5 types\n    * 0: 'Cassava Bacterial Blight (CBB)'\n    * 1: 'Cassava Brown Streak Disease (CBSD)'\n    * 2: 'Cassava Green Mottle (CGM)'\n    * 3: 'Cassava Mosaic Disease (CMD)'\n    * 4: 'Healthy'"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Cassava Leaf Disease Classification"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# Importing  necessary tools\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport tensorflow as tf\nimport tensorflow_hub as hub\nprint('TF version:',tf.__version__)\nprint('TF Hub version:',hub.__version__)\n\n# Check for GPU availbaility, TF inbuilt will initiate GPU and all codes of TF runs on GPU\nprint(\"GPU Available (Yes!!!!)\" if tf.config.list_physical_devices(\"GPU\") else \"GPU not available\")\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Exploratory Data Analysis\n\nExtracting file paths of training images and its respective labels\n\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"labels = pd.read_csv('../input/cassava-leaf-disease-classification/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# To get the info about types of disease\n\ndisease_mapping = pd.read_json('../input/cassava-leaf-disease-classification/label_num_to_disease_map.json',orient='index')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"disease_mapping = dict(disease_mapping[0])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"disease_mapping","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels['label'].value_counts(), len(labels)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Visualizing the given training images and how many images are provided per each class**"},{"metadata":{"trusted":true},"cell_type":"code","source":"labels['label'].value_counts().plot.bar(rot=0,color='red');\nplt.title('Images per class')\nplt.xlabel('Disease Type')\nplt.ylabel('Count');","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Lets get the filenames of each image in training folder\n\nfilenames = ['../input/cassava-leaf-disease-classification/train_images/' + fname for fname in labels['image_id']]\nfilenames[:10], len(filenames)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Checking whether filenames and original images in train folder are equal\nimport os\nif len(os.listdir('../input/cassava-leaf-disease-classification/train_images')) == len(filenames):\n    print(\"File names are matching with original folder!!!\")\nelse:\n    print(\"Filenames are not matching!\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels['classname']  = labels['label'].map(disease_mapping)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_labels = labels['classname']\nimage_labels = np.array(image_labels)\nimage_labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(image_labels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Checking whether labels are matching with filenames\n\nif len(image_labels) == len(filenames):\n    print('image labels and filenames are matching')\nelse:\n    print('Not matching')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# To convert labels to boolean which helps model to train\n\nunqiue_diseases = np.unique(image_labels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"unqiue_diseases","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_labels[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_labels[0] == unqiue_diseases","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"boolean_labels = [image_label == unqiue_diseases for image_label in image_labels]\nboolean_labels[:3]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(boolean_labels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(filenames)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Modelling and Experimentation"},{"metadata":{"trusted":true},"cell_type":"code","source":"#Setting up variables\nX = filenames\ny = boolean_labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# setting number of images to be used for experimenting\nNUM_IMAGES = 21397","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Split them into training and validation sets of total NUM_iMAGES size\nX_train, X_val, y_train, y_val = train_test_split(X[:NUM_IMAGES],\n                                                 y[:NUM_IMAGES], test_size=0.1,\n                                                random_state=42)\n\nlen(X_train), len(y_train), len(X_val), len(y_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_train[:3], y_train[:3]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Preprocessing Images(Turning images into tensors)\n\nTo preprocess our images into tensors we're gonna write a function which does a few operations:\n1. Take an image filepath as input\n2. use Tensorflow to read the file and save it varibale, `image`\n3. Turn our `image` (a jpg) into tensors\n4. Normalize our image(convert colour channel values from 0-255 to 0-1) \n5. Resize the image to a shape of (224,224)\n6. Return the modified `image`"},{"metadata":{"trusted":true},"cell_type":"code","source":"IMG_SIZE = 600\n\n# create a function to preprocess images\ndef process_image(image_path, img_size=IMG_SIZE):\n    \"\"\"\n    Takes an image file path and turns image into tensor\n    \"\"\"\n    #Read a image file\n    image = tf.io.read_file(image_path)\n    #Turn the jpeg image into numerical Tensor with 3 colour channels(R,G,B)\n    image = tf.image.decode_jpeg(image, channels=3)\n    #Convert the color channel values from 0-255 to 0-1\n    image = tf.image.convert_image_dtype(image, tf.float32)  #Normalizing the values for higher efficiency\n    #resize our image to desired value(600,600)\n    image = tf.image.resize(image, size=[IMG_SIZE,IMG_SIZE])\n\n    return image","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# create a function to return tuple (image,label)\ndef get_image_label(image_path, label):\n    \"\"\"\n    Takes an image file path and its associated label,\n    process the image and return the tuple(image,label)\n    \"\"\"\n    image = process_image(image_path)\n    return image,label","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#define thw batch size, 32 is a good start\nBATCH_SIZE = 64\n\n#create a function to turn data into batches\ndef create_data_batches(X, y=None, batch_size=BATCH_SIZE, valid_data=False, test_data=False):\n    \"\"\"\n    Creates batches out of image(X) and label(y) pairs\n    Shuffles the data if it's training data but doesn't shuffle if it's validation data.\n    Also accepts the test data as input\n    \"\"\"\n    #If data is test dataset, we probably dont have labels\n    if test_data:\n        print('Creating test data batches....')\n        data = tf.data.Dataset.from_tensor_slices((tf.constant(X))) #only filepaths\n        data_batch = data.map(process_image).batch(BATCH_SIZE)  #preprocessing the images and dividing into the batches\n        return data_batch\n\n    #if it is validation dataset, we dont need to shuffle\n    elif valid_data:\n        print('Creating validation data batches....')\n        data = tf.data.Dataset.from_tensor_slices((tf.constant(X), #filepaths\n                                                   tf.constant(y))) #labels\n        data_batch = data.map(get_image_label).batch(BATCH_SIZE)\n        return data_batch\n\n    else:\n        print('Creating training data batches...')\n        #turn filepaths and labels into tensors\n        data = tf.data.Dataset.from_tensor_slices((tf.constant(X),\n                                                   tf.constant(y)))\n\n        #Shuffling pathnames and labels before mapping image processor is faster than shuffing the images\n        data = data.shuffle(buffer_size=len(X)) #shuffles all filenames and labels which are tensors form\n\n        #create (image,label) tuples and this also turns shuffled image paths and labels into preproccesed images and labels\n        data = data.map(get_image_label)\n\n        #Tuning the training data into batches\n        data_batch = data.batch(BATCH_SIZE)\n\n        return data_batch\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Create training and validation data batches\ntrain_data = create_data_batches(X_train, y_train)\nvalid_data = create_data_batches(X_val, y_val, valid_data=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Let's check element spec \ntrain_data.element_spec, valid_data.element_spec","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**To visualize the data batches**"},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n#create a function for viewing images in batches\ndef show_25_images(images,labels):\n    \"\"\"\n    Displays plot of 25 images and labels from data batch\n    \"\"\"\n    #setup the figure\n    plt.figure(figsize=(12,12))\n    #Loop through 25 images for displaying them\n    for i in range(25):\n        #create subplots of 5 rows and 5 columns\n        ax = plt.subplot(5,5,i+1)\n        #display am image\n        plt.imshow(images[i])\n        # Add label as the title\n        plt.title(unqiue_diseases[labels[i].argmax()])\n        # Turn the grid lines off\n        plt.axis('off')\n    plt.tight_layout(w_pad=1.5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# To get first data batch of the train data\ntrain_images, train_labels = next(train_data.as_numpy_iterator())\ntrain_images, train_labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Let's visualize\nshow_25_images(train_images,train_labels)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Building a model\n\nBefore building our model, we need to define few things:\n* The input shape (our image shape in form of Tensors) of our model.\n* The output shape (image labels in form of Tensors) of our model.\n* The URL of the model we want use is from Tensorflow hub: https://tfhub.dev/google/imagenet/mobilenet_v2_130_224/classification/4"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Setup input shape to the model\nINPUT_SHAPE = [None, IMG_SIZE, IMG_SIZE, 3] #batch, height,width,colour channels\n\n#Setup output shape of our model\nOUTPUT_SHAPE = len(unqiue_diseases)\n\n\n# setup model URL from Tensorflow Hub\nMODEL_URL = \"https://tfhub.dev/google/bit/m-r50x1/1\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n# creating function to build keras model\ndef create_model(input_shape=INPUT_SHAPE,output_shape=OUTPUT_SHAPE,model_url=MODEL_URL):\n    print(\"Building the model....\",model_url)\n\n    # setup model layers\n    model = tf.keras.Sequential([\n    hub.KerasLayer(model_url), #layer 1(input layer)\n    #tf.keras.layers.Dense(1024, activation='relu', name='hidden_layer'),\n    tf.keras.layers.Dense(units=output_shape,\n                        activation='softmax') # Layer 2 (output layer) it will be in shape of 120 since 120 breeds\n    ])\n\n    # Compile the model\n    model.compile(\n        loss=tf.keras.losses.CategoricalCrossentropy(),\n        optimizer=tf.keras.optimizers.Adam(),  #mostly used optimizer in many models\n        metrics=['accuracy']  #parameter to mesaure to prediction is correct or wrong\n    )\n\n    # Build the model\n    model.build(input_shape)  #This was the input where mobilev2 model is trained ON\n\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = create_model()\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Load Tensorboard notebook extension\n%load_ext tensorboard","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Tensorboard callback\nimport datetime\n\n#create a function to build Tensorboard callback\ndef create_tensorboard_callback():\n    # create a log dictionary for storing Tensorboard logs\n    logdir = os.path.join(\"./\",'logs_', #Make sure we get the logs whenever we run experiment\n                           datetime.datetime.now().strftime(\"%Y%m%d-%H%M%S\"))\n    \n    return tf.keras.callbacks.TensorBoard(logdir)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# create early stopping call back\n\nearly_stopping = tf.keras.callbacks.EarlyStopping(monitor='accuracy', #we gonna monitor validation accuracy metric\n                                                  patience=2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"NUM_EPOCHS = 100","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Build a function to train and return a trained model\ndef train_model():\n    \"\"\"\n    Trains a given model and return the trained version\n    \"\"\"\n    #create a model\n    create_model()\n\n    # create new Tensorboard session everytime we train\n    tensorboard = create_tensorboard_callback()\n\n    #Fit the model to data and also passing the callbacks\n    model.fit(x=train_data,\n              epochs=NUM_EPOCHS,\n              validation_data=valid_data,\n              validation_freq=1,\n              callbacks=[tensorboard,early_stopping])\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#model = train_model()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## training on full datset\nfull_data = create_data_batches(X,y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"full_data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tensorboard = create_tensorboard_callback()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''\nmodel.fit(x=full_data,\n         epochs=NUM_EPOCHS,\n              verbose=1,\n              callbacks=[tensorboard,early_stopping])\n'''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def save_model(model,suffix=None):\n    \"\"\"\n    Savea a given model in models directory and appends a suffix(string)\n    \"\"\"\n    \n    #create a model directory pathname with current time\n    modeldir = os.path.join('./',\n                            datetime.datetime.now().strftime(\"%Y%m%d-%H%M%s\"))\n    \n    model_path = modeldir + \"-\" + suffix + \".h5\" #save the fomat of model .h5 extension\n    print(f\"Saving model to {model_path}....\")\n    model.save(model_path)\n\n    return model_path","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_model(model_path):\n    \"\"\"\n    Loads the model from saved path\n    \"\"\"\n    print(f\"Loading the saved model from {model_path}..\")\n    model = tf.keras.models.load_model(model_path,\n                                       custom_objects={'KerasLayer':hub.KerasLayer})  #since we have used tf hub model, we have to mention \n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_path = save_model(model, suffix=\"full-model\") ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = load_model(model_path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_path = '../input/cassava-leaf-disease-classification/test_images/'\ntest_filenames = ['../input/cassava-leaf-disease-classification/test_images/' + fname for fname in os.listdir(test_path)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_filenames[:3], len(test_filenames)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_data = create_data_batches(test_filenames,test_data=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions = model.predict(test_data, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions[:5], len(predictions)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# creating dataframe for submission file\npreds_df = pd.DataFrame(columns=['image_id','label'])\npreds_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_ids = [os.path.splitext(path)[0] + os.path.splitext(path)[1] for path in os.listdir(test_path)]\npreds_df['image_id'] = test_ids","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#valid_ids = [id.split('../input/cassava-leaf-disease-classification/train_images/')[1] for id in X_val]\n#preds_df['image_id'] = valid_ids","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#preds_df['label'] = [prediction.argmax(axis=1) for prediction in predictions]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds_df['label'] = predictions.argmax(axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#preds_df['label'] = [np.argmax(prediction) for prediction in predictions]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds_df.dtypes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds_df.to_csv('./submission.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}