{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport os\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n\nimport tensorflow as tf\n# config = tf.compat.v1.ConfigProto()\n# config.gpu_options.allow_growth = True\n# session =tf.compat.v1.InteractiveSession(config=config)\n\nimport matplotlib.pyplot as plt\n\nfrom PIL import Image\nfrom tensorflow.keras.models import Sequential,Model\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import *\nfrom tensorflow.python.keras.applications.efficientnet import EfficientNetB4\nfrom tensorflow.keras.applications.resnet50 import preprocess_input as process_resnet\nfrom tensorflow.keras.applications.densenet import preprocess_input as process_densenet\nfrom tensorflow.keras.applications.efficientnet import preprocess_input as process_efficientnet\nfrom tensorflow.keras.optimizers import Nadam\nfrom tensorflow.keras.callbacks import EarlyStopping\n\nprint(tf.__version__)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Loading the datasets**","metadata":{}},{"cell_type":"code","source":"general_path = '../input/cassava-leaf-disease-classification/'\n\ntrain = pd.read_csv(general_path + 'train.csv')\ntrain['label'] = train['label'].astype('str')\ntrain.sample(5)","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"names_of_disease = pd.read_json(general_path + 'label_num_to_disease_map.json', typ='series')\nnames_of_disease","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Showing some pictures","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 12))\nfor i in range(9):\n    plt.subplot(3, 3, i + 1)\n    image = Image.open(general_path + 'train_images/' + train.iloc[i]['image_id'])\n    array = np.array(image)\n    plt.imshow(array)\n    label=train.iloc[i]['label']\n    plt.title(f'{names_of_disease[int(label)]}')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sizes = []\nfor i in range(1, len(train), 250):\n    image = Image.open(general_path + 'train_images/' + train.iloc[i]['image_id'])\n    array = np.array(image)\n    sizes.append(array.shape)\nprint('Picture size', set(sizes))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_width, img_height = 224, 224","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Augmentation","metadata":{}},{"cell_type":"code","source":"datagen = ImageDataGenerator(validation_split=0.1,\n                             rotation_range = 40,\n                             width_shift_range = 0.2,\n                             height_shift_range = 0.2,\n                             shear_range = 0.2,\n                             zoom_range = 0.2,\n                             vertical_flip=True,\n                             horizontal_flip=True)\n\ntrain_datagen_flow = datagen.flow_from_dataframe(\n    dataframe=train,\n    directory=general_path + 'train_images',\n    x_col='image_id',\n    y_col='label',\n    target_size=(img_width, img_height),\n    batch_size=64,\n    subset='training',\n    seed=12345)\n\n\nvalid_datagen_flow = datagen.flow_from_dataframe(\n    dataframe=train,\n    directory=general_path + 'train_images',\n    x_col='image_id',\n    y_col='label',\n    target_size=(img_width, img_height),\n    batch_size=64,\n    class_mode = 'categorical',\n    subset='validation',\n    seed=12345)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"current_balance = train['label'].value_counts(normalize=True)\ncurrent_balance","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Models","metadata":{}},{"cell_type":"code","source":"SHAPE = (224,224,3)\n\ninp = Input(SHAPE)\n##############################################################\n######################Model One###############################\n\n #one\nconv_one_0 = Conv2D(filters=16, kernel_size=(5, 5), activation='relu')(inp)\nbatch_one_0 = BatchNormalization(axis=3)(conv_one_0)\n\nconv_one_1 = Conv2D(filters=16, kernel_size=(5, 5), activation='relu')(batch_one_0)\nmaxpool_one = MaxPooling2D(pool_size=(2, 2))(conv_one_1)\nbatch_one_1 = BatchNormalization(axis=3)(maxpool_one)\ndrop_one = Dropout(0.25)(batch_one_1)\n    \n#two\nconv_two_0 = Conv2D(filters=32, kernel_size=(5, 5), activation='relu')(drop_one)\nbatch_two_0 = BatchNormalization(axis=3)(conv_two_0)\n    \nconv_two_1 = Conv2D(filters=32, kernel_size=(5, 5), activation='relu')(batch_two_0)\nmaxpool_two = MaxPooling2D(pool_size=(2, 2))(conv_two_1)\nbatch_two_1 = BatchNormalization(axis=3)(maxpool_two)\ndrop_two = Dropout(0.25)(batch_two_1)\n    \n#three\nflat_three = Flatten()(drop_two)\n# dense_three_0 = Dense(128, activation='relu')(flat_three)  # Fully connected layer\n# batch_three_0 = BatchNormalization(axis=3)(dense_three_0)\n# drop_three_0 = Dropout(0.5)(batch_three_0)\n    \n# dense_three_1 = Dense(60, activation=\"relu\")(drop_three_0)  # Fully connected layer\n# batch_three_1 = BatchNormalization(axis=3)(dense_three_1)\n# out = Dropout(0.5)(batch_three_1)\n\none= Model(inputs=inp,outputs=flat_three)\n#one.summary()  \n   ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##############################################################\n######################Model Two############################### \n    \n#one\nconv_one_0_1 = Conv2D(filters=64, kernel_size=(3, 3), activation='relu')(inp)\nbatch_one_0_1= BatchNormalization(axis=3)(conv_one_0_1)\n    \nconv_one_1_1 = Conv2D(filters=64, kernel_size=(3, 3), activation='relu')(batch_one_0_1)\nmaxpool_one_1 = MaxPooling2D(pool_size=(2, 2))(conv_one_1_1)\nbatch_one_1_1 = BatchNormalization(axis=3)(maxpool_one_1)\ndrop_one      = Dropout(0.25)(batch_one_1_1)\n\n#two\nconv_two_0_1 = Conv2D(filters=128, kernel_size=(3, 3), activation='relu')(batch_one_1_1)\nbatch_two_0_1 = BatchNormalization(axis=3)(conv_two_0_1)\n    \nconv_two_1_1 = Conv2D(filters=128, kernel_size=(3, 3), activation='relu')(batch_two_0_1)\nmaxpool_two_1 = MaxPooling2D(pool_size=(2, 2))(conv_two_1_1)\nbatch_two_1_1 = BatchNormalization(axis=3)(maxpool_two_1)\ndrop_two = Dropout(0.25)(batch_two_1_1)\n    \n#three\nconv_three_0_1 = Conv2D(filters=256, kernel_size=(3, 3), activation='relu')(batch_two_1_1)\nbatch_three_0_1 = BatchNormalization(axis=3)(conv_three_0_1)\n    \nconv_three_1_1 = Conv2D(filters=256, kernel_size=(3, 3), activation='relu')(batch_three_0_1)\nmaxpool_three_1 = MaxPooling2D(pool_size=(2, 2))(conv_three_1_1)\nbatch_three_1_1 = BatchNormalization(axis=3)(maxpool_three_1)\ndrop_three = Dropout(0.25)(batch_three_1_1)\n    \n#four\nflat_four_1 = Flatten()(batch_three_1_1)\n# dense_four_0_1 = Dense(60, activation='relu')(flat_four_1)  # Fully connected layer\n# #batch_four_0_1 = BatchNormalization()(dense_four_0_1)\n# out_2 = Dropout(0.5)(dense_four_0_1)\n    \n#dense_four_1_1 = Dense(60, activation=\"relu\")(drop_four_0_1)  # Fully connected layer\n#batch_four_1_1 = BatchNormalization()(dense_four_1_1)\n#out_2 = Dropout(0.5)(batch_four_1_1)\n\ntwo= Model(inputs=inp,outputs=flat_four_1)\n#two.summary()    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Custom layer for weighted learnable ensemble ###\n\nclass weightedEnsemble(tf.keras.layers.Layer):\n    def __init__(self,n_output):\n        super(weightedEnsemble,self).__init__()\n        self.W = tf.Variable(initial_value = tf.random.uniform(shape=[1,1,n_output],minval=0,maxval=1),trainable=True)\n        \n    def call(self,inputs):\n        #inputs is list of tensor of shape[(n_batch,n_feat), ..., (n_batch,n_feat)]\n        #expand last dim of each input passed [(n_batch, n_feat, 1), ...,(n_batch, n_feat, 1)]\n        inputs = [tf.expand_dims(i,-1) for i in inputs]\n        inputs = Concatenate(axis=-1)(inputs) #(n_batch, n_feat, n_inputs)\n        weights = tf.nn.softmax(self.W, axis=-1) #(1,1,n_inputs)\n        # weights sum up to one on last dim\n        return tf.reduce_sum(weights*inputs, axis=-1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#weighted ensemble\n# partial_output = [one.outputs[0],two.output]\n# x = weightedEnsemble(n_output=len(partial_output))(partial_output)\n\nmerge = Concatenate()([one.outputs[0],two.outputs[0]])\n#merge = Average()([one.outputs[0],two.outputs[0]])\nhidden1 = Dense(512, activation='relu')(merge)\nhidden2 = Dense(512, activation='relu')(hidden1)\n\noutput = Dense(5, activation='softmax')(hidden2)\nmodel = Model(inputs=inp, outputs = output, name='ensemble')\nopt = tf.keras.optimizers.Nadam(lr = 0.001)\nmodel.compile(loss='categorical_crossentropy', optimizer=opt, metrics='categorical_accuracy')\nmodel.summary()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"es = EarlyStopping(monitor='val_accuracy', mode='auto', restore_best_weights=True, verbose=1, patience=7)\nhistory = model.fit(train_datagen_flow,\n                    validation_data=valid_datagen_flow, \n                    epochs=50,\n                    #callbacks=[early_stop, reduce_lr], \n                    #use_multiprocessing=True,\n                    #shuffle=True,\n                    callbacks=[es],\n                    verbose=2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"model.save('cassava_model'+'.h5') ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# visualization of loss and accuracy","metadata":{}},{"cell_type":"code","source":"def visualize_training(history, lw = 2):\n    plt.figure(figsize=(10,10))\n    plt.subplot(2,1,1)\n    plt.plot(history.history['categorical_accuracy'], label = 'training', marker = '*', linewidth = lw)\n    plt.plot(history.history['val_categorical_accuracy'], label = 'validation', marker = 'o', linewidth = lw, color='red')\n    plt.title('Accuracy Comparison')\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy')\n    plt.grid(True)\n    plt.legend(fontsize = 'x-large')\n    \n\n    plt.subplot(2,1,2)\n    plt.plot(history.history['loss'], label = 'training', marker = '*', linewidth = lw)\n    plt.plot(history.history['val_loss'], label = 'validation', marker = 'o', linewidth = lw, color='red')\n    plt.title('Loss Comparison')\n    plt.xlabel('Epochs')\n    plt.ylabel('Loss')\n\n    plt.legend(fontsize = 'x-large')\n    plt.grid(True)\n    plt.show()\n    \nvisualize_training(history)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# Prediction accuracy on train data\nscore_tr = model.evaluate_generator(train_datagen_flow, verbose=1)\nprint(\"Prediction accuracy on train data =\", score_tr[1])\n\n# Prediction accuracy on test data\nscore_ts = model.evaluate_generator(valid_datagen_flow, verbose=1)\nprint(\"Prediction accuracy on test data =\", score_ts[1])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(columns=['image_id','label'])\nfor image_name in os.listdir(general_path + 'test_images'):\n    image_path = os.path.join(general_path + 'test_images', image_name)\n    image = tf.keras.preprocessing.image.load_img(image_path)\n    resized_image = image.resize((img_width, img_height))\n    numpied_image = np.expand_dims(resized_image, 0)\n    tensored_image = tf.cast(numpied_image, tf.float32)\n    prediction = model.predict(tensored_image)\n    prediction = np.argmax(prediction,axis=1)\n\n    submission = submission.append(pd.DataFrame({'image_id': image_name,\n                                                 'label': prediction}))\n    \nsubmission","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}