{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import random\nfrom numpy.random import seed\nfrom tensorflow.random import set_seed\n\nseed_value = 42\nrandom.seed(seed_value)\nseed(seed_value)\nset_seed(seed_value)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nimport random\nimport os\nimport cv2\nimport sys\nfrom pylab import rcParams\nfrom PIL import Image\nfrom tqdm import tqdm\nwarnings.filterwarnings('ignore')\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import Input\nfrom tensorflow.keras.models import Model, load_model\nfrom tensorflow.keras.layers import Dense, Flatten, Dropout, Activation, Input, GlobalAveragePooling2D\nfrom tensorflow.keras.callbacks import ModelCheckpoint, ReduceLROnPlateau, EarlyStopping\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.mixed_precision import experimental as mixed_precision\nfrom sklearn.model_selection import StratifiedKFold","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train = pd.read_csv(\"../input/cassava-leaf-disease-classification/train.csv\")\ndf_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train[\"label\"] = df_train[\"label\"].astype(str) #convert to str as we want to use Categorical Cross Entropy (CCE) later on\ndf_train.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.set_style(\"whitegrid\")\nplt.figure(figsize=(10,8))\nsns.countplot(df_train[\"label\"], edgecolor=\"black\", palette=\"mako\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"batch_size=32\nimage_size=300\n\ninput_shape = (image_size, image_size, 3)\ntarget_size = (image_size, image_size)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_augmentation = tf.keras.Sequential(\n    [\n        tf.keras.layers.experimental.preprocessing.RandomCrop(image_size, image_size),\n        tf.keras.layers.experimental.preprocessing.RandomFlip(\"horizontal_and_vertical\"),\n        tf.keras.layers.experimental.preprocessing.RandomRotation(0.25),\n        tf.keras.layers.experimental.preprocessing.RandomContrast(0.2)\n    ])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def DataGenerator(train_set, val_set):\n    \n    train_datagen = ImageDataGenerator().flow_from_dataframe(\n                  dataframe = train_set,\n                  directory='../input/cassava-leaf-disease-classification/train_images',\n                  x_col='image_id',\n                  y_col='label',\n                  target_size=target_size,\n                  batch_size=batch_size,\n                  shuffle=True,\n                  class_mode='sparse',\n                  seed=seed_value)\n\n    val_datagen = ImageDataGenerator().flow_from_dataframe(\n                dataframe = val_set,\n                directory='../input/cassava-leaf-disease-classification/train_images',\n                x_col='image_id',\n                y_col='label',\n                target_size=target_size,\n                batch_size=batch_size,\n                shuffle=False,\n                class_mode='sparse',\n                seed=seed_value)\n    \n    return train_datagen, val_datagen","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"epochs = 8\ntotal_steps = (int(len(df_train)*0.8/batch_size)+1)*epochs\n\nlr = tf.keras.experimental.CosineDecay(initial_learning_rate=1e-3, decay_steps=total_steps)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def build_model():\n    base_model = EfficientNetB0(include_top=False, weights=\"imagenet\", input_shape=input_shape)\n\n    # Rebuild top\n    inputs = Input(shape=input_shape)\n    base = base_model(inputs)\n    pooling = GlobalAveragePooling2D()(base)\n    outputs = Dense(5, activation=\"softmax\", dtype='float32')(pooling) #necessary for mixed-precision training to work properly\n\n    # Compile\n    model = Model(inputs=inputs, outputs=outputs)\n    optimizer = tf.keras.optimizers.Adam(learning_rate=lr)\n    model.compile(optimizer=optimizer, loss=\"sparse_categorical_crossentropy\", metrics=['accuracy'])\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fold_number = 0\nn_splits = 5\noof_accuracy = []\n\ntf.keras.backend.clear_session()\nskf = StratifiedKFold(n_splits=n_splits, random_state=seed_value)\nfor train_index, val_index in skf.split(df_train[\"image_id\"], df_train[\"label\"]):\n    train_set = df_train.loc[train_index]\n    val_set = df_train.loc[val_index]\n    train_datagen, val_datagen = DataGenerator(train_set, val_set)\n    model = build_model()\n    print(\"Training fold no.: \" + str(fold_number+1))\n\n    model_name = \"effnetb0 \"\n    fold_name = \"fold.h5\"\n    filepath = model_name + str(fold_number+1) + fold_name\n    callbacks = [ModelCheckpoint(filepath=filepath, monitor='val_accuracy', save_best_only=True)]\n\n    history = model.fit(train_datagen, epochs=epochs, validation_data=val_datagen, callbacks=callbacks)\n    oof_accuracy.append(max(history.history[\"val_accuracy\"]))\n    fold_number += 1\n    if fold_number == n_splits:\n        print(\"Training finished!\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Average Out-Of-Fold Accuracy: {:.2f}\".format(np.mean(oof_accuracy)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"models = [] \nfor i in range(5):\n    effnet = load_model(\"./effnetb0 \" + str(i+1) + \"fold.h5\")\n    #effnet1 = load_model('../input/pretrain-effnetb0-1f/')\n#     models.append(effnet1)\n    \n    models.append(effnet)\n\nmodel_one = models[0]\nmodel_two = models[1]\nmodel_three = models[2]\nmodel_four = models[3]\nmodel_five = models[4]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv(\"../input/cassava-leaf-disease-classification/train.csv\")\nval_list = []\n\nskf = StratifiedKFold(n_splits=5, random_state=seed_value)\nfor train_index, val_index in skf.split(df[\"image_id\"], df[\"label\"]):\n    val_list.append(val_index)\n\none_fold = df.loc[val_list[0]]\ntwo_fold = df.loc[val_list[1]]\nthree_fold = df.loc[val_list[2]]\nfour_fold = df.loc[val_list[3]]\nfive_fold = df.loc[val_list[4]]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tta = tf.keras.Sequential(\n    [\n        tf.keras.layers.experimental.preprocessing.RandomFlip(\"horizontal_and_vertical\"),\n        tf.keras.layers.experimental.preprocessing.RandomRotation(0.25),\n        tf.keras.layers.experimental.preprocessing.RandomContrast(0.2)\n    ]\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def duplicate_image(img_path, image_size=image_size, tta_runs=2):\n\n    img = Image.open(img_path)\n    img = img.resize((image_size, image_size))\n    img_height, img_width = img.size\n    img = np.array(img)\n    \n    img_list = []\n    for i in range(tta_runs):\n        img_list.append(img)\n  \n    return np.array(img_list)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def predict_with_tta(image_filename, folder, tta_runs=2):\n    \n    #apply TTA to each of the 3 images and sum all predictions for each local image\n    localised_predictions = []\n    local_image_list = duplicate_image(folder+image_filename)\n    for local_image in local_image_list:\n        local_image = tf.expand_dims(local_image,0)\n        augmented_images = [tta(local_image) for i in range(tta_runs)]\n        predictions = effnet.predict(np.array(augmented_images[0]))\n        localised_predictions.append(np.sum(predictions, axis=0))\n    \n    #sum all predictions from all 3 images and retrieve the index of the highest value\n    global_predictions = np.sum(np.array(localised_predictions),axis=0)\n    max_value = max(global_predictions)\n    final_prediction = np.argmax(global_predictions)\n    \n    return [final_prediction, max_value, global_predictions]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_folder = \"../input/cassava-leaf-disease-classification/train_images/\"\ntrain_image = \"980448273.jpg\"\npredictions = predict_with_tta(train_image, train_folder)\n\nprint(\"Predicted Label: \", predictions[0])\nprint(\"Predicted Label Value: \", predictions[1])\nprint(\"Predicted One-Hot Label: \", predictions[2])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Confidence Level: {:.2f}\".format(predictions[1]/2*100), \"%\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def predict_image_list(image_list, folder):\n    predictions = []\n    values = []\n    with tqdm(total=len(image_list)) as pbar:\n        for image_filename in image_list:\n            pbar.update(1)\n            predictions.append(predict_with_tta(image_filename, folder)[0])\n            values.append(predict_with_tta(image_filename, folder)[1])\n    return [predictions, values]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = model_one\nplaceholder = predict_image_list(one_fold[\"image_id\"], train_folder)\none_fold[\"pred\"] = placeholder[0]\none_fold[\"value\"] = placeholder[1]\none_fold.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = model_two\nplaceholder = predict_image_list(two_fold[\"image_id\"], train_folder)\ntwo_fold[\"pred\"] = placeholder[0]\ntwo_fold[\"value\"] = placeholder[1]\ntwo_fold.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = model_three\nplaceholder = predict_image_list(three_fold[\"image_id\"], train_folder)\nthree_fold[\"pred\"] = placeholder[0]\nthree_fold[\"value\"] = placeholder[1]\nthree_fold.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = model_four\nplaceholder = predict_image_list(four_fold[\"image_id\"], train_folder)\nfour_fold[\"pred\"] = placeholder[0]\nfour_fold[\"value\"] = placeholder[1]\nfour_fold.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = model_five\nplaceholder = predict_image_list(five_fold[\"image_id\"], train_folder)\nfive_fold[\"pred\"] = placeholder[0]\nfive_fold[\"value\"] = placeholder[1]\nfive_fold.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Denoising The Data\nthreshold = 2*0.8","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"mask1 = (one_fold[\"label\"] != one_fold[\"pred\"]) & (one_fold[\"value\"] >= threshold)\none_list = one_fold[mask1].index.to_list()\n\nmask2 = (two_fold[\"label\"] != two_fold[\"pred\"]) & (two_fold[\"value\"] >= threshold)\ntwo_list = two_fold[mask2].index.to_list()\n\nmask3 = (three_fold[\"label\"] != three_fold[\"pred\"]) & (three_fold[\"value\"] >= threshold)\nthree_list = three_fold[mask3].index.to_list()\n\nmask4 = (four_fold[\"label\"] != four_fold[\"pred\"]) & (four_fold[\"value\"] >= threshold)\nfour_list = four_fold[mask4].index.to_list()\n\nmask5 = (five_fold[\"label\"] != five_fold[\"pred\"]) & (five_fold[\"value\"] >= threshold)\nfive_list = five_fold[mask5].index.to_list()\n\ncombined_list = list(np.unique(one_list + two_list + three_list + four_list + five_list))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"temp = df_train.iloc[combined_list]\ntemp[\"label\"].value_counts","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"perc = len(temp)/len(df_train)*100\nprint(\"Percentage of Data To Be Removed: {:.2f}\".format(perc), \"%\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = df_train.drop(combined_list, axis=\"index\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.reset_index(drop=True, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fold_number = 0\nn_splits = 3\noof_accuracy = []\n\ntf.keras.backend.clear_session()\nskf = StratifiedKFold(n_splits=n_splits, random_state=seed_value)\nfor train_index, val_index in skf.split(df[\"image_id\"], df[\"label\"]):\n    train_set = df.loc[train_index]\n    val_set = df.loc[val_index]\n    train_datagen, val_datagen = DataGenerator(train_set, val_set)\n    model = build_model()\n    print(\"Training fold no.: \" + str(fold_number+1))\n\n    model_name = \"denoised effnetb0 \"\n    fold_name = \"fold.h5\"\n    filepath = model_name + str(fold_number+1) + fold_name\n    callbacks = [ModelCheckpoint(filepath=filepath, monitor='val_accuracy', save_best_only=True)]\n\n    history = model.fit(train_datagen, epochs=epochs, validation_data=val_datagen, callbacks=callbacks)\n    oof_accuracy.append(max(history.history[\"val_accuracy\"]))\n    fold_number += 1\n    if fold_number == n_splits:\n        print(\"Training finished!\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"models = []\nfor i in range(n_splits):\n    deffnet = load_model(\"./denoised effnetb0 \" + str(i+1) + \"fold.h5\")\n    models.append(deffnet)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#print(models)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ss = pd.read_csv(os.path.join('../input/cassava-leaf-disease-classification', \"sample_submission.csv\"))\npreds = []\nresults = []\n\n\nfor image_id in ss.image_id:\n    image = Image.open(os.path.join('../input/cassava-leaf-disease-classification', \"test_images\", image_id))\n    image = image.resize((image_size, image_size))\n    image = np.expand_dims(image, axis = 0)\n    for model in models:\n        preds.append(np.argmax(model.predict(image/255)))\n    res = max(set(preds), key = preds.count)\n        \n    results.append(res)\n    \n\nss['label'] = results\nss.to_csv('submission.csv', index = False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}