{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Adapted from https://towardsdatascience.com/ensembling-convnets-using-keras-237d429157eb"},{"metadata":{"trusted":true},"cell_type":"code","source":"import tensorflow as tf\nfrom keras.callbacks import History\nfrom keras.callbacks import ModelCheckpoint, TensorBoard\nfrom keras.datasets import cifar10\nfrom keras.engine import training\nfrom keras.layers import Conv2D, MaxPooling2D, GlobalAveragePooling2D, Dropout, Activation, Average\nfrom keras.losses import categorical_crossentropy\nfrom keras.models import Model, Input\nfrom keras.optimizers import Adam\n\nimport keras\nfrom keras.preprocessing import image\nfrom keras.models import Sequential\nfrom keras.layers import Conv2D, MaxPool2D, Flatten, Dense, Dropout, BatchNormalization, Input, GlobalAveragePooling2D\nfrom keras.utils.vis_utils import plot_model\nfrom keras.callbacks import ModelCheckpoint,EarlyStopping,ReduceLROnPlateau\nfrom tensorflow.keras.layers.experimental import preprocessing\nfrom tensorflow.keras.applications import InceptionResNetV2\n\nimport matplotlib.pyplot as plt\n\n\nfrom keras.utils import to_categorical\nfrom tensorflow.python.framework.ops import Tensor\nfrom typing import Tuple, List\nimport glob\nimport numpy as np\nimport os\nfrom keras.models import load_model\nfrom keras.utils import to_categorical\nfrom numpy import dstack\n\nfrom sklearn.metrics import classification_report\n\nimport pandas as pd","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tf.random.set_seed(23)\n\n\nIMG_SIZE_incres = 320\nIMG_SIZE_effnet = 333\nBATCH_SZ = 320","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"effnet_path = \"../input/effnetpp/eff.h5\"\nincRes_path = \"../input/inceptionresnet/inceptionResNetv2_Sun_6pm.h5\"\n\nmodel_paths = [effnet_path, incRes_path]\n\nlist_of_models = [load_model(m_p) for m_p in model_paths]\neffModel = list_of_models[0]\nincResModel = list_of_models[1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/cassava-leaf-disease-classification/train.csv\")\ntrain['label'] = train['label'].astype('string')\ntrain.head()\n\ndiseases = pd.read_json(\"/kaggle/input/cassava-leaf-disease-classification/label_num_to_disease_map.json\", typ='series')\ndiseases","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"datagen_incres = image.ImageDataGenerator(rotation_range=360,\n                                width_shift_range=0.1,\n                                height_shift_range=0.1,\n                                brightness_range=[0.2,1.5],\n                                shear_range=25,\n                                zoom_range=0.3,\n                                channel_shift_range=0.1,\n                                horizontal_flip=True,\n                                vertical_flip=True,\n                                rescale=1/255,\n                                validation_split=0.15)\n\nval_datagen_incres = image.ImageDataGenerator(rescale=1/255,\n                                       validation_split = 0.2)\n\ntrain_generator_incres = datagen_incres.flow_from_dataframe(\n    dataframe=train,\n    directory=\"/kaggle/input/cassava-leaf-disease-classification/train_images\",\n    x_col='image_id',\n    y_col='label',\n    target_size=(IMG_SIZE_incres, IMG_SIZE_incres),\n    batch_size=32,\n    subset='training',\n    shuffle = True,\n    class_mode='categorical'\n)\n\nval_generator_incres = val_datagen_incres.flow_from_dataframe(\n    dataframe=train,\n    directory=\"/kaggle/input/cassava-leaf-disease-classification/train_images\",\n    x_col='image_id',\n    y_col='label',\n    target_size=(IMG_SIZE_incres, IMG_SIZE_incres),\n    batch_size=32,\n    subset='validation',\n    class_mode = 'categorical',\n    shuffle = True\n)\n\n# Same as val_generator, except that shuffle = False\ntest_generator_incres = val_datagen_incres.flow_from_dataframe(\n    dataframe=train,\n    directory=\"/kaggle/input/cassava-leaf-disease-classification/train_images\",\n    x_col='image_id',\n    y_col='label',\n    target_size=(IMG_SIZE_incres, IMG_SIZE_incres),\n    batch_size=BATCH_SZ,\n    subset='validation',\n    class_mode = 'categorical',\n    shuffle = False\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"datagen_effnet = image.ImageDataGenerator(rotation_range=360,\n                                width_shift_range=0.1,\n                                height_shift_range=0.1,\n                                brightness_range=[0.2,1.5],\n                                shear_range=25,\n                                zoom_range=0.3,\n                                channel_shift_range=0.1,\n                                horizontal_flip=True,\n                                vertical_flip=True,\n                                rescale=1/255,\n                                validation_split=0.15)\n\nval_datagen_effnet = image.ImageDataGenerator(rescale=1/255,\n                                       validation_split = 0.2)\n\ntrain_generator_effnet = datagen_effnet.flow_from_dataframe(\n    dataframe=train,\n    directory=\"/kaggle/input/cassava-leaf-disease-classification/train_images\",\n    x_col='image_id',\n    y_col='label',\n    target_size=(IMG_SIZE_effnet, IMG_SIZE_effnet),\n    batch_size=32,\n    subset='training',\n    shuffle = True,\n    class_mode='categorical'\n)\n\nval_generator_effnet = val_datagen_effnet.flow_from_dataframe(\n    dataframe=train,\n    directory=\"/kaggle/input/cassava-leaf-disease-classification/train_images\",\n    x_col='image_id',\n    y_col='label',\n    target_size=(IMG_SIZE_effnet, IMG_SIZE_effnet),\n    batch_size=32,\n    subset='validation',\n    class_mode = 'categorical',\n    shuffle = True\n)\n\n# Same as val_generator, except that shuffle = False\ntest_generator_effnet = val_datagen_effnet.flow_from_dataframe(\n    dataframe=train,\n    directory=\"/kaggle/input/cassava-leaf-disease-classification/train_images\",\n    x_col='image_id',\n    y_col='label',\n    target_size=(IMG_SIZE_effnet, IMG_SIZE_effnet),\n    batch_size=BATCH_SZ,\n    subset='validation',\n    class_mode = 'categorical',\n    shuffle = False\n)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Check that dataloaders working properly"},{"metadata":{"trusted":true},"cell_type":"code","source":"# test_imgs, labels = next(test_generator)\n# print(test_imgs.shape)\n\n# plt.figure(figsize=(20,10))\n# for i in range(25):\n#     plt.subplot(5,5,i+1)\n#     plt.xticks([])\n#     plt.yticks([])\n#     plt.grid(False)\n#     plt.imshow(test_imgs[i])\n#     label1 = np.argmax(labels[i])\n#     plt.xlabel(diseases.get(label1))\n# plt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"testing with just one to make sure the averaging works\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"#test_imgs_incres, labels_incres = next(test_generator_incres)\n#test_imgs_effnet, labels_effnet = next(test_generator_effnet)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#np.mean(labels_incres == labels_effnet)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#test = iter(test_generator)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#predictions = [incResModel.predict_on_batch(test_imgs_incres), effModel.predict_on_batch(test_imgs_effnet)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#avg_preds = np.average(predictions, axis=0)\n#avg_preds","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#avg_preds.argmax(axis=1)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Now let's try scaling this up"},{"metadata":{"trusted":true},"cell_type":"code","source":"# test_labels = []\n# test_prob_preds = []\n# test_class_preds = []\n\n# while True:\n#     try:\n#         imgs_incres, labels_incres = next(test_generator_incres)\n#         imgs_effnet, labels_effnet = next(test_generator_effnet)\n#     except StopIteration:\n#         break   \n#     else: # this is executed if the try clause does not raise an exception\n#         test_labels.append(labels_incres)\n\n#         predictions = [incResModel.predict_on_batch(imgs_incres), effModel.predict_on_batch(imgs_effnet)]\n#         avg_preds = np.average(predictions, axis=0)\n#         class_preds = avg_preds.argmax(axis=1)\n\n#         test_prob_preds.append(avg_preds)\n#         test_class_preds.append(class_preds)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#test_acc = np.mean(test_class_preds == test_labels)\n#test_acc","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import cv2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_sub = pd.read_csv('/kaggle/input/cassava-leaf-disease-classification/sample_submission.csv')\n\ndef getIndividProbPred(image_str, IMG_SIZE, model):\n    img = tf.keras.preprocessing.image.load_img('../input/cassava-leaf-disease-classification/test_images/' + image_str)\n    img = tf.keras.preprocessing.image.img_to_array(img)\n    img = tf.keras.preprocessing.image.smart_resize(img, (IMG_SIZE, IMG_SIZE))\n    img = tf.reshape(img, (-1, IMG_SIZE, IMG_SIZE, 3))\n    prediction = model.predict(img/255)\n    return prediction\n\ndef getIRpred(img_id): return getIndividProbPred(img_id, IMG_SIZE_incres, incResModel)\ndef getEffpred(img_id): return getIndividProbPred(img_id, IMG_SIZE_effnet, effModel)\n\ndef getEnsembPredForImg(img_id):\n    predictions = [getIRpred(img_id), getEffpred(img_id)]\n    avg_preds = np.average(predictions, axis=0)\n    class_pred = avg_preds.argmax(axis=1)\n    return class_pred.item(0)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ss = pd.read_csv('../input/cassava-leaf-disease-classification/sample_submission.csv')\npreds = [getEnsembPredForImg(img_id) for img_id in ss.image_id]\npreds","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nsubmission = pd.DataFrame({'image_id': ss.image_id, 'label': preds})\nsubmission\n\nsubmission.to_csv('submission.csv', index = False)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# labelNames = [cidx for cidx in train_generator.class_indices]\n\n# predictions = []\n\n# # loop over the models\n# for model in list_of_models:\n#      # use the current model to make predictions on the testing data,\n#      # then store these predictions in the aggregate predictions list\n#      predictions.append(model.predict(test_generator, batch_size=BATCH_SZ))\n# # average the probabilities across all model predictions, then show\n# # a classification report\n# predictions = np.average(predictions, axis=0)\n\n# print(classification_report(test_generator.classes,\n#      predictions.argmax(axis=1), target_names=labelNames))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#list_of_models[0].input_shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n#labelNames = [cidx for cidx in train_generator.class_indices]\n#print(classification_report(test_generator.classes,\n#     predictions.argmax(axis=1), target_names=labelNames))\n\n#print(classification_report(test_generator.classes, y_pred, target_names=labelNames))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}