{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport random\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-03T18:04:32.346196Z","iopub.execute_input":"2021-07-03T18:04:32.346534Z","iopub.status.idle":"2021-07-03T18:04:32.354867Z","shell.execute_reply.started":"2021-07-03T18:04:32.34646Z","shell.execute_reply":"2021-07-03T18:04:32.354088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n\n\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications.vgg16 import VGG16\nfrom tensorflow.keras.applications.vgg19 import VGG19\nfrom tensorflow.keras.applications.inception_v3 import InceptionV3\nfrom tensorflow.keras.applications.inception_resnet_v2 import InceptionResNetV2\nfrom tensorflow.keras.applications.densenet import DenseNet121, DenseNet169, DenseNet201 \nfrom tensorflow.keras.applications.xception import Xception\nfrom tensorflow.keras.applications.resnet50 import ResNet50\nfrom tensorflow.keras.applications.resnet_v2 import ResNet50V2, ResNet101V2, ResNet152V2\nfrom tensorflow.keras.applications.nasnet import NASNetLarge\nfrom tensorflow.keras.applications import EfficientNetB7, EfficientNetB2\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint, TensorBoard\n\nfrom sklearn.model_selection import KFold, train_test_split\n\nfrom tensorflow.keras.layers import Flatten,Dense,Dropout,BatchNormalization\nfrom tensorflow.keras.models import Model,Sequential\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, BatchNormalization, GlobalAveragePooling2D\nfrom keras.optimizers import Adam,SGD,Adagrad,Adadelta,RMSprop\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, EarlyStopping, ModelCheckpoint","metadata":{"execution":{"iopub.status.busy":"2021-07-03T18:04:32.35909Z","iopub.execute_input":"2021-07-03T18:04:32.359682Z","iopub.status.idle":"2021-07-03T18:04:37.851788Z","shell.execute_reply.started":"2021-07-03T18:04:32.359645Z","shell.execute_reply":"2021-07-03T18:04:37.850974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH_IMG = '/kaggle/input/resized-plant2021/img_sz_384/'\nPATH_TAB = '/kaggle/input/plant-pathology-2021-fgvc8/'\nPATH_SAVE = '/kaggle/working/'\nos.listdir(PATH_TAB)[:5]","metadata":{"execution":{"iopub.status.busy":"2021-07-03T18:04:37.853169Z","iopub.execute_input":"2021-07-03T18:04:37.853487Z","iopub.status.idle":"2021-07-03T18:04:37.863449Z","shell.execute_reply.started":"2021-07-03T18:04:37.853454Z","shell.execute_reply":"2021-07-03T18:04:37.862441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.chdir(PATH_TAB)","metadata":{"execution":{"iopub.status.busy":"2021-07-03T18:04:37.865656Z","iopub.execute_input":"2021-07-03T18:04:37.866099Z","iopub.status.idle":"2021-07-03T18:04:37.870274Z","shell.execute_reply.started":"2021-07-03T18:04:37.866061Z","shell.execute_reply":"2021-07-03T18:04:37.86911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"train.csv\")\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-03T18:04:37.872136Z","iopub.execute_input":"2021-07-03T18:04:37.872497Z","iopub.status.idle":"2021-07-03T18:04:37.928355Z","shell.execute_reply.started":"2021-07-03T18:04:37.872464Z","shell.execute_reply":"2021-07-03T18:04:37.927587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Análisis Exploratorio","metadata":{}},{"cell_type":"code","source":"freq = df['labels'].value_counts()\nfreq.index","metadata":{"execution":{"iopub.status.busy":"2021-07-03T18:04:37.9296Z","iopub.execute_input":"2021-07-03T18:04:37.92995Z","iopub.status.idle":"2021-07-03T18:04:37.943861Z","shell.execute_reply.started":"2021-07-03T18:04:37.929915Z","shell.execute_reply":"2021-07-03T18:04:37.942894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sns.set_theme(style=\"whitegrid\")\nfreq.plot.bar()\nplt.savefig(PATH_SAVE + \"freq_plot.svg\")","metadata":{"execution":{"iopub.status.busy":"2021-07-03T18:04:37.945054Z","iopub.execute_input":"2021-07-03T18:04:37.945433Z","iopub.status.idle":"2021-07-03T18:04:38.266939Z","shell.execute_reply.started":"2021-07-03T18:04:37.945397Z","shell.execute_reply":"2021-07-03T18:04:38.266016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocesamiento","metadata":{"execution":{"iopub.status.busy":"2021-06-17T15:58:41.786565Z","iopub.execute_input":"2021-06-17T15:58:41.787143Z","iopub.status.idle":"2021-06-17T15:58:41.7916Z","shell.execute_reply.started":"2021-06-17T15:58:41.787094Z","shell.execute_reply":"2021-06-17T15:58:41.790687Z"}}},{"cell_type":"code","source":"freq = df['labels'].value_counts()\nlabels = freq.index\nlabels","metadata":{"execution":{"iopub.status.busy":"2021-07-03T18:05:28.329065Z","iopub.execute_input":"2021-07-03T18:05:28.329411Z","iopub.status.idle":"2021-07-03T18:05:28.339973Z","shell.execute_reply.started":"2021-07-03T18:05:28.32938Z","shell.execute_reply":"2021-07-03T18:05:28.338939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class6_mask = df['labels'].isin(labels[:6]) ","metadata":{"execution":{"iopub.status.busy":"2021-07-03T18:06:09.16652Z","iopub.execute_input":"2021-07-03T18:06:09.166872Z","iopub.status.idle":"2021-07-03T18:06:09.174079Z","shell.execute_reply.started":"2021-07-03T18:06:09.166841Z","shell.execute_reply":"2021-07-03T18:06:09.17328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class6_df = df[class6_mask]\nclass6_df['labels'].unique()","metadata":{"execution":{"iopub.status.busy":"2021-07-03T18:06:12.537142Z","iopub.execute_input":"2021-07-03T18:06:12.537461Z","iopub.status.idle":"2021-07-03T18:06:12.545817Z","shell.execute_reply.started":"2021-07-03T18:06:12.537433Z","shell.execute_reply":"2021-07-03T18:06:12.544853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_test_split_mask = np.random.rand(class6_df.shape[0]) < 0.8","metadata":{"execution":{"iopub.status.busy":"2021-07-03T18:06:16.650916Z","iopub.execute_input":"2021-07-03T18:06:16.651297Z","iopub.status.idle":"2021-07-03T18:06:16.655945Z","shell.execute_reply.started":"2021-07-03T18:06:16.651266Z","shell.execute_reply":"2021-07-03T18:06:16.655086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = class6_df[train_test_split_mask]\ntest_df = class6_df[~train_test_split_mask]\n\nprint(f\"Numero de imagenes en class6_df : {class6_df.shape[0]}\")\nprint(f\"Numero de imagenes en train_df : {train_df.shape[0]}\")\nprint(f\"Numero de imagenes en test_df : {test_df.shape[0]}\")","metadata":{"execution":{"iopub.status.busy":"2021-07-03T18:06:18.749391Z","iopub.execute_input":"2021-07-03T18:06:18.749704Z","iopub.status.idle":"2021-07-03T18:06:18.75941Z","shell.execute_reply.started":"2021-07-03T18:06:18.749675Z","shell.execute_reply":"2021-07-03T18:06:18.758567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Algunas imagenes","metadata":{}},{"cell_type":"code","source":"#plot function\ndef plot_images():\n    idx = np.random.choice(class6_df.shape[0], 20, replace=False)\n    print(\"Pictures: {}\".format(idx))\n    plt.figure(figsize=(15,10))\n    for i in range(20):\n        plt.subplot(5,5,i+1)\n        img_name = class6_df.iloc[i][0]\n        plt.imshow(plt.imread(\"../../input/plant-pathology-2021-fgvc8/train_images/\" + img_name))\n        plt.title(class6_df.iloc[i][1])\n        plt.axis('off')\n    plt.savefig(PATH_SAVE + \"samples.svg\")","metadata":{"execution":{"iopub.status.busy":"2021-07-03T18:06:23.453602Z","iopub.execute_input":"2021-07-03T18:06:23.453926Z","iopub.status.idle":"2021-07-03T18:06:23.460264Z","shell.execute_reply.started":"2021-07-03T18:06:23.453897Z","shell.execute_reply":"2021-07-03T18:06:23.459183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_images()","metadata":{"execution":{"iopub.status.busy":"2021-07-03T18:06:23.747316Z","iopub.execute_input":"2021-07-03T18:06:23.747618Z","iopub.status.idle":"2021-07-03T18:06:49.519467Z","shell.execute_reply.started":"2021-07-03T18:06:23.74759Z","shell.execute_reply":"2021-07-03T18:06:49.5141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot function\ndef plot_images_complex():\n    mask_complex = class6_df['labels'] == \n    idx = np.random.choice(class6_df.shape[0], 20, replace=False)\n    print(\"Pictures: {}\".format(idx))\n    plt.figure(figsize=(15,10))\n    for i in range(20):\n        plt.subplot(5,5,i+1)\n        img_name = class6_df.iloc[i][0]\n        plt.imshow(plt.imread(\"../../input/plant-pathology-2021-fgvc8/train_images/\" + img_name))\n        plt.title(class6_df.iloc[i][1])\n        plt.axis('off')\n    plt.savefig(PATH_SAVE + \"samples.svg\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generacion del Modelo","metadata":{}},{"cell_type":"code","source":"img_size = 224\nn_epochs = 10\nn_class = 6\nlearn_rate=0.001","metadata":{"execution":{"iopub.status.busy":"2021-06-21T10:14:37.362484Z","iopub.execute_input":"2021-06-21T10:14:37.362833Z","iopub.status.idle":"2021-06-21T10:14:37.369032Z","shell.execute_reply.started":"2021-06-21T10:14:37.362802Z","shell.execute_reply":"2021-06-21T10:14:37.368128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 10-Fold Cross Validation","metadata":{}},{"cell_type":"code","source":"num_folds = 10","metadata":{"execution":{"iopub.status.busy":"2021-06-21T10:05:36.019085Z","iopub.execute_input":"2021-06-21T10:05:36.019488Z","iopub.status.idle":"2021-06-21T10:05:36.03144Z","shell.execute_reply.started":"2021-06-21T10:05:36.019456Z","shell.execute_reply":"2021-06-21T10:05:36.030036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model_name(k):\n    return 'model_' + str(k) + '.h5'","metadata":{"execution":{"iopub.status.busy":"2021-06-21T10:05:36.033798Z","iopub.execute_input":"2021-06-21T10:05:36.034314Z","iopub.status.idle":"2021-06-21T10:05:36.04373Z","shell.execute_reply.started":"2021-06-21T10:05:36.034277Z","shell.execute_reply":"2021-06-21T10:05:36.04278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datagen = ImageDataGenerator(rescale=1./255.)\nkfold = KFold(n_splits=num_folds, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2021-06-21T10:05:36.044978Z","iopub.execute_input":"2021-06-21T10:05:36.048188Z","iopub.status.idle":"2021-06-21T10:05:36.056591Z","shell.execute_reply.started":"2021-06-21T10:05:36.04792Z","shell.execute_reply":"2021-06-21T10:05:36.055531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"VALIDATION_ACCURACY = []\nVALIDATION_LOSS = []\n\nr_fig, r_ax = plt.subplots(num_folds, 2, figsize = (10,20)) # Para representar los resultados\n\nfold_var = 1\n\nfor train_index, val_index in kfold.split(np.zeros(train_df.shape[0]),train_df['labels']):\n    print('*'*15 + ' Model ' + str(fold_var) + ' ' + '*'*15)\n    \n    train_data = train_df.iloc[train_index]\n    valid_data = train_df.iloc[val_index]\n\n    train_generator=datagen.flow_from_dataframe(\n        dataframe=train_data,\n        directory=PATH_IMG,\n        x_col=\"image\",\n        y_col=\"labels\",\n        batch_size=32,\n        seed=123,\n        shuffle=True,\n        class_mode=\"categorical\",\n        target_size=(224,224)\n        )\n\n    valid_generator=datagen.flow_from_dataframe(\n        dataframe=valid_data,\n        directory=PATH_IMG,\n        x_col=\"image\",\n        y_col=\"labels\",\n        seed=123,\n        shuffle=True,\n        class_mode=\"categorical\",\n        target_size=(224,224)\n        )\n    \n    # Here you can replace the EfficientNetB7 architecture with any other of the loaded ones and observe the result\n    # VGG16, VGG19, InceptionV3, InceptionResNetV2, DenseNet121, DenseNet169, DenseNet201, Xception,\n    # ResNet50, ResNet50V2, ResNet101V2, ResNet152V2, NASNetLarge, EfficientNetL2\n\n    use_model = InceptionResNetV2(weights='imagenet',\n                                 include_top=False, pooling='avg',\n                                 input_shape=(img_size, img_size, 3))\n\n    model = Sequential()\n    model.add(use_model)\n    model.add(Dense(n_class, activation=\"softmax\"))\n    \n    model.compile(optimizer='nadam', loss='categorical_crossentropy',metrics=['accuracy'])\n        \n    # CREATE CALLBACKS\n    checkpoint = ModelCheckpoint(PATH_SAVE + get_model_name(fold_var), \n                                 monitor='val_accuracy', verbose=1, \n                                 save_best_only=True, mode='max')\n    \n    my_callbacks = [checkpoint]\n    \n    history = model.fit(\n            train_generator,\n            callbacks = my_callbacks,\n            validation_data=valid_generator, epochs=n_epochs)\n    \n    \n    # LOAD BEST MODEL to evaluate the performance of the model\n    model.load_weights(PATH_SAVE + get_model_name(fold_var))\n    \n    results = model.evaluate(valid_generator)\n    results = dict(zip(model.metrics_names,results))\n    \n    VALIDATION_ACCURACY.append(results['accuracy'])\n    VALIDATION_LOSS.append(results['loss'])\n    \n    r_ax[fold_var-1, 0].plot(history.history['loss'], color='green', label='Train')\n    r_ax[fold_var-1, 0].plot(history.history['val_loss'], color='blue', label='Validation')\n    r_ax[fold_var-1, 0].set_title('Loss')\n    r_ax[fold_var-1, 0].set_xlabel('Step')\n    r_ax[fold_var-1, 0].legend()\n    r_ax[fold_var-1, 1].plot(history.history['accuracy'], color='purple', label='Train')\n    r_ax[fold_var-1, 1].plot(history.history['val_accuracy'], color='orange', label='Validation')\n    r_ax[fold_var-1, 1].set_title('Accuracy')\n    r_ax[fold_var-1, 1].set_xlabel('Steps')\n    r_ax[fold_var-1, 1].legend()\n    \n    fold_var += 1\n    print('*'*40)\n    \nr_fig.savefig(PATH_SAVE + 'results_' + str(num_folds) + '-validation.svg')","metadata":{"execution":{"iopub.status.busy":"2021-06-19T09:23:17.625052Z","iopub.execute_input":"2021-06-19T09:23:17.625601Z","iopub.status.idle":"2021-06-19T13:51:24.266068Z","shell.execute_reply.started":"2021-06-19T09:23:17.625557Z","shell.execute_reply":"2021-06-19T13:51:24.265155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"r_fig.set_size_inches(8, 35)\nr_fig.tight_layout()\nr_fig.savefig(PATH_SAVE + 'results_' + str(num_folds) + '-validation.svg')\nr_fig","metadata":{"execution":{"iopub.status.busy":"2021-06-19T14:10:02.772136Z","iopub.execute_input":"2021-06-19T14:10:02.772486Z","iopub.status.idle":"2021-06-19T14:10:05.311149Z","shell.execute_reply.started":"2021-06-19T14:10:02.772455Z","shell.execute_reply":"2021-06-19T14:10:05.31014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('*'*40)\nprint('Score per fold')\nprint('*'*40)\n\nfor i in range(num_folds):\n    print(f\"Fold {i+1} - Loss: {np.round(VALIDATION_LOSS[i], 4)} - Accuracy: {np.round(VALIDATION_ACCURACY[i], 4)}\")\n\nprint('*'*40)\nprint('Average scores for all folds:')\nprint(f\"Accuracy: {np.array(VALIDATION_ACCURACY).mean().round(4)} +/- {np.array(VALIDATION_ACCURACY).std().round(4)}\")\nprint(f\"Loss: {np.array(VALIDATION_LOSS).mean().round(4)}\")\nprint('*'*40)","metadata":{"execution":{"iopub.status.busy":"2021-06-19T14:26:21.380596Z","iopub.execute_input":"2021-06-19T14:26:21.380919Z","iopub.status.idle":"2021-06-19T14:26:21.393575Z","shell.execute_reply.started":"2021-06-19T14:26:21.38089Z","shell.execute_reply":"2021-06-19T14:26:21.392688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluate Optimal Model (Model 8)","metadata":{}},{"cell_type":"code","source":"os.listdir()","metadata":{"execution":{"iopub.status.busy":"2021-06-21T10:15:24.866797Z","iopub.execute_input":"2021-06-21T10:15:24.867149Z","iopub.status.idle":"2021-06-21T10:15:24.873762Z","shell.execute_reply.started":"2021-06-21T10:15:24.867119Z","shell.execute_reply":"2021-06-21T10:15:24.872474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testgen = ImageDataGenerator(rescale=1./255.)\n\ntest_generator=testgen.flow_from_dataframe(\n    dataframe=test_df,\n    directory=PATH_IMG,\n    x_col=\"image\",\n    y_col=\"labels\",\n    seed=123,\n    shuffle=True,\n    class_mode=\"categorical\",\n    target_size=(224,224)\n    )","metadata":{"execution":{"iopub.status.busy":"2021-06-21T10:15:25.677688Z","iopub.execute_input":"2021-06-21T10:15:25.678014Z","iopub.status.idle":"2021-06-21T10:15:33.541468Z","shell.execute_reply.started":"2021-06-21T10:15:25.677979Z","shell.execute_reply":"2021-06-21T10:15:33.540664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"use_model = InceptionResNetV2(weights='imagenet',\n                                 include_top=False, pooling='avg',\n                                 input_shape=(img_size, img_size, 3))\n\nmodel_opt = Sequential()\nmodel_opt.add(use_model)\nmodel_opt.add(Dense(n_class, activation=\"softmax\"))\n\nmodel_opt.compile(optimizer='nadam', loss='categorical_crossentropy',metrics=['accuracy'])\n\nmodel_opt.load_weights('../plantasmodeloinceptionresnetv2/model_8.h5')\n\nresults_opt = model_opt.evaluate(test_generator)\nresults_opt = dict(zip(model_opt.metrics_names,results_opt))","metadata":{"execution":{"iopub.status.busy":"2021-06-21T10:15:33.544617Z","iopub.execute_input":"2021-06-21T10:15:33.544908Z","iopub.status.idle":"2021-06-21T10:16:19.774522Z","shell.execute_reply.started":"2021-06-21T10:15:33.544882Z","shell.execute_reply":"2021-06-21T10:16:19.773658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results_opt","metadata":{"execution":{"iopub.status.busy":"2021-06-21T10:17:04.747186Z","iopub.execute_input":"2021-06-21T10:17:04.747539Z","iopub.status.idle":"2021-06-21T10:17:04.7549Z","shell.execute_reply.started":"2021-06-21T10:17:04.747496Z","shell.execute_reply":"2021-06-21T10:17:04.754036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}