{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nimport glob\nimport shutil\nimport json\nimport keras\nimport itertools\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport tensorflow as tf\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nfrom collections import Counter\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import GlobalAveragePooling2D, Flatten, Dense, Dropout, BatchNormalization\nfrom tensorflow.keras.optimizers import RMSprop, Adam, SGD\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint, ReduceLROnPlateau\n\nwork_dir = '../input/cassava-leaf-disease-classification/'\nos.listdir(work_dir) \ntrain_path = '/kaggle/input/cassava-leaf-disease-classification/train_images'\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = pd.read_csv(work_dir + 'train.csv')\nprint(Counter(data['label'])) # Checking the frequencies of the labels","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"print(\"Using TensorFlow version %s\" % tf.__version__)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# # Verification de utilisation du TPU, sinon affichage 1 alors activer le TPU pour avoir les 8 coeurs a utiliser\n# try:\n#     tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n#     print('Device:', tpu.master())\n#     tf.config.experimental_connect_to_cluster(tpu)\n#     tf.tpu.experimental.initialize_tpu_system(tpu)\n#     strategy = tf.distribute.experimental.TPUStrategy(tpu)\n# except:\n#     strategy = tf.distribute.get_strategy()\n# print('Number of replicas:', strategy.num_replicas_in_sync)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = pd.read_csv(work_dir + 'train.csv')\nprint(Counter(data['label'])) # Checking the frequencies of the labels\n\ndata['label'].hist()\n\n# Importing the json file with labels\n\nf = open(work_dir + 'label_num_to_disease_map.json')\nreal_labels = json.load(f)\nreal_labels = {int(k):v for k,v in real_labels.items()}\n\n# Defining the working dataset\ndata['class_name'] = data.label.map(real_labels)\nprint(data.head(10))\nprint(data['class_name'].unique())\n\n\nmask = data['label'] ==4\nclassHealthy = data[mask]\n\nmask = data['label'] ==3\nclassCMD = data[mask]\n\nmask = data['label'] ==2\nclassCGM = data[mask]\nmask = data['label'] ==1\nclassCBSD = data[mask]\n\nmask = data['label'] ==0\nclassCBB = data[mask]\n\n\nclass0 = classCBB.sample(frac=0.99)\nclass1 = classCBSD.sample(frac=0.9)\nclass2 = classCGM.sample(frac=0.9)\nclass3 = classCMD.sample(frac=0.9)\nclass4 = classHealthy.sample(frac=0.9)\n\n\n\n\nframes=[class0,class1,class2,class3,class4]\nfinalData = pd.concat(frames)\nfinalData.head(10)\nprint(len(finalData))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain,val = train_test_split(finalData, test_size = 0.05, random_state = 42, stratify = finalData['class_name'])\n\n# Importing the data using ImageDataGenerator\n\nfrom keras.preprocessing.image import ImageDataGenerator\n\nIMG_SIZE =  300 #300\nsize = (IMG_SIZE,IMG_SIZE)\nn_CLASS = 5\n\ndatagen = ImageDataGenerator(\n                    preprocessing_function = tf.keras.applications.efficientnet.preprocess_input,\n                    rotation_range = 60,\n                    width_shift_range = 0.2,\n                    height_shift_range = 0.2,\n                    shear_range = 0.2,\n                    zoom_range = 0.2,\n                    horizontal_flip = True,\n                    vertical_flip = True,\n                    fill_mode = 'nearest')\n\ntrain_set = datagen.flow_from_dataframe(train,\n                         directory = train_path,\n                         seed=42,\n                         x_col = 'image_id',\n                         y_col = 'class_name',\n                         target_size = size,\n                         class_mode = 'categorical',\n                         interpolation = 'nearest',\n                         shuffle = True,\n                         batch_size = 64)\n\nval_set = datagen.flow_from_dataframe(val,\n                         directory = train_path,\n                         seed=42,\n                         x_col = 'image_id',\n                         y_col = 'class_name',\n                         target_size = size,\n                         class_mode = 'categorical',\n                         interpolation = 'nearest',\n                         shuffle = True,\n                         batch_size = 64)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# def create_model():\n    \n#     model = Sequential()\n#     # initialize the model with input shape as (224,224,3)\n#     model.add(tf.keras.applications.EfficientNetB0(input_shape = (IMG_SIZE, IMG_SIZE, 3), include_top = False, weights = 'imagenet' ))\n# #     model.load_weights('../input/noisystudent/efficientnet-b0_noisy-student_notop.h5', by_name=True)\n#     model.add(GlobalAveragePooling2D())\n#     model.add(Flatten())\n#     model.add(Dense(256, activation = 'relu', bias_regularizer=tf.keras.regularizers.L1L2(l1=0.01, l2=0.001)))\n#     model.add(BatchNormalization())\n#     model.add(Dropout(0.7))\n#     model.add(Dense(32, activation = 'relu', bias_regularizer=tf.keras.regularizers.L1L2(l1=0.01, l2=0.001)))\n#     model.add(BatchNormalization())\n#     model.add(Dropout(0.7))\n#     model.add(Dense(n_CLASS, activation = 'softmax'))\n    \n#     return model\n\n# leaf_model = create_model()\n# # leaf_model = leaf_model.\n\n# leaf_model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def create_model():\n    \n    model = Sequential()\n    # initialize the model with input shape as (224,224,3)\n    model.add(tf.keras.applications.EfficientNetB0(input_shape = (IMG_SIZE, IMG_SIZE, 3), include_top = False, weights = 'imagenet' ))\n#     model.load_weights('../input/noisystudent/efficientnet-b0_noisy-student_notop.h5', by_name=True)\n    model.add(GlobalAveragePooling2D())\n    model.add(Flatten())\n#     model.add(Dense(256, bias_regularizer=tf.keras.regularizers.L1L2(l1=0.01, l2=0.001)))\n    model.add(Dense(256))\n#     model.add(BatchNormalization())\n    model.add(keras.layers.Activation(\"relu\"))\n    model.add(Dropout(0.7))\n    model.add(Dense(32))\n#     model.add(Dense(32, bias_regularizer=tf.keras.regularizers.L1L2(l1=0.01, l2=0.001)))\n#     model.add(BatchNormalization())\n    model.add(keras.layers.Activation(\"relu\"))\n    model.add(Dropout(0.7))\n    model.add(Dense(n_CLASS, activation = 'softmax'))\n    \n    return model\n\nleaf_model = create_model()\n# leaf_model = leaf_model.\n\nleaf_model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"EPOCHS = 50\nSTEP_SIZE_TRAIN = train_set.n // train_set.batch_size\nSTEP_SIZE_VALID = val_set.n // val_set.batch_size\nprint(STEP_SIZE_TRAIN)\nprint(STEP_SIZE_TRAIN)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def Model_fit():\n    \n    #leaf_model = None\n    \n    leaf_model = create_model()\n#     leaf_model = keras.load_weights('../input/noisystudent/efficientnet-l2_noisy-student_notop.h5')\n\n    '''Compiling the model'''\n\n    loss = tf.keras.losses.CategoricalCrossentropy(from_logits = False,\n                                                   label_smoothing=0.001,\n                                                   name='categorical_crossentropy' )\n    \n    leaf_model.compile(optimizer = Adam(learning_rate = 0.001), #2e-4\n                        loss = loss, #'categorical_crossentropy'\n                        metrics = ['categorical_accuracy']) #'acc'\n\n    # Stop training when the val_loss has stopped decreasing for 5 epochs.\n    es = EarlyStopping(monitor='val_loss', mode='min', patience=5,\n                       restore_best_weights=True, verbose=1)\n\n    # Save the model with the minimum validation loss\n    checkpoint_cb = ModelCheckpoint(\"Cassava_best_modelEffNetB0-NS.h5\",\n                                    save_best_only=True,\n                                    monitor = 'val_loss',\n                                    mode='min')\n\n    # reduce learning rate\n    reduce_lr = ReduceLROnPlateau(monitor = 'val_loss',\n                                  factor = 0.3,\n                                  patience = 3,\n                                  min_lr = 1e-6,\n                                  mode = 'min',\n                                  verbose = 1)\n\n    history = leaf_model.fit(train_set,\n                             validation_data = val_set,\n                             epochs= EPOCHS,\n                             batch_size = 64,\n                             steps_per_epoch = STEP_SIZE_TRAIN,\n                             validation_steps = STEP_SIZE_VALID,\n                             callbacks= [es, checkpoint_cb,reduce_lr])\n\n    leaf_model.save('Cassava_modelEffNetB0-NS'+'.h5')  \n\n    return history","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history = Model_fit()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"acc = history.history['categorical_accuracy']\nval_acc = history.history['val_categorical_accuracy']\n\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs_range = range(EPOCHS)\n\n# plt.figure(figsize=(8, 8))\n# plt.subplot(1, 2, 1)\nplt.plot(acc, label='Training Accuracy')\nplt.plot(val_acc, label='Validation Accuracy')\nplt.legend(loc='lower right')\nplt.title('Training and Validation Accuracy')\nplt.show()\n\n# plt.subplot(1, 2, 2)\nplt.plot(loss, label='Training Loss')\nplt.plot(val_loss, label='Validation Loss')\nplt.legend(loc='upper right')\nplt.title('Training and Validation Loss')\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import keras\n\nfinal_model = keras.models.load_model('Cassava_best_modelEffNetB0-NS.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Importing the json file with labels\n\nf = open(work_dir + 'label_num_to_disease_map.json')\nreal_labels = json.load(f)\nreal_labels = {int(k):v for k,v in real_labels.items()}\n\n# Defining the working dataset\ndata['class_name'] = data.label.map(real_labels)\n\nclass0 = classCBB.sample(frac=0.99)\nclass1 = classCBSD.sample(frac=0.7)\nclass2 = classCGM.sample(frac=0.7)\nclass3 = classCMD.sample(frac=0.7)\nclass4 = classHealthy.sample(frac=0.7)\n\n\n\n\nframes=[class0,class1,class2,class3,class4]\nfinalData = pd.concat(frames)\nfinalData.head(10)\nprint(len(finalData))\n\n# Spliting the data\nfrom sklearn.model_selection import train_test_split\n\ntrain,val = train_test_split(finalData, test_size = 0.1, random_state = 40, stratify = finalData['class_name'])\n\n# Importing the data using ImageDataGenerator\n\nfrom keras.preprocessing.image import ImageDataGenerator\n\nIMG_SIZE = 300 # 300\nsize = (IMG_SIZE,IMG_SIZE)\nn_CLASS = 5\n\ndatagen = ImageDataGenerator(\n                    preprocessing_function = tf.keras.applications.efficientnet.preprocess_input,\n                    rotation_range =40,\n                    width_shift_range = 0.2,\n                    height_shift_range = 0.2,\n                    shear_range = 0.2,\n                    zoom_range = 0.2,\n                    horizontal_flip = True,\n                    vertical_flip = True,\n                    fill_mode = 'nearest')\n\ntrain_set = datagen.flow_from_dataframe(train,\n                         directory = train_path,\n                         seed= 40,\n                         x_col = 'image_id',\n                         y_col = 'class_name',\n                         target_size = size,\n                         #color_mode=\"rgb\",\n                         class_mode = 'categorical',\n                         interpolation = 'nearest',\n                         shuffle = True,\n                         batch_size = 64)\nvalDatagen = ImageDataGenerator(\n                    preprocessing_function = tf.keras.applications.efficientnet.preprocess_input\n                    )\n\nval_set = valDatagen.flow_from_dataframe(val,\n                         directory = train_path,\n                         seed=40,\n                         x_col = 'image_id',\n                         y_col = 'class_name',\n                         target_size = size,\n                         #color_mode=\"rgb\",\n                         class_mode = 'categorical',\n                         interpolation = 'nearest',\n                         shuffle = True,\n                         batch_size = 64)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"EPOCHS = 25\nSTEP_SIZE_TRAIN = train_set.n//train_set.batch_size\nSTEP_SIZE_VALID = val_set.n//val_set.batch_size","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def Model_fit1(model):\n#     with strategy.scope():\n    leaf_model = model\n\n\n    '''Compiling the model'''\n\n    loss = tf.keras.losses.CategoricalCrossentropy(from_logits = False,\n                                                   label_smoothing=0.01,\n                                                   name='categorical_crossentropy' )\n\n    leaf_model.compile(optimizer =  Adam(learning_rate = 2e-4),\n                        loss = loss, #'categorical_crossentropy'\n                        metrics = ['categorical_accuracy']) #'acc'\n    \n    # Stop training when the val_loss has stopped decreasing for 5 epochs.\n    es = EarlyStopping(monitor='val_loss', mode='min', patience=7,\n                       restore_best_weights=True, verbose=1)\n    \n    # Save the model with the minimum validation loss\n    checkpoint_cb = ModelCheckpoint(\"Cassava_best_modelEffNetB0v1-NS.h5\",\n                                    save_best_only=True,\n                                    monitor = 'categorical_accuracy',\n                                    mode='max')\n    \n    # reduce learning rate\n    reduce_lr = ReduceLROnPlateau(monitor = 'val_loss',\n                                  factor = 0.1,\n                                  patience = 2,\n                                  min_lr = 1e-6,\n                                  mode = 'min',\n                                  verbose = 1)\n    \n    history = leaf_model.fit(train_set,\n                             validation_data = val_set,\n                             epochs= EPOCHS,\n                             batch_size = 64,\n                             steps_per_epoch = STEP_SIZE_TRAIN,\n                             validation_steps = STEP_SIZE_VALID,\n                             callbacks=[es, checkpoint_cb])\n    \n    leaf_model.save('Cassava_modelEffNetB0v1-NS'+'.h5')\n    \n    return history","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history = Model_fit1(final_model)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"acc = history.history['categorical_accuracy']\nval_acc = history.history['val_categorical_accuracy']\n\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs_range = range(EPOCHS)\n\n# plt.figure(figsize=(8, 8))\n# plt.subplot(1, 2, 1)\nplt.plot(acc, label='Training Accuracy')\nplt.plot(val_acc, label='Validation Accuracy')\nplt.legend(loc='lower right')\nplt.title('Training and Validation Accuracy')\nplt.show()\n\n# plt.subplot(1, 2, 2)\nplt.plot(loss, label='Training Loss')\nplt.plot(val_loss, label='Validation Loss')\nplt.legend(loc='upper right')\nplt.title('Training and Validation Loss')\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}