{"cells":[{"metadata":{"_uuid":"d0e29a00-a599-418d-9faa-bf4965f384c8","_cell_guid":"afb46ac2-6826-4c50-ac8b-41023c8d1f8b","trusted":true,"papermill":{"duration":0.020649,"end_time":"2021-01-10T05:50:54.761279","exception":false,"start_time":"2021-01-10T05:50:54.74063","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Cassava Leaf Disease Classification\n\nInstantiates the  architecture."},{"metadata":{"trusted":true},"cell_type":"code","source":"#!wget https://storage.googleapis.com/cloud-tpu-checkpoints/efficientnet/noisystudent/noisy_student_efficientnet-b3.tar.gz","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#!tar -xf noisy_student_efficientnet-b3.tar.gz","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#!wget https://raw.githubusercontent.com/tensorflow/tensorflow/master/tensorflow/python/keras/applications/efficientnet_weight_update_util.py","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#!python efficientnet_weight_update_util.py --model b3 --notop --ckpt  ./noisy-student-efficientnet-b3/model.ckpt --o efficientnetb3_notop.h5","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.mixed_precision import experimental as mixed_precision\n\npolicy = mixed_precision.Policy('mixed_float16')\nmixed_precision.set_policy(policy)\n\nprint('Compute dtype: %s' % policy.compute_dtype)\nprint('Variable dtype: %s' % policy.variable_dtype)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f1648544-f5b4-430f-9e75-1776bd89df7f","_cell_guid":"c966c872-462c-4975-91e4-050cedc6ef24","trusted":true,"papermill":{"duration":0.01959,"end_time":"2021-01-10T05:50:59.229878","exception":false,"start_time":"2021-01-10T05:50:59.210288","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Check GPU state"},{"metadata":{"_uuid":"bd83aa32-a322-4441-9669-5bff0a66ccf7","_cell_guid":"8511c4d0-5b29-400e-a678-aa07c4e5a176","trusted":true,"execution":{"iopub.execute_input":"2021-01-10T05:51:01.620963Z","iopub.status.busy":"2021-01-10T05:51:01.620091Z","iopub.status.idle":"2021-01-10T05:51:01.628498Z","shell.execute_reply":"2021-01-10T05:51:01.629139Z"},"papermill":{"duration":2.380086,"end_time":"2021-01-10T05:51:01.629442","exception":false,"start_time":"2021-01-10T05:50:59.249356","status":"completed"},"tags":[]},"cell_type":"code","source":"import tensorflow as tf\n\nfrom tensorflow.python.framework.config import set_memory_growth\ngpus = tf.config.experimental.list_physical_devices('GPU')\nif gpus:\n    try:\n        # Currently, memory growth needs to be the same across GPUs\n        for gpu in gpus:\n            tf.config.experimental.set_memory_growth(gpu, False)\n        logical_gpus = tf.config.experimental.list_logical_devices('GPU')\n        print(len(gpus), \"Physical GPUs,\", len(logical_gpus), \"Logical GPUs\")\n    except RuntimeError as e:\n        # Memory growth must be set before GPUs have been initialized\n        print(e)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"29ae3ef5-d162-4ddd-9554-62c896a8e041","_cell_guid":"93db8cae-6599-4e2a-b869-3962a6857a04","trusted":true,"execution":{"iopub.execute_input":"2021-01-10T05:50:54.807074Z","iopub.status.busy":"2021-01-10T05:50:54.806507Z","iopub.status.idle":"2021-01-10T05:50:59.14212Z","shell.execute_reply":"2021-01-10T05:50:59.141053Z"},"papermill":{"duration":4.361857,"end_time":"2021-01-10T05:50:59.142236","exception":false,"start_time":"2021-01-10T05:50:54.780379","status":"completed"},"tags":[]},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport tensorflow_addons as tfa\nfrom tensorflow.keras import layers\nfrom tensorflow import keras\nfrom keras.layers import Dense, Flatten, Activation, Dropout\nimport seaborn as sns\n\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.optimizers import Adamax\n\nimport matplotlib.pyplot as plt\nfrom keras.models import Sequential\n\nprint(\"Tensorflow version \" + tf.__version__)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a01ac5ec-f39c-463c-87e5-89a40ebcc10b","_cell_guid":"f36021a6-eb90-4c53-8eea-1ace23d9b19f","trusted":true,"execution":{"iopub.execute_input":"2021-01-10T05:50:59.188728Z","iopub.status.busy":"2021-01-10T05:50:59.187966Z","iopub.status.idle":"2021-01-10T05:50:59.190716Z","shell.execute_reply":"2021-01-10T05:50:59.190274Z"},"papermill":{"duration":0.027083,"end_time":"2021-01-10T05:50:59.190809","exception":false,"start_time":"2021-01-10T05:50:59.163726","status":"completed"},"tags":[]},"cell_type":"code","source":"TARGET_SIZE = 350\nimg_width = TARGET_SIZE\nimg_height = TARGET_SIZE\nEPOCHS = 20\nepochs =EPOCHS;\nlast_epoch=0;\nBATCH_SIZE = 24\nSEED = 128\nDROPOUT_RATE = 0.4\nvalidation_split = 0.3\n\ngeneral_path = '/kaggle/input/cassava-leaf-disease-classification/'\n\nNOISY_IMAGENET = '../input/keras-efficientnetb3-noisy-student/efficientnetb3_notop.h5'","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"06b17448-bac6-466b-ad27-9361c313cc47","_cell_guid":"20732563-912b-421a-9369-81bd1e46105d","trusted":true,"papermill":{"duration":0.034357,"end_time":"2021-01-10T05:51:01.699252","exception":false,"start_time":"2021-01-10T05:51:01.664895","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Read train data"},{"metadata":{"_uuid":"bb341555-8910-48ec-9c9c-03595db94cfd","_cell_guid":"d3b4ab4a-9b90-40fc-807e-958970970b26","trusted":true,"execution":{"iopub.execute_input":"2021-01-10T05:51:01.779773Z","iopub.status.busy":"2021-01-10T05:51:01.778891Z","iopub.status.idle":"2021-01-10T05:51:01.855592Z","shell.execute_reply":"2021-01-10T05:51:01.856392Z"},"papermill":{"duration":0.122738,"end_time":"2021-01-10T05:51:01.856558","exception":false,"start_time":"2021-01-10T05:51:01.73382","status":"completed"},"tags":[]},"cell_type":"code","source":"train = pd.read_csv('../input/cassava-leaf-disease-classification/train.csv')\ndisplay(train.head())\nprint(f'Train samples: {len(train)}')\ntrain['label'] = train['label'].astype('str')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Configure the dataset for performance\nLet's make sure to use buffered prefetching so you can yield data from disk without having I/O become blocking. These are two important methods you should use when loading data.\n\n[Dataset.cache()](https://www.tensorflow.org/api_docs/python/tf/data/Dataset#cache) keeps the images in memory after they're loaded off disk during the first epoch. This will ensure the dataset does not become a bottleneck while training your model. If your dataset is too large to fit into memory, you can also use this method to create a performant on-disk cache.\n\n[Dataset.prefetch()](https://www.tensorflow.org/api_docs/python/tf/data/Dataset#prefetch) overlaps data preprocessing and model execution while training.\n\n\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"AUTOTUNE = tf.data.experimental.AUTOTUNE\n\n#train = train.cache().shuffle(1000).prefetch(buffer_size=AUTOTUNE)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0858bda1-b9ee-4653-903a-06d3ad520b72","_cell_guid":"3d0a57f1-be6c-411a-b566-abae3fb3fe03","trusted":true,"papermill":{"duration":0.020207,"end_time":"2021-01-10T05:51:01.901656","exception":false,"start_time":"2021-01-10T05:51:01.881449","status":"completed"},"tags":[]},"cell_type":"markdown","source":"Image augmentation using Keras Image Data Generator\nImage augmentation is a technique of applying different transformations to original images which results in multiple transformed copies of the same image. Each copy, however, is different from the other in certain aspects depending on the augmentation techniques you apply like shifting, rotating, flipping, etc.\n\nApplying these small amounts of variations on the original image does not change its target class but only provides a new perspective of capturing the object in real life. And so, we use it is quite often for building deep learning models. Keras ImageDataGenerator is a gem! It lets you augment your images in real-time while your model is still training! You can apply any random transformations on each training image as it is passed to the model. This will not only make your model robust but will also save up on the overhead memory.\n\nAdvantages of using Keras Image Data Generator:\nThe main benefit of using the Keras ImageDataGenerator class is that it is designed to provide real-time data augmentation. Meaning it is generating augmented images on the fly while your model is still in the training stage. But it only returns the transformed images and does not add it to the original corpus of images. If it was, in fact, the case, then the model would be seeing the original images multiple times which would definitely overfit our model.\n\nAnother advantage of ImageDataGenerator is that it requires lower memory usage. This is so because without using this class, we load all the images at once. But on using it, we are loading the images in batches which saves a lot of memory."},{"metadata":{"_uuid":"e7d08376-6845-4216-ba6f-aae4a913fe50","_cell_guid":"aaf53080-e011-4773-a1b8-56e385d5e61e","trusted":true,"execution":{"iopub.execute_input":"2021-01-10T05:51:01.991508Z","iopub.status.busy":"2021-01-10T05:51:01.981287Z","iopub.status.idle":"2021-01-10T05:51:02.157191Z","shell.execute_reply":"2021-01-10T05:51:02.158084Z"},"papermill":{"duration":0.236547,"end_time":"2021-01-10T05:51:02.158245","exception":false,"start_time":"2021-01-10T05:51:01.921698","status":"completed"},"tags":[]},"cell_type":"code","source":"\ntrain_datagen = ImageDataGenerator(\n    horizontal_flip=True,\n    vertical_flip=True,\n    rotation_range=20,\n    shear_range=20,\n    zoom_range=0.2,\n    height_shift_range=0.1,\n    width_shift_range=0.1,\n    validation_split=validation_split,\n    rescale=1.0/255.0 # The ImageDataGenerator class can be used to rescale pixel values from the range of 0-255 to the range 0-1 preferred for neural network models.\n)\n\ntrain_datagen_flow = train_datagen.flow_from_dataframe(\n    dataframe=train,\n    directory=general_path + 'train_images',\n    x_col='image_id',\n    y_col='label',\n    target_size=(img_width, img_height),\n    batch_size=BATCH_SIZE,\n    seed=SEED,\n    subset='training',\n    class_mode='sparse',\n    validate_filenames=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"28000236-6c0d-4e9e-9286-a627d19a100a","_cell_guid":"4e70d516-9657-4bba-916a-0c0d268649aa","trusted":true,"papermill":{"duration":0.020294,"end_time":"2021-01-10T05:51:02.200355","exception":false,"start_time":"2021-01-10T05:51:02.180061","status":"completed"},"tags":[]},"cell_type":"markdown","source":"Since we have provided validation_split=0.2 as a parameter, out of 21397 images our train data generator has generated 17118 images for train data.\nFor each image it's size will be (300, 300, 3)."},{"metadata":{"_uuid":"f6b86348-2ccf-4676-b346-0453ec6eba31","_cell_guid":"e5c6b429-613a-4992-9d36-4ac4016ff6d3","trusted":true,"execution":{"iopub.execute_input":"2021-01-10T05:51:02.247542Z","iopub.status.busy":"2021-01-10T05:51:02.246669Z","iopub.status.idle":"2021-01-10T05:51:02.366509Z","shell.execute_reply":"2021-01-10T05:51:02.367044Z"},"papermill":{"duration":0.146809,"end_time":"2021-01-10T05:51:02.367165","exception":false,"start_time":"2021-01-10T05:51:02.220356","status":"completed"},"tags":[]},"cell_type":"code","source":"valid_datagen = keras.preprocessing.image.ImageDataGenerator(\n    horizontal_flip=True,\n    vertical_flip=True,\n    rotation_range=20,\n    shear_range=20,\n    zoom_range=0.2,\n    height_shift_range=0.1,\n    width_shift_range=0.1,\n    validation_split=validation_split,\n    rescale=1.0/255.0 # The ImageDataGenerator class can be used to rescale pixel values from the range of 0-255 to the range 0-1 preferred for neural network models.\n)\n\nvalid_datagen_flow = valid_datagen.flow_from_dataframe(\n    dataframe=train,\n    directory=general_path + 'train_images',\n    x_col='image_id',\n    y_col='label',\n    target_size=(img_width, img_height),\n    batch_size=BATCH_SIZE,\n    seed=SEED,\n    subset='validation',\n    class_mode='sparse',\n    validate_filenames=False\n)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Build Model"},{"metadata":{"_uuid":"c26d2872-224b-466a-9af6-12b79b609f3f","_cell_guid":"1ba08ef5-e9c6-48cd-8d2d-5f26754482fc","trusted":true,"papermill":{"duration":0.02072,"end_time":"2021-01-10T05:51:02.408498","exception":false,"start_time":"2021-01-10T05:51:02.387778","status":"completed"},"tags":[]},"cell_type":"markdown","source":"A Sequential model is appropriate for a plain stack of layers where each layer has exactly one input tensor and one output tensor."},{"metadata":{"trusted":true},"cell_type":"code","source":"data_augmentation = keras.Sequential(\n    [\n        layers.experimental.preprocessing.RandomFlip(\"horizontal\"),\n        layers.experimental.preprocessing.RandomRotation(0.1),\n    ]\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"aa916b43-6dcb-4111-a14d-9beb0454c351","_cell_guid":"2993daf3-088c-48d0-a38d-b61f43b3d730","trusted":true,"execution":{"iopub.execute_input":"2021-01-10T05:51:02.612138Z","iopub.status.busy":"2021-01-10T05:51:02.611554Z","iopub.status.idle":"2021-01-10T05:51:06.094293Z","shell.execute_reply":"2021-01-10T05:51:06.093287Z"},"papermill":{"duration":3.540969,"end_time":"2021-01-10T05:51:06.094415","exception":false,"start_time":"2021-01-10T05:51:02.553446","status":"completed"},"tags":[]},"cell_type":"code","source":"#model.add(keras.applications.Xception(\n#    input_shape=(img_height, img_width, 3), \n#    weights=Xception_NOTOP, \n#    include_top=False)\n#)\nbase_model = keras.applications.EfficientNetB3(\n    include_top = False, \n    weights = NOISY_IMAGENET, \n    input_shape=(img_height, img_width, 3))\n\nbase_model.trainable = False\n\n#model.add(efn.EfficientNetB3(\n#    include_top = False, \n#    weights = NOISY_STUDENT_B3, \n#    input_shape=(img_height, img_width, 3),\n#))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Fine-tuning\nOnce your model has converged on the new data, you can try to unfreeze all or part of the base model and retrain the whole model end-to-end with a very low learning rate.\n\nThis is an optional last step that can potentially give you incremental improvements. It could also potentially lead to quick overfitting -- keep that in mind."},{"metadata":{"trusted":true},"cell_type":"code","source":"inputs = keras.Input(shape=(img_width, img_height, 3))\n# We make sure that the base_model is running in inference mode here,\n# by passing `training=False`. This is important for fine-tuning, as you will\n# learn in a few paragraphs.\n\nx = base_model(inputs, training=False)\n# Convert features of shape `base_model.output_shape[1:]` to vectors\nx = keras.layers.GlobalAveragePooling2D()(x)\nx = keras.layers.Dense(128, activation='relu')(x)\nx = keras.layers.Dropout(DROPOUT_RATE)(x)\n# A Dense classifier with 5 unit (SparseCategoricalCrossentropy)\noutputs = layers.Dense(5, activation='softmax', dtype='float32', name='predictions')(x)\nmodel = keras.Model(inputs, outputs)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9b167012-6339-48f9-9b8a-d89701726486","_cell_guid":"f2259b21-2a17-4614-bd0f-1f688187a45e","trusted":true,"papermill":{"duration":0.021059,"end_time":"2021-01-10T05:51:02.531187","exception":false,"start_time":"2021-01-10T05:51:02.510128","status":"completed"},"tags":[]},"cell_type":"markdown","source":"The structure of our deep learning model is as follows:\n\n1. Global Average Pooling technique to reduce the image shape and apply pooling on spatial dimensions.\n1. Dense layer to provide the probability of predictions for all 5 classes, acting as a output layer.\n1. If you are new to Deep Learning and want to understand the functionality of Global Average Pooling layer: click here"},{"metadata":{"_uuid":"1f067861-e052-42fc-ba68-4e7b98400b29","_cell_guid":"2183e063-d81a-4a27-8dbe-2ed515259f5f","trusted":true,"execution":{"iopub.execute_input":"2021-01-10T05:51:02.456824Z","iopub.status.busy":"2021-01-10T05:51:02.456276Z","iopub.status.idle":"2021-01-10T05:51:02.488298Z","shell.execute_reply":"2021-01-10T05:51:02.488706Z"},"papermill":{"duration":0.05968,"end_time":"2021-01-10T05:51:02.488829","exception":false,"start_time":"2021-01-10T05:51:02.429149","status":"completed"},"tags":[]},"cell_type":"code","source":"#model = keras.models.Sequential()\n\n#model.add(base_model)\n\n#model.add(keras.layers.GlobalAveragePooling2D())\n\n#model.add(layers.Dropout(DROPOUT_RATE))\n\n#model.add(keras.layers.Dense(128, activation='relu'))\n\n#model.add(layers.Dropout(DROPOUT_RATE))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bbb9d5b7-0630-4738-831f-853412cbdde5","_cell_guid":"9efa12c3-d5e1-488d-836f-af8a757f417a","trusted":true,"papermill":{"duration":0.020908,"end_time":"2021-01-10T05:51:06.188918","exception":false,"start_time":"2021-01-10T05:51:06.16801","status":"completed"},"tags":[]},"cell_type":"markdown","source":"The dense layer is a neural network layer that is connected deeply, which means each neuron in the dense layer receives input from all neurons of its previous layer. T"},{"metadata":{"_uuid":"b9d9dc61-336d-4280-913e-240f8976019e","_cell_guid":"558908da-1a6a-4ebd-9463-d943ee4ac075","trusted":true,"execution":{"iopub.execute_input":"2021-01-10T05:51:06.236278Z","iopub.status.busy":"2021-01-10T05:51:06.235756Z","iopub.status.idle":"2021-01-10T05:51:06.245543Z","shell.execute_reply":"2021-01-10T05:51:06.245123Z"},"papermill":{"duration":0.035558,"end_time":"2021-01-10T05:51:06.245633","exception":false,"start_time":"2021-01-10T05:51:06.210075","status":"completed"},"tags":[]},"cell_type":"code","source":"#model.add(keras.layers.Dense(5, activation='softmax', dtype='float32', name='predictions'))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fd8fdfc8-92fb-4f59-9e3a-50ec1a1c4bf9","_cell_guid":"df481ec1-e6f9-4bf9-9626-1055db83ed58","trusted":true,"papermill":{"duration":0.021924,"end_time":"2021-01-10T05:51:06.505811","exception":false,"start_time":"2021-01-10T05:51:06.483887","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Print model summary"},{"metadata":{"_uuid":"e04cd22e-7054-4616-95d8-63cf21d9fae9","_cell_guid":"9a496c54-8945-4014-83f0-fb8b503ef877","trusted":true,"execution":{"iopub.execute_input":"2021-01-10T05:51:06.568194Z","iopub.status.busy":"2021-01-10T05:51:06.558615Z","iopub.status.idle":"2021-01-10T05:51:06.574224Z","shell.execute_reply":"2021-01-10T05:51:06.574801Z"},"papermill":{"duration":0.047293,"end_time":"2021-01-10T05:51:06.574927","exception":false,"start_time":"2021-01-10T05:51:06.527634","status":"completed"},"tags":[]},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f1e9e600-c643-4d81-adb3-830e8fade358","_cell_guid":"cf9a5b76-107a-4205-9e17-cff9296384a3","trusted":true,"papermill":{"duration":0.021742,"end_time":"2021-01-10T05:51:06.619717","exception":false,"start_time":"2021-01-10T05:51:06.597975","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Callback"},{"metadata":{"_uuid":"0c70a601-0055-4b39-8ff3-efa6d72cc062","_cell_guid":"38b9c7f7-efa0-4ba8-9898-29a602e48c16","trusted":true,"papermill":{"duration":0.021624,"end_time":"2021-01-10T05:51:06.663154","exception":false,"start_time":"2021-01-10T05:51:06.64153","status":"completed"},"tags":[]},"cell_type":"markdown","source":"A callback is an object that can perform actions at various stages of training (e.g. at the start or end of an epoch, before or after a single batch, etc).  In this model we will be using 3 callbacks as below:"},{"metadata":{"_uuid":"b771924f-8770-4989-8b77-0d161dced889","_cell_guid":"72301f52-7561-4c7a-9bd7-fb2f89f1ed66","trusted":true,"papermill":{"duration":0.021918,"end_time":"2021-01-10T05:51:06.751098","exception":false,"start_time":"2021-01-10T05:51:06.72918","status":"completed"},"tags":[]},"cell_type":"markdown","source":"1. ModelCheckpoint : Callback to save the Keras model or model weights at some frequency."},{"metadata":{"_uuid":"8561e8a8-4a39-493c-a049-02b7a37533d3","_cell_guid":"2d6307ee-aed2-4351-b8d1-39c8f35a0706","trusted":true,"execution":{"iopub.execute_input":"2021-01-10T05:51:06.801661Z","iopub.status.busy":"2021-01-10T05:51:06.799935Z","iopub.status.idle":"2021-01-10T05:51:06.802276Z","shell.execute_reply":"2021-01-10T05:51:06.802697Z"},"papermill":{"duration":0.029661,"end_time":"2021-01-10T05:51:06.802805","exception":false,"start_time":"2021-01-10T05:51:06.773144","status":"completed"},"tags":[]},"cell_type":"code","source":"model_checkpoint = keras.callbacks.ModelCheckpoint(\n    './best_weights.h5',\n    monitor=\"val_loss\",\n    verbose=1,\n    save_best_only=True,\n    save_weights_only=True,\n    mode=\"min\"\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4d2f1258-ef0d-4172-a9bc-49d012c86791","_cell_guid":"6e4401c6-ba16-4ced-8b87-785e18c0f05e","trusted":true,"papermill":{"duration":0.021786,"end_time":"2021-01-10T05:51:06.846305","exception":false,"start_time":"2021-01-10T05:51:06.824519","status":"completed"},"tags":[]},"cell_type":"markdown","source":"2. EarlyStopping : Stop training when a monitored metric has stopped improving."},{"metadata":{"_uuid":"2ae57950-5b1a-4e14-b29e-960843372d08","_cell_guid":"59dce581-e3d2-41e8-89b9-1941fd533080","trusted":true,"execution":{"iopub.execute_input":"2021-01-10T05:51:06.895699Z","iopub.status.busy":"2021-01-10T05:51:06.893963Z","iopub.status.idle":"2021-01-10T05:51:06.896318Z","shell.execute_reply":"2021-01-10T05:51:06.896736Z"},"papermill":{"duration":0.028718,"end_time":"2021-01-10T05:51:06.896835","exception":false,"start_time":"2021-01-10T05:51:06.868117","status":"completed"},"tags":[]},"cell_type":"code","source":"early_stopping = keras.callbacks.EarlyStopping(\n    monitor=\"val_loss\",\n    min_delta=0.0005,\n    patience=5,\n    verbose=1,\n    mode=\"min\",\n    restore_best_weights=True,\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e35ed7fe-e61f-4067-93a5-a39f19f1d2f5","_cell_guid":"941d43b7-aee8-4b02-8733-912b71c8b072","trusted":true,"papermill":{"duration":0.021694,"end_time":"2021-01-10T05:51:06.942001","exception":false,"start_time":"2021-01-10T05:51:06.920307","status":"completed"},"tags":[]},"cell_type":"markdown","source":"3. ReduceLROnPlateau : Reduce learning rate when a metric has stopped improving."},{"metadata":{"trusted":true},"cell_type":"code","source":"# Using an LR ramp up because fine-tuning a pre-trained model.\n# Starting with a high LR would break the pre-trained weights.\n\nLR_START = 0.00001\nLR_MAX = 0.0001\nLR_MIN = 0.00001\nLR_RAMPUP_EPOCHS = 3\nLR_SUSTAIN_EPOCHS = 0\nLR_EXP_DECAY = 0.85\n\ndef lrfn(epoch):\n    if epoch < LR_RAMPUP_EPOCHS:\n        lr = (LR_MAX - LR_START) / LR_RAMPUP_EPOCHS * epoch + LR_START\n    elif epoch < LR_RAMPUP_EPOCHS + LR_SUSTAIN_EPOCHS:\n        lr = LR_MAX\n    else:\n        #cosine decay\n        progress = (epoch - LR_RAMPUP_EPOCHS) / (EPOCHS - LR_RAMPUP_EPOCHS)\n        lr = LR_MAX * (0.5 * (1.0 + tf.math.cos(np.pi * ((1.0 * progress) % 1.0))))\n        \n        #exponential decay\n        #lr = (LR_MAX - LR_MIN) * LR_EXP_DECAY**(epoch - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS) + LR_MIN\n    return lr\n    \n#setting verbose=True allows us to see LR in model training\nlr_callback_dimitry = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose = True)\n\n#visualizing the learning rate schedule\nrng = [i for i in range(EPOCHS)]\ny = [lrfn(x) for x in rng]\n\nsns.set(style='whitegrid')\nplt.figure(figsize=(13, 5))\nplt.xlabel('Epoch')\nplt.ylabel('Learning Rate')\nplt.plot(rng, y)\nprint(\"Learning rate schedule: {:.3g} to {:.3g} to {:.3g}\".format(y[0], max(y), y[-1]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"initial_learning_rate = 0.0001\n\ndef lr_exp_decay(epoch, lr):\n    k = 0.2\n    return initial_learning_rate * tf.math.exp(-k*epoch)\n\n#setting verbose=True allows us to see LR in model training\nlr_callback = tf.keras.callbacks.LearningRateScheduler(lr_exp_decay, verbose = True)\n\n#visualizing the learning rate schedule\nrng = [i for i in range(EPOCHS)]\ny = [lr_exp_decay(x, initial_learning_rate) for x in rng]\n\nsns.set(style='whitegrid')\nplt.figure(figsize=(13, 5))\nplt.xlabel('Epoch')\nplt.ylabel('Learning Rate')\nplt.plot(rng, y)\n\nprint(\"Learning rate schedule: {:.3g} to {:.3g} to {:.3g}\".format(y[0], max(y), y[-1]))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"da1c8097-fab3-40ea-bbf2-0bc6fcf994e5","_cell_guid":"3db9196f-9a54-4e6b-a882-fc00f8fb0756","trusted":true,"execution":{"iopub.execute_input":"2021-01-10T05:51:06.99141Z","iopub.status.busy":"2021-01-10T05:51:06.989974Z","iopub.status.idle":"2021-01-10T05:51:06.992428Z","shell.execute_reply":"2021-01-10T05:51:06.992852Z"},"papermill":{"duration":0.029,"end_time":"2021-01-10T05:51:06.99295","exception":false,"start_time":"2021-01-10T05:51:06.96395","status":"completed"},"tags":[]},"cell_type":"code","source":"# Exponential decay\n\nreduce_lr = keras.callbacks.ReduceLROnPlateau(\n    monitor=\"val_loss\",\n    factor=0.1,\n    patience=2,\n    verbose=1,\n    mode=\"min\",\n    min_delta=0.0001,\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import datetime\nlog_dir = \"./\" + datetime.datetime.now().strftime(\"%Y%m%d-%H%M%S\")\ntensorboard_callback = tf.keras.callbacks.TensorBoard(log_dir=log_dir, histogram_freq=1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"96d9a616-5aa5-494d-9f2c-77ad3b330294","_cell_guid":"16c8c395-3610-4afc-afb1-17f613d2bae3","trusted":true},"cell_type":"markdown","source":"4.Stream the epoch loss to a file in JSON format. The file content is not well-formed JSON but rather has a JSON object per line."},{"metadata":{"_uuid":"babb80ef-38b2-42e0-bef2-abad19809677","_cell_guid":"c030ed17-f8bf-4f25-a19a-284b4dba11be","trusted":true},"cell_type":"code","source":"import json\njson_log = open('/kaggle/working/loss_log.json', mode='w', buffering=1)\njson_logging_callback = keras.callbacks.LambdaCallback(\n    on_train_begin=lambda logs: json_log.write('{ \"train\": [ \\n'),\n    on_epoch_end=lambda epoch,logs: json_log.write(json.dumps({'epoch': epoch, 'loss': logs['loss'], 'acc':logs['accuracy']}) + ',\\n'),\n    on_train_end=lambda logs: json_log.write(']\\n}'),\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1ea7bce4-7b14-430c-a720-b32d9eb0a44c","_cell_guid":"1dc65e7e-f26e-4230-8355-46c1029bb793","trusted":true,"papermill":{"duration":0.021876,"end_time":"2021-01-10T05:51:07.036965","exception":false,"start_time":"2021-01-10T05:51:07.015089","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Model Compile"},{"metadata":{"_uuid":"10fe4aa1-ba40-45d9-95fd-c9389d96d1ad","_cell_guid":"5eb699a9-c764-4037-bfb4-c2ac5e88d191","trusted":true,"papermill":{"duration":0.021868,"end_time":"2021-01-10T05:51:07.081004","exception":false,"start_time":"2021-01-10T05:51:07.059136","status":"completed"},"tags":[]},"cell_type":"markdown","source":"For this model we will be using Adamax optimizer and since our output labels are categorical we will be using SparseCategoricalAccuracy as a loss function.\n\nUse this crossentropy loss function when there are two or more label classes. We expect labels to be provided as integers.\n\ntf.keras.metrics.SparseCategoricalAccuracy\n\nThis metric creates two local variables, total and count that are used to compute the frequency with which y_pred matches y_true. This frequency is ultimately returned as sparse categorical accuracy: an idempotent operation that simply divides total by count\n"},{"metadata":{"_uuid":"abb6413f-7c2d-45b5-934f-fb6d3e8dd033","_cell_guid":"ad85a0ab-2088-410b-9a3f-fb9406a02a0a","trusted":true,"execution":{"iopub.execute_input":"2021-01-10T05:51:07.139991Z","iopub.status.busy":"2021-01-10T05:51:07.13912Z","iopub.status.idle":"2021-01-10T05:51:07.147179Z","shell.execute_reply":"2021-01-10T05:51:07.146708Z"},"papermill":{"duration":0.044331,"end_time":"2021-01-10T05:51:07.14727","exception":false,"start_time":"2021-01-10T05:51:07.102939","status":"completed"},"tags":[]},"cell_type":"code","source":"loss = tf.keras.losses.SparseCategoricalCrossentropy(from_logits=False)\noptimizer = tf.keras.optimizers.Adam(learning_rate=0.001)\n\nmodel.compile(\n            loss=loss,\n            optimizer='adam',\n            metrics=['sparse_categorical_accuracy', 'accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"616f54bb-9bba-4907-baa3-ba0977df3244","_cell_guid":"c18bad17-8ce4-458c-85a2-2e6dd9dce0c1","trusted":true,"papermill":{"duration":0.021961,"end_time":"2021-01-10T05:51:07.19135","exception":false,"start_time":"2021-01-10T05:51:07.169389","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Start Training"},{"metadata":{"_uuid":"08686560-785f-424d-be56-37d964663d1d","_cell_guid":"ac4be2d3-3ced-4408-8948-538d845cabd3","trusted":true,"papermill":{"duration":0.022004,"end_time":"2021-01-10T05:51:07.235479","exception":false,"start_time":"2021-01-10T05:51:07.213475","status":"completed"},"tags":[]},"cell_type":"markdown","source":"We have defined everything we need, it's time to train the model..."},{"metadata":{"_uuid":"9b46ede3-fc74-4699-8a8e-bd30b389120d","_cell_guid":"8f4e0702-0260-4b5d-9410-12b901e5b4f8","trusted":true,"execution":{"iopub.execute_input":"2021-01-10T05:51:07.286163Z","iopub.status.busy":"2021-01-10T05:51:07.285614Z","iopub.status.idle":"2021-01-10T08:33:29.651181Z","shell.execute_reply":"2021-01-10T08:33:29.649172Z"},"papermill":{"duration":9742.393762,"end_time":"2021-01-10T08:33:29.651338","exception":false,"start_time":"2021-01-10T05:51:07.257576","status":"completed"},"tags":[]},"cell_type":"code","source":"print(\"Started Training...\")\n\n# Trains the model for a fixed number of epochs (iterations on a dataset).\nhistory = model.fit(\n    train_datagen_flow,\n    epochs=epochs,\n    steps_per_epoch=(len(train) * (1-validation_split)) // BATCH_SIZE,\n    validation_data=valid_datagen_flow,\n    validation_steps=(len(train) * validation_split) // BATCH_SIZE,\n    callbacks = [\n        model_checkpoint, \n        early_stopping, \n        lr_callback_dimitry,\n        json_logging_callback\n    ],\n    use_multiprocessing=False,\n    verbose=1\n)\nprint(\"Training completed\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.save('./Model_Trained_Freezed')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Do a round of fine-tuning of the entire model\nFinally, let's unfreeze the base model and train the entire model end-to-end with a low learning rate.\n\nImportantly, although the base model becomes trainable, it is still running in inference mode since we passed training=False when calling it when we built the model. This means that the batch normalization layers inside won't update their batch statistics. If they did, they would wreck havoc on the representations learned by the model so far."},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"# Unfreeze the base_model. Note that it keeps running in inference mode\n# since we passed `training=False` when calling it. This means that\n# the batchnorm layers will not update their batch statistics.\n# This prevents the batchnorm layers from undoing all the training\n# we've done so far.\nbase_model.trainable = True\nmodel.summary()\n\nmodel.compile(\n    optimizer=keras.optimizers.Adam(1e-5),  # Low learning rate\n    loss=keras.losses.SparseCategoricalCrossentropy(from_logits=False),\n    metrics=['sparse_categorical_accuracy', 'accuracy'],\n)\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"After 10 epochs, fine-tuning gains us a nice improvement here."},{"metadata":{"trusted":true},"cell_type":"code","source":"epochs = 10\nmodel.fit(train_datagen_flow, epochs=epochs, validation_data=valid_datagen_flow)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.save('./Model_Trained_UnFreezed')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"92845fd9-88e0-4b2d-8380-9569b245635b","_cell_guid":"4def5d1c-2942-44ec-bafd-a22e2b62aa50","trusted":true,"papermill":{"duration":2.002997,"end_time":"2021-01-10T08:33:46.448212","exception":false,"start_time":"2021-01-10T08:33:44.445215","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Submission"},{"metadata":{"_uuid":"91d722c5-7cfe-428e-9b63-a92f10c5c199","_cell_guid":"8d7111df-b9f7-4bf1-9b82-190ba8f7f32c","trusted":true,"execution":{"iopub.execute_input":"2021-01-10T08:33:51.169332Z","iopub.status.busy":"2021-01-10T08:33:51.168652Z","iopub.status.idle":"2021-01-10T08:33:52.485542Z","shell.execute_reply":"2021-01-10T08:33:52.484125Z"},"papermill":{"duration":3.669798,"end_time":"2021-01-10T08:33:52.485666","exception":false,"start_time":"2021-01-10T08:33:48.815868","status":"completed"},"tags":[]},"cell_type":"code","source":"submission = pd.DataFrame(columns=['image_id','label'])\nfor image_name in os.listdir(general_path + 'test_images'):\n    image_path = os.path.join(general_path + 'test_images', image_name)\n    image = tf.keras.preprocessing.image.load_img(image_path)\n    resized_image = image.resize((img_width, img_height))\n    numpied_image = np.expand_dims(resized_image, 0)\n    tensored_image = tf.cast(numpied_image, tf.float32)\n    submission = submission.append(pd.DataFrame({'image_id': image_name,\n                                                 'label': model.predict_classes(tensored_image)}))\n\nsubmission.to_csv('/kaggle/working/submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"059d535c-0597-499c-b38e-94133426c587","_cell_guid":"0eccbc5d-ada4-4abd-a405-98cfa708479d","trusted":true,"papermill":{"duration":1.947825,"end_time":"2021-01-10T08:33:33.974399","exception":false,"start_time":"2021-01-10T08:33:32.026574","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Performance\nSince the performance metric for this competition is accuracy, let us plot the train and validation accuracy to monitor our model performance."},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(13, 5))\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title(\"Model Loss\")\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend(['Train', 'Test'])\nplt.ylim(ymax = 2, ymin = 0)\nplt.grid()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(13, 5))\nplt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\nplt.title('Model Accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend(['Train','Test'])\nplt.grid()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(13, 5))\nplt.plot(history.history['sparse_categorical_accuracy'])\nplt.plot(history.history['val_sparse_categorical_accuracy'])\nplt.title('Sparse Categorical Accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Sparse Categorical Accuracy')\nplt.legend(['Train','Test'])\nplt.grid()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"388aacb5-dc60-4645-a981-518a31194b56","_cell_guid":"b784df81-40d0-4ca5-bb46-39c5282842fb","trusted":true,"execution":{"iopub.execute_input":"2021-01-10T08:33:42.051567Z","iopub.status.busy":"2021-01-10T08:33:42.05061Z","iopub.status.idle":"2021-01-10T08:33:42.501162Z","shell.execute_reply":"2021-01-10T08:33:42.500552Z"},"papermill":{"duration":2.400111,"end_time":"2021-01-10T08:33:42.501267","exception":false,"start_time":"2021-01-10T08:33:40.101156","status":"completed"},"tags":[]},"cell_type":"code","source":"import json\nfile = open('/kaggle/working//loss_log.json', 'r')\ncountriesStr = file.read()\ncountriesStr = countriesStr[::-1].replace(',', '', 1)[::-1]\nfile.close()\n\nwith open('/kaggle/working//loss_log.json', 'w') as file:\n    file.write(countriesStr)\n    file.close()\n\njsonFile = open('/kaggle/working//loss_log.json', 'r')\njson_array = json.load(jsonFile)\n\nfor item in json_array['train']:\n    last_epoch = item['epoch']\n    \nlast_epoch+=1\n\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\nepochs_range = range(last_epoch)\n\nplt.figure(figsize=(8, 8))\nplt.subplot(1, 2, 1)\nplt.plot(epochs_range, acc, label='Training Accuracy')\nplt.plot(epochs_range, val_acc, label='Validation Accuracy')\nplt.legend(loc='lower right')\nplt.title('Training and Validation Accuracy')\n\nplt.subplot(1, 2, 2)\nplt.plot(epochs_range, loss, label='Training Loss')\nplt.plot(epochs_range, val_loss, label='Validation Loss')\nplt.legend(loc='upper right')\nplt.title('Training and Validation Loss')\nplt.show()\n#json_log.close()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f3245985-989f-4a8a-9a04-032e6c78661e","_cell_guid":"edd38d90-84ca-45a0-83bc-814ff714766d","trusted":true,"execution":{"iopub.execute_input":"2021-01-10T08:33:56.552427Z","iopub.status.busy":"2021-01-10T08:33:56.551195Z","iopub.status.idle":"2021-01-10T08:33:56.553822Z","shell.execute_reply":"2021-01-10T08:33:56.554264Z"},"papermill":{"duration":2.062046,"end_time":"2021-01-10T08:33:56.554412","exception":false,"start_time":"2021-01-10T08:33:54.492366","status":"completed"},"tags":[]},"cell_type":"code","source":"from sys import exit\nexit()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}