{"cells":[{"metadata":{},"cell_type":"markdown","source":"## Package Imports"},{"metadata":{"_uuid":"eaa1560f-fae6-400a-97c3-e097537bc309","_cell_guid":"49fc82eb-48ea-492f-a9a9-b3d0183c7f6c","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import models\nfrom tensorflow.keras import layers","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Image Data Generator with a simple rescaling and random horizontal flip augmentation"},{"metadata":{"_uuid":"1577beb3-8a98-47a9-82a3-a91d4cd54a4b","_cell_guid":"bdd81773-a557-4010-ab42-5eb34d2f2db2","trusted":true},"cell_type":"code","source":"proj_dir = '../input/cassava-leaf-disease-classification/train_images/'\ntrain = pd.read_csv('../input/cassava-leaf-disease-classification/train.csv')\ntrain.loc[:,'label'] = train.loc[:,'label'].astype('str')\nBATCH_SIZE = 64\nSPLIT = 0.2\n\ntrain_datagen = ImageDataGenerator(rescale=1./255,\n        horizontal_flip=True,\n        validation_split=SPLIT)\n\ntrain_generator = train_datagen.flow_from_dataframe(\n        train,\n        directory = proj_dir,\n        x_col = 'image_id',\n        y_col = 'label',\n        target_size=(448, 448),\n        batch_size=BATCH_SIZE,\n        subset = 'training',\n        class_mode='categorical')\n\nval_generator = train_datagen.flow_from_dataframe(\n        train,\n        directory = proj_dir,\n        x_col = 'image_id',\n        y_col = 'label',\n        target_size=(448, 448),\n        batch_size=BATCH_SIZE,\n        subset = 'validation',\n        class_mode='categorical')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a045058a-37d9-41ef-8b85-ace3d3c8f5e2","_cell_guid":"6d887234-5388-4356-babb-5de0f90d331a","trusted":true},"cell_type":"markdown","source":"## Xception Model pretrained on imagenet dataset has been used\nHave loaded the model from another notebook, it can be directly imported in this notebook itself by enabling the internet connection\n\nThe output is passed through 2 dense layers and finally through the output softmax layer to get the class probabilities"},{"metadata":{"_uuid":"96ec55fd-c093-4ca0-8a5e-b6c4394377c3","_cell_guid":"72a3aa7b-f0ff-4f4b-b907-f68f13a86bce","trusted":true},"cell_type":"code","source":"conv_base = tf.keras.models.load_model('../input/pretrained-models/xception')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1ea4e40e-7913-461d-b209-802bb994df65","_cell_guid":"ebda61b0-dbf9-4b25-89cc-19cdf5fc1ac4","trusted":true},"cell_type":"code","source":"conv_base.summary()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"22a7f6da-297a-495f-911d-218f396f56bc","_cell_guid":"a4b79179-9b09-4c5a-925d-09b5e0973f12","trusted":true},"cell_type":"code","source":"model = models.Sequential()\nmodel.add(conv_base)\nmodel.add(layers.MaxPooling2D((2,2)))\nmodel.add(layers.Flatten())\nmodel.add(layers.Dense(1024,activation='relu',kernel_regularizer=tf.keras.regularizers.l2(l2=0.005)))\nmodel.add(layers.Dense(256,activation='relu',kernel_regularizer=tf.keras.regularizers.l2(l2=0.005)))\nmodel.add(layers.Dense(5,activation='softmax'))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Have just fined tuned the final layer of xception for now\nUnfreezing the upstream layers in xception further leads to increase in accuracy"},{"metadata":{"_uuid":"0bd18b56-24c8-4482-8d7b-0c4f0a6cf3d4","_cell_guid":"8d6e16e4-cc94-42d6-9130-403bc53c72b6","trusted":true},"cell_type":"code","source":"print(len(conv_base.trainable_weights))\nconv_base.trainable=True\n\nset_trainable=False\n\nfor layer in conv_base.layers :\n    if layer.name == 'block14_sepconv1' :\n        set_trainable=True\n    if set_trainable:\n        layer.trainable=True\n    else:\n        layer.trainable=False\n\nprint(len(conv_base.trainable_weights))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4068ac4c-4aa7-4c28-8e11-b65825ee2a84","_cell_guid":"de1c877b-138b-4e89-8233-be79505fbe16","trusted":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Label smoothing has been done to account for the available noise in the dataset"},{"metadata":{"_uuid":"5298881c-aa2d-4e20-a3c0-2231507fde4e","_cell_guid":"573c5c84-d1e7-4aff-816c-ec8cc18a16de","trusted":true},"cell_type":"code","source":"from keras import optimizers\n\nmodel.compile(loss=tf.keras.losses.CategoricalCrossentropy(label_smoothing=0.2),optimizer='Adamax',metrics=['acc'])\n\ncallback = tf.keras.callbacks.EarlyStopping(\n    monitor='val_loss', min_delta=0.05, patience=3, verbose=0,\n    mode='min', baseline=None, restore_best_weights=True\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"62c7a5d4-8bc7-46d3-aa1f-a80c6066aee5","_cell_guid":"ac6ca9bd-c62a-42e8-9344-c711668052f6","trusted":true},"cell_type":"code","source":"history = model.fit_generator(train_generator,epochs=20,steps_per_epoch=int(len(train)*(1-SPLIT)/BATCH_SIZE),callbacks=[callback],validation_data=val_generator,validation_steps=int(len(train)*SPLIT/BATCH_SIZE))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Plotting the accuracy and loss for training and validations sets"},{"metadata":{"_uuid":"0143f43f-6748-4011-8af1-35f7eef459ca","_cell_guid":"aa3f0dc2-99c1-4464-b7e0-67191d811017","trusted":true},"cell_type":"code","source":"def smooth_points(points,factor=0.85):\n    smoothed_points=[]\n    for point in points:\n        if smoothed_points:\n            previous=smoothed_points[-1]\n            smoothed_points.append(previous*factor+point*(1-factor))\n        else:\n            smoothed_points.append(point)\n    return smoothed_points\n\nimport matplotlib.pyplot as plt\n\nacc=history.history['acc']\nval_acc=history.history['val_acc']\n\nloss=history.history['loss']\nval_loss=history.history['val_loss']\n\nepochs=range(1,len(acc)+1)\n\nplt.plot(epochs,acc,'bo',label='Training acc')\nplt.plot(epochs,val_acc,'b',label='Validation acc')\nplt.title('Training and validation accuracy')\nplt.legend()\n\nplt.figure()\n\nplt.plot(epochs,loss,'bo',label='Training loss')\nplt.plot(epochs,val_loss,'b',label='Validation loss')\nplt.title('Training and validation loss')\nplt.legend()\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e7d684bf-d000-468b-be52-d4f0374ecd74","_cell_guid":"392037df-620c-41e0-9110-a1b99ecf9d7e","trusted":true},"cell_type":"markdown","source":"> ## Based on the accuracy and validation loss curve ;15 epochs was found to be appropriate "},{"metadata":{"_uuid":"381cab62-f91e-46f5-8e08-8e6ebb3d64ca","_cell_guid":"1c49a805-c8fd-41bf-ae87-06408337aef2","trusted":true},"cell_type":"code","source":"test_generator.reset()\n\npred = model.predict_generator(test_generator,verbose=1,steps = len(test))\n\npredicted_class_indices = np.argmax(pred,axis=1)\n\nlabels = (train_generator.class_indices)\nlabels = dict((v,k) for k,v in labels.items())\npredictions = [labels[k] for k in predicted_class_indices]\n\nfilenames=test_generator.filenames\nresults=pd.DataFrame({\"image_id\":filenames,\n                      \"label\":predictions})","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cfa15937-1264-4024-aaf4-272fec04690b","_cell_guid":"98720e4d-79e3-4721-9a45-304180e36dc5","trusted":true},"cell_type":"code","source":"results.to_csv('/kaggle/working/submission.csv',index=False)\nmodel.save('/kaggle/working/model')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}