{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Calling out all required packages for prediction"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt \nimport plotly.express as px\nimport os\nimport cv2\nfrom PIL import Image\nimport keras\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nimport tensorflow as tf\nfrom tensorflow.keras import models, layers\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras.applications import EfficientNetB7\nfrom tensorflow.keras.optimizers import Adam,Nadam","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Calling GPU servies"},{"metadata":{"trusted":true},"cell_type":"code","source":"try: # detect TPUs\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver() # TPU detection\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept ValueError: # no TPU found, detect GPUs\n    #strategy = tf.distribute.MirroredStrategy() # for GPU or multi-GPU machines\n    strategy = tf.distribute.get_strategy() # default strategy that works on CPU and single GPU\n    #strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy() # for clusters of multi-GPU machines\n\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\n\nprint(\"Number of accelerators: \", strategy.num_replicas_in_sync)\nprint(\"BATCH_SIZE: \", str(BATCH_SIZE))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"reading training data"},{"metadata":{"trusted":true},"cell_type":"code","source":"PATH_ = '../input/cassava-leaf-disease-classification/'\n\ntrain_imag = PATH_+ 'train_images/'\ntrain = pd.read_csv(PATH_+'train.csv')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"declaring requried variables\nTarget size is for dividing the no of pixels this will differ from efficientNet B0 to B7 please click on the link for more inforation about the efficientnet in keras\n[https://keras.io/api/applications/efficientnet/](http://)"},{"metadata":{"trusted":true},"cell_type":"code","source":"BATCH_SIZE =16 #Mini-Batch Gradient Descent\nSTEPS_PER_EPOCH = len(train)*0.8 / BATCH_SIZE\nVALIDATION_STEPS = len(train)*0.2 / BATCH_SIZE\nEPOCHS = 30\nTARGET_SIZE = 224","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"1. Converting the train label to string to since it's in integer and working on classification problem\n2. Creating additional data using image data generator with below parameter for training the model with differnet set visuals\n    rotation range is used to change the angle of the image with given range \n    Zoom range is used to capture the image the with certain range of zoom\n    horizontal/ vertical flip used for trun the image for different visuals \n\n3. implementing image data generator for training and validation\n\nMore information on preprocessing https://keras.io/api/preprocessing/image/ "},{"metadata":{"trusted":true},"cell_type":"code","source":"train.label = train.label.astype('str')\n\ntrain_datagen = ImageDataGenerator(validation_split = 0.2,\n                                     rotation_range = 40,\n                                     zoom_range = 0.4,\n                                     horizontal_flip = True,\n                                     vertical_flip = True,\n                                     fill_mode = 'nearest',\n                                     shear_range = 0.15,\n                                     height_shift_range = 0.15,\n                                     width_shift_range = 0.15,\n                                     featurewise_center = True,\n                                     featurewise_std_normalization = True)\n\ntrain_generator = train_datagen.flow_from_dataframe(train,\n                         directory = os.path.join('../input/cassava-leaf-disease-classification/train_images'),\n                         subset = \"training\",\n                         x_col = \"image_id\",\n                         y_col = \"label\",\n                         target_size = (TARGET_SIZE, TARGET_SIZE),\n                         batch_size = BATCH_SIZE,\n                         class_mode = \"sparse\",\n                         shuffle= True)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nvalidation_datagen = ImageDataGenerator(validation_split = 0.2)\n\nvalidation_generator = validation_datagen.flow_from_dataframe(train,\n                         directory = os.path.join('../input/cassava-leaf-disease-classification/train_images'),\n                         subset = \"validation\",\n                         x_col = \"image_id\",\n                         y_col = \"label\",\n                         target_size = (TARGET_SIZE, TARGET_SIZE),\n                         batch_size = BATCH_SIZE,\n                         class_mode = \"sparse\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"calling the Efficientnet B7 model with parameter"},{"metadata":{"trusted":true},"cell_type":"code","source":"eff_b7 = EfficientNetB7(include_top=False, input_tensor=None,\n    pooling=None, input_shape=(TARGET_SIZE, TARGET_SIZE, 3), classifier_activation='softmax')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"1. Calling the efficientnetB7 \n2. adding global max pooling \n3. adding dense layer of 5 neurons(5 different classes) and activation method as softmax"},{"metadata":{"trusted":true},"cell_type":"code","source":"model = eff_b7.output\nmodel = layers.GlobalMaxPooling2D()(model)\nmodel = layers.Dense(5, activation = \"softmax\")(model)\nmodel = models.Model(eff_b7.input, model)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Adding Nadam optimizer used to identify the gradient which is faster than adam \n\n[https://keras.io/api/optimizers/Nadam/](http://)"},{"metadata":{"trusted":true},"cell_type":"code","source":"model.compile(optimizer = Nadam(lr = 0.001),\n              loss = \"sparse_categorical_crossentropy\",\n              metrics = [\"acc\"])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Our model summary which contains efficientnet B7, global max pooling and dense layer of 5 neurons "},{"metadata":{"trusted":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"ModelCheckpoint: Callback to save the Keras model or model weights at some frequency.\n\nEarlyStopping: Stop training when a monitored metric has stopped improving.\n\nReduceLROnPlateau: Reduce learning rate when a metric has stopped improving.\n\nfor more information use this link [https://keras.io/api/callbacks/](http://)"},{"metadata":{"trusted":true},"cell_type":"code","source":"model_save = ModelCheckpoint('./effici_b7_best_weights.h5', \n                              save_best_only = True, \n                              save_weights_only = True,\n                              monitor = 'val_loss', \n                              mode = 'min', verbose = 1)\nearly_stop = EarlyStopping(monitor = 'val_loss', min_delta = 0.001,\n                           patience = 5, mode = 'min', verbose = 1,\n                           restore_best_weights = True)\nreduce_lr = ReduceLROnPlateau(monitor = 'val_loss',factor = 0.3,\n                              patience = 2, min_delta = 0.001,\n                              mode = 'min', verbose = 1) #reduced learning rate","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history = model.fit(\n     train_generator,\n     steps_per_epoch = STEPS_PER_EPOCH,\n     epochs = EPOCHS,\n     validation_data = validation_generator,\n     validation_steps = VALIDATION_STEPS,\n     callbacks = [model_save, early_stop, reduce_lr])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Plotting Training Accuray vs Value Accuracy"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(15, 5))\nplt.plot(history.history['acc'], 'b*-', label=\"train_acc\")\nplt.plot(history.history['val_acc'], 'r*-', label=\"val_acc\")\nplt.grid()\nplt.title(\"train_acc vs val_acc\")\nplt.ylabel(\"Accuracy\")\nplt.xlabel(\"Epochs\")\nplt.legend()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Plotting Training loss vs Value loss"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(15, 5))\nplt.plot(history.history['loss'], 'b*-', label=\"train_loss\")\nplt.plot(history.history['val_loss'], 'r*-', label=\"val_loss\")\nplt.grid()\nplt.title(\"train_loss - val_loss\")\nplt.ylabel(\"Loss\")\nplt.xlabel(\"Epochs\")\nplt.legend()\nplt.show()\n\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"loading the trained model to submit the test prediction"},{"metadata":{},"cell_type":"markdown","source":"Prediction on test images"},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = []\nsample_sub = pd.read_csv('../input/cassava-leaf-disease-classification/sample_submission.csv')\n\nfor image in sample_sub.image_id:\n    img = keras.preprocessing.image.load_img('../input/cassava-leaf-disease-classification/test_images/' + image)\n    img = keras.preprocessing.image.img_to_array(img)\n    img = keras.preprocessing.image.smart_resize(img, (TARGET_SIZE, TARGET_SIZE))\n    img = np.expand_dims(img, 0)\n    prediction = model.predict(img)\n    preds.append(np.argmax(prediction))\n\nmy_submission = pd.DataFrame({'image_id': sample_sub.image_id, 'label': preds})\nmy_submission.to_csv('submission.csv', index=False) ","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}