{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Setup"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom kaggle_datasets import KaggleDatasets\nfrom tensorflow import keras\nfrom functools import partial\nfrom sklearn.model_selection import train_test_split\nfrom PIL import Image\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.layers import Dense, Dropout\n# from tensorflow.keras.applications import EfficientNetB0\n# from tensorflow.keras.applications import EfficientNetB4\nfrom tensorflow.keras.applications import EfficientNetB5\nfrom keras.callbacks import EarlyStopping\nfrom pathlib import Path\n\nprint(\"Tensorflow version \" + tf.__version__)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"root_path = '../input/cassava-leaf-disease-classification/'","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Read the input data"},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv(root_path + 'train.csv')\ntrain['label'] = train['label'].astype('string')\ntrain.sample(5)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Disease catagory"},{"metadata":{"trusted":true},"cell_type":"code","source":"names_of_disease = pd.read_json(root_path + 'label_num_to_disease_map.json', typ='series')\nnames_of_disease","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from PIL import Image,ImageFilter\nimport os\n\nplt.figure(figsize=(16, 12))\nfor i in range(9):\n    plt.subplot(3, 3, i + 1)\n    image = Image.open(root_path + 'train_images/' + train.iloc[i]['image_id'])\n    print(root_path + 'train_images/' + train.iloc[i]['image_id'])\n    array = np.array(image)\n    plt.imshow(array)\n    label=train.iloc[i]['label']\n    print(label)\n    plt.title(f'{names_of_disease[int(label)]}')\n    break\nplt.show()\n\n  \n#Read image\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sizes = []\nfor i in range(1, len(train), 250):\n    image = Image.open(root_path + 'train_images/' + train.iloc[i]['image_id'])\n    array = np.array(image)\n    sizes.append(array.shape)\nprint('Picture size', set(sizes))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"128*800/600.0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# img_width, img_height = 224, 224\n# img_width, img_height = 128, 128\n# img_width, img_height = 164, 164\nimg_width, img_height = 256, 256","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Check Data input Distribution"},{"metadata":{"trusted":true},"cell_type":"code","source":"train['label'].value_counts(normalize=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Training"},{"metadata":{"trusted":true},"cell_type":"code","source":"datagen = ImageDataGenerator(validation_split=0.2,\n                             vertical_flip=True,\n                             horizontal_flip=True,\n                             rotation_range=90,\n                             brightness_range=[0.5,1.0],\n                             shear_range=25,\n                             zoom_range=[0.5,1.0]\n                            )                            \n\ntrain_datagen_flow = datagen.flow_from_dataframe(\n    dataframe=train,\n    directory=root_path + 'train_images',\n    x_col='image_id',\n    y_col='label',\n    target_size=(img_width, img_height),\n    batch_size=32,\n    subset='training',\n    shuffle = True,\n    #seed=12345,\n    class_mode='categorical'\n)\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Test (Validating)"},{"metadata":{"trusted":true},"cell_type":"code","source":"valid_datagen_flow = datagen.flow_from_dataframe(\n    dataframe=train,\n    directory=root_path + 'train_images',\n    x_col='image_id',\n    y_col='label',\n    target_size=(img_width, img_height),\n    batch_size=32,\n    subset='validation',\n    #seed=12345,\n    class_mode = 'categorical',\n    shuffle = True\n)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Build the Model"},{"metadata":{},"cell_type":"markdown","source":"# Running the model"},{"metadata":{},"cell_type":"markdown","source":"### * Version 2: 856/856 [==============================] - 1894s 2s/step - loss: 1.1880 - accuracy: 0.6146 - val_loss: 1.1774 - val_accuracy: 0.6165\n**\n* Version 3: 856/856 [==============================] - 274s 321ms/step - loss: 0.8554 - accuracy: 0.6764 - val_loss: 0.8232 - val_accuracy: 0.6843\n* Version 4: \n    image size 64x64  accuracy: 0.6823\n    856/856 [==============================] - 652s 762ms/step - loss: 0.8481 - accuracy: 0.6823 - val_loss: 0.8613 - val_accuracy: 0.6686\n* Version 5:\n    convolution... filters=64, kernel_size=4\n* Version 6: \n    back to basic  accuracy: 0.64  after epocs =8\n* Version 8: \n    Early stopping, epocs 100\n* Version 9: EfficientNetB0 epocs 10 --> accuracy: 0.7385\n\n* Version 11-13: EfficientNetB0 epocs 50 --> accuracy: 0.84  validation=0.72\n* Version 14: EfficientNetB0 epocs 20 --> accuracy:?\n* Version 15: TODO: Batchsize 128 and dropout 0.6  make the model worst\n* Version 16: Batchsize 32 + flatten +drop+ dense512 +ephocs =10  accuracy: 0.74  validation=0.72 \n* Version 17: image size 128x128 accuracy: 0.81  validation=0.79\n* Version 18: image size 256x256 , epoch=6 (<8H) occuracy 0.84 validation=0.8441\n* version 19: image size 128X170 , epoch=6  occuracy=0.787 Valid=0.787\n* version 20: image size 164X164 , epoch=6 , change from B0 to EfficientNetB4\n* version 21: image size 164X164 , epoch=8 , change from B0 to EfficientNetB4 -accuaracy 0.81 validation 0.79\n* Version 22: image size 164X164 , epoch=6 , change from B4 to EfficientNetB5  accuracy 078 valid=0.79\n* Version 23: fix output submission\n* version 25: save model file  +epoch=10  accuracy: 0.8186 - val_loss: 0.5699 - val_accuracy: 0.8004  (score 0.772)\n* version 26: 256X256 +epoch=10 loss: 0.4181 - accuracy: 0.8583 - val_loss: 0.4416 - val_accuracy: 0.8537  (score 0.862)\n* version 26: 256X256 BatchNormalization(32) +epoch=10 got worst\n* Version:  use callback for dynamic learn-rate ... callback [... ,lrs]  increase as we goes (big degradation from 0.8 to 0.6)\n* use callback for dynamic learn-rate ... callback [... ,lrs]  decrease as we goes \n\nBase on: https://www.kaggle.com/bununtadiresmenmor/starter-keras-efficientnet?select=sample_submission.csv\n\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"early_stopping = EarlyStopping(monitor='val_loss', verbose=1, patience=4)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"1. # Model EfficientNetB5"},{"metadata":{"trusted":true},"cell_type":"code","source":"model = Sequential()\nmodel.add(EfficientNetB5(include_top = False, weights = \"imagenet\",\n                        input_shape=(img_width, img_height, 3)))\n# model.add(Dropout(0.2))\n# model.add(tf.keras.layers.GlobalAveragePooling2D())\n\nmodel.add(tf.keras.layers.BatchNormalization())\nmodel.add(tf.keras.layers.AveragePooling2D(pool_size=(3, 3)))\n\n\nmodel.add(tf.keras.layers.Flatten())\nmodel.add(Dropout(0.5))\nmodel.add(tf.keras.layers.Dense(512, activation = \"relu\"))\nmodel.add(tf.keras.layers.Dense(64, activation = \"relu\"))\nmodel.add(tf.keras.layers.Dense(5, activation = \"softmax\"))\n# model.add(Dropout(0.5))\noptimizer = tf.keras.optimizers.Adam(learning_rate=1e-3)\n# optimizer = 'adam'\nmodel.compile(optimizer = optimizer,\n            loss = \"categorical_crossentropy\",\n            metrics = [\"accuracy\"])\n\n# from tensorflow.keras.applications import EfficientNetB0\n\n# with strategy.scope():\n#     inputs = layers.Input(shape=(img_width, img_height, 3))\n#     x = img_augmentation(inputs)\n#     outputs = EfficientNetB0(include_top=True, weights=None, classes=NUM_CLASSES)(x)\n\n#     model = tf.keras.Model(inputs, outputs)\n#     model.compile(\n#         optimizer=\"adam\", loss=\"categorical_crossentropy\", metrics=[\"accuracy\"]\n#     )\n\n# model.summary()\n\n# epochs = 40  # @param {type: \"slider\", min:10, max:100}\n# hist = model.fit(ds_train, epochs=epochs, validation_data=ds_test, verbose=2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras import utils\n\nutils.plot_model(model)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.callbacks import LearningRateScheduler\nlrs = LearningRateScheduler(my_learning_rate)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def my_learning_rate(epoch, lrate):\n\n if (epoch < 5) :\n  lrate = (1e-3)\n elif (epoch < 10) :\n  lrate = 1e-4\n else:\n  lrate = 1e-5\n    \n return lrate\n\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"my_learning_rate(9, 1e-4)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Run and save model"},{"metadata":{"trusted":true},"cell_type":"code","source":"history = model.fit_generator(train_datagen_flow,\n                            epochs = 15,\n                            validation_data = valid_datagen_flow,\n                             callbacks = [early_stopping,lrs])\n\n\nmodel.save('Cassava_model'+'.h5') ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n\ndef plot_hist(history):\n    plt.plot(history.history[\"accuracy\"])\n    plt.plot(history.history[\"val_accuracy\"])\n    plt.title(\"model accuracy\")\n    plt.ylabel(\"accuracy\")\n    plt.xlabel(\"epoch\")\n    plt.legend([\"train\", \"validation\"], loc=\"upper left\")\n    plt.show()\n\n\nplot_hist(history)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history_df = pd.DataFrame(history.history)\nhistory_df.loc[:, ['loss', 'val_loss']].plot()\nhistory_df.loc[:, ['accuracy', 'val_accuracy']].plot()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Logs"},{"metadata":{"trusted":true},"cell_type":"code","source":"# import os\n# import keras\n# RUN_NAME = 'run 1 with 25 nodes'\n# logger = keras.callbacks.TensorBoard(\n#     log_dir='kaggle/working/logs/ {}'.format(RUN_NAME),\n#     histogram_freq=5,\n#     write_graph=True\n# )\n# os.getcwd()\n# tf.tensorboard --logdir=/logs/\n# tensorboard --logdir=summaries","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Reload the model trained weights"},{"metadata":{"trusted":true},"cell_type":"code","source":"# model.load_weights('model_weights.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# os.path.join(root_path + 'test_images', image_name)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Predict Test Image"},{"metadata":{},"cell_type":"markdown","source":"Submission"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Evaluating the model\n\nimport keras\n\nfinal_model = keras.models.load_model('Cassava_model.h5')\n\nsubmission = pd.DataFrame(columns=['image_id','label'])\n\nfor image_name in os.listdir(root_path + 'test_images'):\n    image_path = os.path.join(root_path + 'test_images', image_name)\n    image = tf.keras.preprocessing.image.load_img(image_path)\n    resized_image = image.resize((img_width, img_height))\n    numpied_image = np.expand_dims(resized_image, 0)\n    tensored_image = tf.cast(numpied_image, tf.float32)\n    submission = submission.append(pd.DataFrame({'image_id': image_name,\n                                                 'label': final_model.predict_classes(tensored_image)}))\n\nsubmission","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.to_csv('submission.csv', index = False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}