{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport os\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm_notebook\nfrom tensorflow.keras import models, layers\nfrom tensorflow.keras import callbacks\nfrom tensorflow.keras.callbacks import LearningRateScheduler\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import shuffle\n%matplotlib inline\nfrom tensorflow.keras.applications import EfficientNetB4\nimport time\nimport json\nimport seaborn as sns\nfrom sklearn.model_selection import KFold\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"target_dir = '../input/cassava-leaf-disease-classification/'\n\nos.listdir(target_dir)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train = pd.read_csv(os.path.join(target_dir, \"train.csv\"))\ndf_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.countplot(df_train.label, edgecolor = 'black',\n              palette = sns.color_palette(\"viridis\", 5))\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\ndf_train.image_id = df_train.image_id.astype('str')\ndf_train.label = df_train.label.astype('str')\ntrain_df, valid_df = train_test_split(df_train, test_size = 0.2, shuffle = True, stratify = df_train['label'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"batch_size = 16\nheight, width = 400, 400\n#rescale=1./127.5-1\ntrain_generator = ImageDataGenerator(preprocessing_function = tf.keras.applications.efficientnet.preprocess_input).flow_from_dataframe(train_df, directory=target_dir+'train_images',\n                                                                          target_size=(height, width),\n                                                                          x_col='image_id',\n                                                                          y_col='label',\n                                                                          color_mode='rgb',\n                                                                          batch_size = batch_size,\n                                                                          shuffle = True,\n                                                                          rotation_range=90,\n                                                                          width_shift_range=0.2,\n                                                                          height_shift_range=0.2,\n                                                                          shear_range=0.2,\n                                                                          zoom_range=0.2,\n                                                                          channel_shift_range=10,\n                                                                          horizontal_flip=True,\n                                                                          fill_mode='nearest',\n                                                                          validate_filenames=False)                                                                        \n                                                                    \nvalid_generator = ImageDataGenerator(preprocessing_function = tf.keras.applications.efficientnet.preprocess_input).flow_from_dataframe(valid_df, directory=target_dir+'train_images',\n                                                                         target_size=(height, width),\n                                                                         x_col='image_id',\n                                                                         y_col='label',\n                                                                         color_mode='rgb',\n                                                                         batch_size = batch_size,\n                                                                         shuffle = False, \n                                                                         validate_filenames=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"early_stopping_callback = tf.keras.callbacks.EarlyStopping(monitor='val_accuracy', patience=5)\n\nmodel_path = './best_model_EfficientNetB4_cassava-leaf.h5'  # 模型儲存的位置\n\n\n# 建立 Checkpoint\ncheckpoint = \\\n    callbacks.ModelCheckpoint(model_path,\n                              verbose=1,\n                              monitor='val_accuracy',  # 儲存模型的指標\n                              save_best_only=True,  # 是否只儲存最好的\n                              mode='max')           # 與指標搭配模式","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#定義learning rate,根據 epoch 要如何變動\ndef schedule(epoch):  \n    if epoch < 10:\n        return 0.0005\n    elif epoch < 20:\n        return 0.0003\n    elif epoch < 30:\n        return 0.0001\n    else:\n        return 0.00005\n    \nclass CustomCallback(tf.keras.callbacks.Callback):\n    def on_epoch_begin(self, epoch, logs=None):\n        print()\n        print(self.model.optimizer.learning_rate)\n        print(f\"目前的learning rate是{float(tf.kears.baclend.get_value(self.model.optimizer.lr))}\")\n        \nlr_schedule = callbacks.LearningRateScheduler(schedule, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"EfficientNetB4_model =EfficientNetB4(\n    include_top=False, weights='imagenet',input_shape=(height, width, 3))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for layer in EfficientNetB4_model.layers:\n    layer.trainable = True","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.regularizers import l1_l2\n### initializing CNN   #加上API\nnum_class = 5\nl1_alpha, l2_alpha = [1e-6, 1e-6]\nmodel = tf.keras.models.Sequential([\n     EfficientNetB4_model,\n     tf.keras.layers.GlobalAveragePooling2D(),\n     tf.keras.layers.Flatten(),\n     tf.keras.layers.Dense(256, activation='relu',bias_regularizer=tf.keras.regularizers.L1L2(l1=0.01, l2=0.01)),\n     tf.keras.layers.Dropout(0.4),\n     tf.keras.layers.Dense(num_class, activation='softmax')\n])\nopt = tf.keras.optimizers.Adam()\nmodel.compile(optimizer = opt,\n              loss = 'categorical_crossentropy',   #sparse_categorical_crossentropy\n              metrics = ['accuracy'])\n\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history = model.fit_generator(train_generator,\n                    #steps_per_epoch=64,\n                    epochs=50,\n                    verbose=1, #0不顯示,1,2顯示數值 \n                    callbacks=[lr_schedule, checkpoint, early_stopping_callback],\n                    validation_data=valid_generator,\n                    validation_steps=None,\n                    validation_freq=1,\n                    class_weight=None,\n                    max_queue_size=10,\n                    workers=1,\n                    use_multiprocessing=False,\n                    shuffle=True                              \n)\n\n#儲存model\nmodel.save(\"./best_model_EfficientNetB4_cassava-leaf.h5\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#視覺化訓練過程\nplt.plot(model.history.history['loss'], 'b', label='train_generator')\nplt.plot(model.history.history['val_loss'], 'r', label='valid_generator')\nplt.legend()\nplt.title('Loss')\nplt.show()\n\nplt.plot(model.history.history['accuracy'], 'b', label='train_generator')\nplt.plot(model.history.history['val_accuracy'], 'r', label='valid_generator')\nplt.legend(loc=4)\nplt.title('Accuracy')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.models import load_model\nticks_begin = time.time()\nmodel_ = load_model(\"./best_model_EfficientNetB4_cassava-leaf.h5\")\nticks_end = time.time()\ntime_cost = ticks_end - ticks_begin\nloss, acc = model_.evaluate_generator(valid_generator)\nprint('\\n','val_accuracy分數:',round(acc,3),'\\n','val_loss分數:',round(loss,3))\nprint('time', round(time_cost,3))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ss = pd.read_csv(os.path.join(target_dir, \"sample_submission.csv\"))\nss","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from PIL import Image\n\npreds = []\n\nfor image_id in ss.image_id:\n    image = Image.open(os.path.join(target_dir,  \"test_images\", image_id))\n    image = image.resize((height, width))\n    image = np.expand_dims(image, axis = 0)\n    preds.append(np.argmax(model.predict(image)))\n\nss['label'] = preds\nss","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"result=ss.to_csv('submission.csv', index = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}