{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport time\nimport shutil\nimport random\nimport cv2\nimport pandas as pd\nimport seaborn as sn\nimport tensorflow as tf\nimport tensorflow_hub as hub\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import classification_report\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import EfficientNetB7\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.applications.efficientnet import preprocess_input\nfrom tensorflow.keras.models import *\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.optimizers import *\nfrom tensorflow.keras.utils import *\nfrom tensorflow.keras.callbacks import *\nfrom tensorflow.keras.initializers import *\nfrom kaggle_datasets import KaggleDatasets\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2021-11-21T01:46:16.837492Z","iopub.execute_input":"2021-11-21T01:46:16.838157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install tensorflow-addons","metadata":{"execution":{"iopub.status.busy":"2021-11-21T01:31:58.708512Z","iopub.execute_input":"2021-11-21T01:31:58.708794Z","iopub.status.idle":"2021-11-21T01:32:07.008861Z","shell.execute_reply.started":"2021-11-21T01:31:58.708766Z","shell.execute_reply":"2021-11-21T01:32:07.007258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow_addons as tfa","metadata":{"execution":{"iopub.status.busy":"2021-11-21T01:32:43.858827Z","iopub.execute_input":"2021-11-21T01:32:43.859172Z","iopub.status.idle":"2021-11-21T01:32:43.997684Z","shell.execute_reply.started":"2021-11-21T01:32:43.85914Z","shell.execute_reply":"2021-11-21T01:32:43.996836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\n# Detect hardware, return appropriate distribution strategy\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()  # TPU detection. No parameters necessary if TPU_NAME environment variable is set. On Kaggle this is always the case.\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy() # default distribution strategy in Tensorflow. Works on CPU and single GPU.\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2021-11-21T01:08:27.711469Z","iopub.execute_input":"2021-11-21T01:08:27.711681Z","iopub.status.idle":"2021-11-21T01:08:33.694189Z","shell.execute_reply.started":"2021-11-21T01:08:27.711656Z","shell.execute_reply":"2021-11-21T01:08:33.693301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GCS_DS_PATH = KaggleDatasets().get_gcs_path(\"fgvc8aug\")\n\n# Configuration\nEPOCHS = 30\nBATCH_SIZE = 8 * strategy.num_replicas_in_sync\nIM_Z = 768\nCLASSES = 6","metadata":{"execution":{"iopub.status.busy":"2021-11-21T01:27:09.308918Z","iopub.execute_input":"2021-11-21T01:27:09.309561Z","iopub.status.idle":"2021-11-21T01:27:09.659861Z","shell.execute_reply.started":"2021-11-21T01:27:09.309515Z","shell.execute_reply":"2021-11-21T01:27:09.658997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GCS_DS_PATH","metadata":{"execution":{"iopub.status.busy":"2021-11-21T01:08:34.07509Z","iopub.execute_input":"2021-11-21T01:08:34.075631Z","iopub.status.idle":"2021-11-21T01:08:34.08464Z","shell.execute_reply.started":"2021-11-21T01:08:34.075588Z","shell.execute_reply":"2021-11-21T01:08:34.083739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def format_train_path(fname):\n    return GCS_DS_PATH+'/data_full_augmentation_images/data_full_augmentation/images/'+fname\n\ndef format_test_path(fname):\n    return GCS_DS_PATH+'/test_images/'+fname","metadata":{"execution":{"iopub.status.busy":"2021-11-21T01:08:34.086224Z","iopub.execute_input":"2021-11-21T01:08:34.086842Z","iopub.status.idle":"2021-11-21T01:08:34.093536Z","shell.execute_reply.started":"2021-11-21T01:08:34.086805Z","shell.execute_reply":"2021-11-21T01:08:34.092738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir=\"../input/fgvc8aug/data_full_augmentation_images/data_full_augmentation/images\"\ntest_dir=\"../input/plant-pathology-2021-fgvc8/test_images\"\ndf_train=pd.read_csv('../input/fgvc8aug/data.csv')\ndf_sub = pd.read_csv('../input/plant-pathology-2021-fgvc8/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2021-11-21T01:24:08.747778Z","iopub.execute_input":"2021-11-21T01:24:08.74808Z","iopub.status.idle":"2021-11-21T01:24:08.810415Z","shell.execute_reply.started":"2021-11-21T01:24:08.748036Z","shell.execute_reply":"2021-11-21T01:24:08.809601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_paths = df_train.image.apply(format_train_path)\ntest_paths = df_sub.image.apply(format_test_path)","metadata":{"execution":{"iopub.status.busy":"2021-11-21T01:25:48.536459Z","iopub.execute_input":"2021-11-21T01:25:48.536854Z","iopub.status.idle":"2021-11-21T01:25:48.55944Z","shell.execute_reply.started":"2021-11-21T01:25:48.536827Z","shell.execute_reply":"2021-11-21T01:25:48.558453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train[[\"image\", \"labels\"]]\nmlb = MultiLabelBinarizer().fit(df_train.labels.apply(lambda x : x.split()))\nlabels = pd.DataFrame(mlb.transform(df_train.labels.apply(lambda x : x.split())), columns = mlb.classes_)\n\nlabels = pd.concat([df_train['image'], labels], axis=1)\nlabels.head()","metadata":{"execution":{"iopub.status.busy":"2021-11-21T01:24:11.197748Z","iopub.execute_input":"2021-11-21T01:24:11.198024Z","iopub.status.idle":"2021-11-21T01:24:11.468379Z","shell.execute_reply.started":"2021-11-21T01:24:11.197987Z","shell.execute_reply":"2021-11-21T01:24:11.467475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = np.float32(labels.loc[:, 'complex':'scab'].values)\ntrain_paths, valid_paths, train_labels, valid_labels =\\\ntrain_test_split(train_paths, train_labels, test_size=0.1)","metadata":{"execution":{"iopub.status.busy":"2021-11-21T01:24:14.208058Z","iopub.execute_input":"2021-11-21T01:24:14.208334Z","iopub.status.idle":"2021-11-21T01:24:14.223503Z","shell.execute_reply.started":"2021-11-21T01:24:14.208306Z","shell.execute_reply":"2021-11-21T01:24:14.222575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decode_image(filename, label=None, image_size=(IM_Z, IM_Z)):\n    bits = tf.io.read_file(filename)\n    image = tf.image.decode_jpeg(bits, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0\n    image = tf.image.resize(image, image_size)\n    \n    if label is None:\n        return image\n    else:\n        return image, label","metadata":{"execution":{"iopub.status.busy":"2021-11-21T01:25:17.08954Z","iopub.execute_input":"2021-11-21T01:25:17.090141Z","iopub.status.idle":"2021-11-21T01:25:17.095321Z","shell.execute_reply.started":"2021-11-21T01:25:17.090104Z","shell.execute_reply":"2021-11-21T01:25:17.094582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((train_paths, train_labels))\n    .map(decode_image, num_parallel_calls=AUTO)\n    .cache()\n    .repeat()\n    .shuffle(512)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((valid_paths, valid_labels))\n    .map(decode_image, num_parallel_calls=AUTO)\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n\ntest_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(test_paths)\n    .map(decode_image, num_parallel_calls=AUTO)\n    .batch(BATCH_SIZE)\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-21T01:25:18.728911Z","iopub.execute_input":"2021-11-21T01:25:18.729495Z","iopub.status.idle":"2021-11-21T01:25:18.908599Z","shell.execute_reply.started":"2021-11-21T01:25:18.72946Z","shell.execute_reply":"2021-11-21T01:25:18.907851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model():\n    model = tf.keras.models.Sequential(name='EfficientNetB7')\n    \n    model.add(EfficientNetB7(\n        include_top=False,\n        input_shape=(IM_Z, IM_Z, 3),\n        weights='imagenet',\n        pooling='avg'))\n    \n    model.add(tf.keras.layers.Dense(CLASSES, \n        kernel_initializer=tf.keras.initializers.RandomUniform(seed=32),\n        bias_initializer=tf.keras.initializers.Zeros(), name='dense_top'))\n    model.add(tf.keras.layers.Activation('sigmoid', dtype='float32'))\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2021-11-21T01:29:21.927989Z","iopub.execute_input":"2021-11-21T01:29:21.928588Z","iopub.status.idle":"2021-11-21T01:29:21.935872Z","shell.execute_reply.started":"2021-11-21T01:29:21.928524Z","shell.execute_reply":"2021-11-21T01:29:21.935165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    model = get_model()\n        \n    model.compile(\n                loss=tf.keras.losses.BinaryCrossentropy(),\n                optimizer='adam',\n                metrics=[\n                    tf.keras.metrics.BinaryAccuracy(name='accuracy'), \n                    tfa.metrics.F1Score(\n                        num_classes=CLASSES, \n                        average='macro')])\n    model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-11-21T01:32:49.077744Z","iopub.execute_input":"2021-11-21T01:32:49.078041Z","iopub.status.idle":"2021-11-21T01:33:22.896944Z","shell.execute_reply.started":"2021-11-21T01:32:49.078004Z","shell.execute_reply":"2021-11-21T01:33:22.896034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint = ModelCheckpoint(\n    \"./B7-tpu.h5\",\n    monitor = 'val_accuracy',\n    mode = 'max',\n    save_best_only = True,\n    save_weights_only= False ,\n    perior = 1,\n    verbose = 1\n)\n\nearly_stopping = EarlyStopping(\n    monitor = 'val_accuracy',\n    mode = 'auto',\n    min_delta = 0.0001,\n    patience = 3,\n    baseline = None,\n    restore_best_weights = True,\n    verbose = 1\n)\n\nreduce_lr = ReduceLROnPlateau(\n    monitor = 'val_accuracy',\n    mode = 'auto',\n    min_lr = 0.0000001,\n    min_delta =.0001,\n    patience = 2,\n    factor = np.sqrt(0.1),\n    cooldown = 0,\n    verbose = 1\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-21T01:37:15.408222Z","iopub.execute_input":"2021-11-21T01:37:15.408751Z","iopub.status.idle":"2021-11-21T01:37:15.415199Z","shell.execute_reply.started":"2021-11-21T01:37:15.40871Z","shell.execute_reply":"2021-11-21T01:37:15.414458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEPS_PER_EPOCH = train_labels.shape[0] // BATCH_SIZE ","metadata":{"execution":{"iopub.status.busy":"2021-11-21T01:37:16.275515Z","iopub.execute_input":"2021-11-21T01:37:16.275805Z","iopub.status.idle":"2021-11-21T01:37:16.280437Z","shell.execute_reply.started":"2021-11-21T01:37:16.275774Z","shell.execute_reply":"2021-11-21T01:37:16.279576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_dataset, \n    epochs=EPOCHS, \n    callbacks=[checkpoint, early_stopping, reduce_lr, CSVLogger('./B7-hist.log')],\n    steps_per_epoch=STEPS_PER_EPOCH,\n    validation_data=valid_dataset\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-21T01:37:19.067582Z","iopub.execute_input":"2021-11-21T01:37:19.068227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_hist(path):\n    history = pd.read_csv(path)\n\n    acc = history['accuracy']\n    val_acc = history['val_accuracy']\n\n    f1 = history['f1_score']\n    val_f1 = history['val_f1_score']\n\n    loss = history['loss']\n    val_loss = history['val_loss']\n    plt.style.use('fivethirtyeight')\n    plt.figure(figsize=(30, 10))\n\n    plt.subplot(1, 2, 3)\n    plt.plot(acc, label='Training Accuracy')\n    plt.plot(val_acc, label='Validation Accuracy')\n    plt.legend(loc='lower right')\n    plt.ylabel('Accuracy')\n    plt.ylim([min(plt.ylim()), 1])\n    plt.title('Training and Validation Accuracy')\n    plt.xlabel('epoch')\n\n    plt.subplot(1, 2, 3)\n    plt.plot(loss, label='Training Loss')\n    plt.plot(val_loss, label='Validation Loss')\n    plt.legend(loc='upper right')\n    plt.ylabel('Categorical Crossentropy')\n    plt.ylim([min(plt.ylim()), max(plt.ylim())])\n    plt.title('Training and Validation Loss')\n    \n    plt.subplot(1, 3, 3)\n    plt.plot(f1, label='Training F1-score')\n    plt.plot(val_f1, label='Validation F1-score')\n    plt.legend(loc='lower right')\n    plt.ylabel('F1-score')\n    plt.ylim([min(plt.ylim()), max(plt.ylim())])\n    plt.title('Training and Validation F1-score')\n\n    plt.xlabel('epoch')\n    plt.savefig('evaluation.jpg')\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_hist('./B7-hist.log')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}