{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Import thư viện","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2021-11-29T03:29:29.91576Z","iopub.execute_input":"2021-11-29T03:29:29.916137Z","iopub.status.idle":"2021-11-29T03:29:30.854095Z","shell.execute_reply.started":"2021-11-29T03:29:29.916037Z","shell.execute_reply":"2021-11-29T03:29:30.853163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import *\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.optimizers import *\nfrom tensorflow.keras.utils import *\nfrom tensorflow.keras.callbacks import *\nfrom tensorflow.keras.initializers import *\nimport tensorflow as tf\nfrom tensorflow.keras.applications import VGG16\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.applications.vgg16 import preprocess_input\nfrom tensorflow.keras.models import Model\nfrom sklearn.preprocessing import MultiLabelBinarizer\nimport tensorflow_addons as tfa\nfrom kaggle_datasets import KaggleDatasets","metadata":{"execution":{"iopub.status.busy":"2021-11-29T03:29:30.856073Z","iopub.execute_input":"2021-11-29T03:29:30.856324Z","iopub.status.idle":"2021-11-29T03:29:35.97531Z","shell.execute_reply.started":"2021-11-29T03:29:30.856296Z","shell.execute_reply":"2021-11-29T03:29:35.974355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\n# Detect hardware, return appropriate distribution strategy\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()  # TPU detection. No parameters necessary if TPU_NAME environment variable is set. On Kaggle this is always the case.\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy() # default distribution strategy in Tensorflow. Works on CPU and single GPU.\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2021-11-29T03:29:35.97691Z","iopub.execute_input":"2021-11-29T03:29:35.977158Z","iopub.status.idle":"2021-11-29T03:29:41.579614Z","shell.execute_reply.started":"2021-11-29T03:29:35.977131Z","shell.execute_reply":"2021-11-29T03:29:41.578773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EPOCHS = 10\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nWIDTH = 480\nHEIGHT = 480\nCHANNELS = 3\nLEARNING_RATE = 0.001\nCLASSES = 6\nSEED = 32\ntop_dropout_rate = 0.2","metadata":{"execution":{"iopub.status.busy":"2021-11-29T03:29:41.58176Z","iopub.execute_input":"2021-11-29T03:29:41.582159Z","iopub.status.idle":"2021-11-29T03:29:41.587377Z","shell.execute_reply.started":"2021-11-29T03:29:41.582116Z","shell.execute_reply":"2021-11-29T03:29:41.586776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GCS_DS_PATH = KaggleDatasets().get_gcs_path('fgvc8aug')\nTRAIN_PATH = GCS_DS_PATH + \"/data_full_augmentation_images/data_full_augmentation/images/\"\nprint(GCS_DS_PATH)","metadata":{"execution":{"iopub.status.busy":"2021-11-29T03:29:41.588245Z","iopub.execute_input":"2021-11-29T03:29:41.588521Z","iopub.status.idle":"2021-11-29T03:29:42.004011Z","shell.execute_reply.started":"2021-11-29T03:29:41.588492Z","shell.execute_reply":"2021-11-29T03:29:42.003351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_model = 'FGVC8-VGG16.h5'\nhist_path = 'FGVC8-VGG16.log'\ntrain_image = '../input/fgvc8aug/data_full_augmentation_images/data_full_augmentation/images'\ntrain_df = pd.read_csv('../input/fgvc8aug/data.csv', )","metadata":{"execution":{"iopub.status.busy":"2021-11-29T03:29:47.390416Z","iopub.execute_input":"2021-11-29T03:29:47.391013Z","iopub.status.idle":"2021-11-29T03:29:47.464924Z","shell.execute_reply.started":"2021-11-29T03:29:47.390979Z","shell.execute_reply":"2021-11-29T03:29:47.464242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df[[\"image\", \"labels\"]]\nmlb = MultiLabelBinarizer().fit(train_df.labels.apply(lambda x : x.split()))\nlabels = pd.DataFrame(mlb.transform(train_df.labels.apply(lambda x : x.split())), columns = mlb.classes_)\n\nlabels = pd.concat([train_df['image'], labels], axis=1)\nlabels.head()","metadata":{"execution":{"iopub.status.busy":"2021-11-29T03:29:50.197488Z","iopub.execute_input":"2021-11-29T03:29:50.198082Z","iopub.status.idle":"2021-11-29T03:29:50.490102Z","shell.execute_reply.started":"2021-11-29T03:29:50.19805Z","shell.execute_reply":"2021-11-29T03:29:50.489187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def format_path(st):\n    return TRAIN_PATH + st\n\ntrain_paths = labels.image.apply(format_path).values\n\ntrain_labels = np.float32(labels.loc[:, 'complex':'scab'].values)\ntrain_paths, valid_paths, train_labels, valid_labels =\\\ntrain_test_split(train_paths, train_labels, test_size=0.15, random_state=2020)","metadata":{"execution":{"iopub.status.busy":"2021-11-29T03:29:51.307701Z","iopub.execute_input":"2021-11-29T03:29:51.308205Z","iopub.status.idle":"2021-11-29T03:29:51.341452Z","shell.execute_reply.started":"2021-11-29T03:29:51.308153Z","shell.execute_reply":"2021-11-29T03:29:51.340523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_img(filepath,label):\n    image = tf.io.read_file(filepath)\n    image = tf.image.decode_jpeg(image, channels=CHANNELS)\n    image = tf.image.convert_image_dtype(image, tf.float32) \n    image = tf.image.resize(image, [HEIGHT,WIDTH])\n    return image,label","metadata":{"execution":{"iopub.status.busy":"2021-11-29T03:29:51.983217Z","iopub.execute_input":"2021-11-29T03:29:51.983517Z","iopub.status.idle":"2021-11-29T03:29:51.989337Z","shell.execute_reply.started":"2021-11-29T03:29:51.983485Z","shell.execute_reply":"2021-11-29T03:29:51.988391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((train_paths, train_labels))\n    .map(process_img, num_parallel_calls=AUTO)\n    .repeat()\n    .shuffle(512)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((valid_paths, valid_labels))\n    .map(process_img, num_parallel_calls=AUTO)\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-29T03:29:57.057006Z","iopub.execute_input":"2021-11-29T03:29:57.057305Z","iopub.status.idle":"2021-11-29T03:29:57.197585Z","shell.execute_reply.started":"2021-11-29T03:29:57.057273Z","shell.execute_reply":"2021-11-29T03:29:57.19674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Get model","metadata":{}},{"cell_type":"code","source":"def get_model():\n    VGG16_MODEL = tf.keras.applications.VGG16(weights='imagenet' ,include_top=False, input_shape=(HEIGHT, WIDTH, 3))\n    \n    x=VGG16_MODEL.output\n    x=GlobalAveragePooling2D()(x)\n    x=Dense(256,activation='relu')(x)\n    x=Dropout(0.2)(x)\n    x=Dense(128,activation='relu')(x)\n    prediction=Dense(6,activation='sigmoid')(x)\n\n    model=Model(inputs=VGG16_MODEL.input, outputs=prediction)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2021-11-29T03:30:11.507423Z","iopub.execute_input":"2021-11-29T03:30:11.507948Z","iopub.status.idle":"2021-11-29T03:30:11.514634Z","shell.execute_reply.started":"2021-11-29T03:30:11.507916Z","shell.execute_reply":"2021-11-29T03:30:11.513868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    model = get_model()\n\n    f1_score = tfa.metrics.F1Score(num_classes=6, threshold=0.4, average='micro')\n    model.compile(tf.keras.optimizers.Adam(learning_rate=0.0005) , loss='binary_crossentropy', metrics=[f1_score, 'accuracy'])","metadata":{"execution":{"iopub.status.busy":"2021-11-29T03:30:13.323227Z","iopub.execute_input":"2021-11-29T03:30:13.323956Z","iopub.status.idle":"2021-11-29T03:30:16.059811Z","shell.execute_reply.started":"2021-11-29T03:30:13.323922Z","shell.execute_reply":"2021-11-29T03:30:16.058956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-11-29T03:30:18.659769Z","iopub.execute_input":"2021-11-29T03:30:18.660071Z","iopub.status.idle":"2021-11-29T03:30:18.670477Z","shell.execute_reply.started":"2021-11-29T03:30:18.660042Z","shell.execute_reply":"2021-11-29T03:30:18.669565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(model, dpi=60)","metadata":{"execution":{"iopub.status.busy":"2021-11-29T03:30:23.596787Z","iopub.execute_input":"2021-11-29T03:30:23.597128Z","iopub.status.idle":"2021-11-29T03:30:24.538091Z","shell.execute_reply.started":"2021-11-29T03:30:23.597093Z","shell.execute_reply":"2021-11-29T03:30:24.536995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Trainning","metadata":{}},{"cell_type":"code","source":"checkpoint = ModelCheckpoint(\n    final_model,\n    monitor = 'val_accuracy',\n    mode = 'max',\n    save_best_only = True,\n    save_weights_only= False ,\n    perior = 1,\n    verbose = 1\n)\n\nearly_stopping = EarlyStopping(\n    monitor = 'val_accuracy',\n    mode = 'auto',\n    min_delta = 0.0001,\n    patience = 3,\n    baseline = None,\n    restore_best_weights = True,\n    verbose = 1\n)\ndef build_lrfn(lr_start=0.00001, lr_max=0.00005, \n               lr_min=0.00001, lr_rampup_epochs=5, \n               lr_sustain_epochs=0, lr_exp_decay=.8):\n    lr_max = lr_max * strategy.num_replicas_in_sync\n\n    def lrfn(epoch):\n        if epoch < lr_rampup_epochs:\n            lr = (lr_max - lr_start) / lr_rampup_epochs * epoch + lr_start\n        elif epoch < lr_rampup_epochs + lr_sustain_epochs:\n            lr = lr_max\n        else:\n            lr = (lr_max - lr_min) *\\\n                 lr_exp_decay**(epoch - lr_rampup_epochs\\\n                                - lr_sustain_epochs) + lr_min\n        return lr\n    return lrfn","metadata":{"execution":{"iopub.status.busy":"2021-11-29T03:30:30.025001Z","iopub.execute_input":"2021-11-29T03:30:30.025293Z","iopub.status.idle":"2021-11-29T03:30:30.036242Z","shell.execute_reply.started":"2021-11-29T03:30:30.025261Z","shell.execute_reply":"2021-11-29T03:30:30.035281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lrfn = build_lrfn()\nSTEPS_PER_EPOCH = train_labels.shape[0] // BATCH_SIZE\nlr_schedule = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2021-11-29T03:30:31.400908Z","iopub.execute_input":"2021-11-29T03:30:31.401193Z","iopub.status.idle":"2021-11-29T03:30:31.405802Z","shell.execute_reply.started":"2021-11-29T03:30:31.401163Z","shell.execute_reply":"2021-11-29T03:30:31.404989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = model.fit(\n    train_dataset, \n    validation_data = valid_dataset, \n    epochs = EPOCHS,\n    steps_per_epoch=STEPS_PER_EPOCH,\n    callbacks = [lr_schedule, early_stopping, checkpoint, CSVLogger(hist_path)]\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-29T03:30:32.97876Z","iopub.execute_input":"2021-11-29T03:30:32.979084Z"},"trusted":true},"execution_count":null,"outputs":[]}]}