{"cells":[{"metadata":{"_uuid":"c07709ef-f0ed-4311-8ba8-f1f895ba00a5","_cell_guid":"73486ba3-8bb7-4a8f-bca9-cf9c95b859b1","trusted":true},"cell_type":"markdown","source":"## This notebook in nutshell\n\n* Multimodal Deep Learning\n    * Image\n    * Text\n* Dynamic image augmentation rate\n    * GridMask image augmentation (https://arxiv.org/abs/2001.04086)\n    * Rotate, shear, zoom, shift from [Rotation Augmentation GPU/TPU - [0.96+]](https://www.kaggle.com/cdeotte/rotation-augmentation-gpu-tpu-0-96)\n    * [tf.image](https://www.tensorflow.org/api_docs/python/tf/image) functions\n* TF-IDF word representation\n    * L2 normalization\n    * Sublinear Term Frequency\n* EfficientNet B7 (https://arxiv.org/abs/1905.11946)\n* LAMB optimizer (https://arxiv.org/abs/1904.00962)\n* Global Average Pooling (https://arxiv.org/abs/1312.4400)\n* TPU","execution_count":null},{"metadata":{"_uuid":"66345b4c-794f-413c-90f3-18e88e5ae9a5","_cell_guid":"22cd88d0-361c-4a4a-bc69-10e8a20f147c","trusted":true},"cell_type":"markdown","source":"## Changelog\n\n* Version 2 : ?\n    * Change LR\n    * Optimize GridMask to use TPU\n    * Remove unused codes\n    * Merge code from https://www.kaggle.com/williammulianto/fork-of-kernel42b295af53?scriptVersionId=37641353\n* Version 1 : 0.15254\n    * Make sure there's no error","execution_count":null},{"metadata":{"_uuid":"623ff6ab-b389-478a-bb33-0283fb5c598c","_cell_guid":"85c5754b-ba0b-468a-8be6-2d16d44a7162","trusted":true},"cell_type":"markdown","source":"## Install, load and configure library","execution_count":null},{"metadata":{"_uuid":"17e34b54-d8fa-4334-ad33-f1a6effdc269","_cell_guid":"88a00a56-7fb7-408a-b986-968547f17ff3","trusted":true},"cell_type":"code","source":"!pip install --upgrade efficientnet tensorflow_addons tensorflow","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"67cc7fe6-ae69-4f8c-a148-dec8d5aa3d8f","_cell_guid":"16760e82-ac7c-46c4-b0ef-31b275e849b6","trusted":true},"cell_type":"code","source":"import math\nimport re\nimport random\nimport os\nimport tensorflow as tf\nimport tensorflow_addons as tfa\nimport numpy as np\nimport tensorflow.keras.backend as K\nimport efficientnet.tfkeras as efn\nimport efficientnet\nimport itertools\nimport matplotlib\nimport scipy\nimport pandas as pd\nimport sklearn\nfrom matplotlib import pyplot as plt\nfrom datetime import datetime","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e6863204-a209-4f49-94fc-fb328eeb7e1c","_cell_guid":"74714d96-44ec-4d90-b870-29c0e87aa1ba","trusted":true},"cell_type":"code","source":"print(f'Numpy version : {np.__version__}')\nprint(f'Tensorflow version : {tf.__version__}')\nprint(f'Tensorflow Addons version : {tfa.__version__}')\nprint(f'EfficientNet (library) version : {efficientnet.__version__}')\nprint(f'Matplotlib version : {matplotlib.__version__}')\nprint(f'Scipy version : {scipy.__version__}')\nprint(f'Pandas version : {pd.__version__}')\nprint(f'Scikit-Learn version : {sklearn.__version__}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip freeze > requirements.txt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"PRE_TRAINING_TIME_START = datetime.now()\nAUTO = tf.data.experimental.AUTOTUNE","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"SEED = 42\n\nos.environ['PYTHONHASHSEED']=str(SEED)\nrandom.seed(SEED)\nnp.random.seed(SEED)\n\nos.environ['TF_DETERMINISTIC_OPS']=str(SEED)\ntf.random.set_seed(SEED)\n# tf.config.threading.set_inter_op_parallelism_threads(1)\n# tf.config.threading.set_intra_op_parallelism_threads(1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"db35ac57-c2d9-407e-8cfb-ac12931831ca","_cell_guid":"0a918e64-d733-4859-8ce0-2289ec35798c","trusted":true},"cell_type":"markdown","source":"## TPU or GPU detection","execution_count":null},{"metadata":{"_uuid":"295fe847-8f32-4f03-b9fc-dec73b13f846","_cell_guid":"f5ad7339-2b74-4954-a472-5ee0b1c458ee","trusted":true},"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()  # TPU detection. No parameters necessary if TPU_NAME environment variable is set. On Kaggle this is always the case.\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy() # default distribution strategy in Tensorflow. Works on CPU and single GPU.\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"39299429-f9ac-4486-bc84-e9701b883854","_cell_guid":"fef69ec7-a513-45ff-adf2-893e96b0e4f9","trusted":true},"cell_type":"markdown","source":"# Configuration","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls -lha /kaggle/input/","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8d7d5846-689c-4812-9f83-4fbdabfbee1d","_cell_guid":"15239699-06e5-49c2-b028-8676f53361a4","trusted":true},"cell_type":"code","source":"from kaggle_datasets import KaggleDatasets\nIMAGE_SIZE = (512, 512)\n\nGCS_TRAIN_PATHS = [\n    KaggleDatasets().get_gcs_path('tfrecords'),\n    KaggleDatasets().get_gcs_path('tfrecords-2')\n]\nTRAINING_FILENAMES = []\nfor i in GCS_TRAIN_PATHS:\n    TRAINING_FILENAMES.append(tf.io.gfile.glob(i + '/*.tfrecords'))\nTRAINING_FILENAMES = list(itertools.chain.from_iterable(TRAINING_FILENAMES))\n\nGCS_TEST_PATH = KaggleDatasets().get_gcs_path('tfrecords-3')\nTEST_FILENAMES = tf.io.gfile.glob(GCS_TEST_PATH + '/*.tfrecords') # predictions on this dataset should be submitted for the competition\n\nprint(len(TRAINING_FILENAMES))\nprint(len(TEST_FILENAMES))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"EPOCHS = 9\nDO_AUG = True\n\n# BATCH_SIZE = 256\nBATCH_SIZE = 192\ncurrent_epoch = 0 # used to determine augmentation rate\nchance = 0\n\nNUM_TRAINING_IMAGES = 105390\nNUM_TEST_IMAGES = 12186\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"72639119-2c24-424f-8b3e-e19c8ec8d1e7","_cell_guid":"82195b29-693e-4440-8478-7971eda7ae15","trusted":true},"cell_type":"code","source":"CLASSES = [str(c).zfill(2) for c in range(0, 42)]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e6e318a2-7708-406f-b33f-6ce1dcc6c2ef","_cell_guid":"6f314c57-ac5c-409d-90e5-afb416e915c7","trusted":true},"cell_type":"markdown","source":"# Datasets functions","execution_count":null},{"metadata":{"_uuid":"28268dba-da6e-4ebd-b4be-7da94aab2110","_cell_guid":"d0064349-32a9-4fed-80a2-31868d60681d","trusted":true},"cell_type":"code","source":"def decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0  # convert image to floats in [0, 1] range\n    image = tf.reshape(image, [*IMAGE_SIZE, 3]) # explicit size needed for TPU\n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string), # tf.string means bytestring\n        \"words\": tf.io.FixedLenFeature([6633], tf.float32),  # shape [] means single element\n        \"label\": tf.io.FixedLenFeature([], tf.int64),  # shape [] means single element\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    words = example['words']\n    label = tf.cast(example['label'], tf.int32)\n    \n    return ((image, words), label) # returns a dataset of (image, label) pairs\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string), # tf.string means bytestring\n        \"words\": tf.io.FixedLenFeature([6633], tf.float32),  # shape [] means single element\n        \"filename\": tf.io.FixedLenFeature([], tf.string),  # shape [] means single element\n        # class is missing, this competitions's challenge is to predict flower classes for the test dataset\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    words = example['words']\n    filename = example['filename']\n    return ((image, words), filename) # returns a dataset of image(s)\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    # Read from TFRecords. For optimal performance, reading from multiple files at once and\n    # disregarding data order. Order does not matter since we will be shuffling the data anyway.\n\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False # disable order, increase speed\n\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO) # automatically interleaves reads from multiple files\n    dataset = dataset.with_options(ignore_order) # uses data as soon as it streams in, rather than in its original order\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord, num_parallel_calls=AUTO)\n    # returns a dataset of (image, label) pairs if labeled=True or (image, id) pairs if labeled=False\n    return dataset\n\ndef get_training_dataset(do_aug=True):\n    dataset = load_dataset(TRAINING_FILENAMES, labeled=True)\n    if do_aug:\n        dataset = dataset.map(image_augmentation, num_parallel_calls=AUTO)\n    dataset = dataset.repeat() # the training dataset must repeat for several epochs\n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO) # prefetch next batch while training (autotune prefetch buffer size)\n    return dataset\n\ndef get_test_dataset(ordered=False):\n    dataset = load_dataset(TEST_FILENAMES, labeled=False, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO) # prefetch next batch while training (autotune prefetch buffer size)\n    return dataset","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"98e001f9-c2e8-4bc0-84d8-c9d02504e3a6","_cell_guid":"b829d37c-6c9d-486b-9ed1-533d805a49c5","trusted":true},"cell_type":"markdown","source":"# List of image augmentation functions\n\n| Function   | Chance | Range                             |\n| ---------- | ------ | --------------------------------- |\n| Flip       | 50%    | Only Left to right                |\n| Brightness | 50%    | 0.9 to 1.1                        |\n| Contrast   | 50%    | 0.9 to 1.1                        |\n| Saturation | 50%    | 0.9 to 1.1                        |\n| Hue        | 50%    | 0.05                              |\n| Rotate     | 50%    | 17 degrees * normal distribution  |\n| Shear      | 50%    | 5.5 degrees * normal distribution |\n| Zoom Out   | 33%    | 1.0 - (normal distribution / 8.5) |\n| Shift      | 33%    | 18 pixel * normal distribution    |\n| GridMask   | 50%    | 100 - 160 pixel black rectangle   |\n|            |        | with same pixel range for gap     |","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"## Rotate, shear, zoom, shift","execution_count":null},{"metadata":{"_uuid":"299e214a-a4a5-456f-88e2-d1975caf9b24","_cell_guid":"afb27760-e259-4623-92ee-2a5be4d6adb9","trusted":true},"cell_type":"code","source":"@tf.function\ndef get_mat(rotation, shear, height_zoom, width_zoom, height_shift, width_shift):\n    # returns 3x3 transformmatrix which transforms indicies\n        \n    # CONVERT DEGREES TO RADIANS\n    rotation = math.pi * rotation / 180.\n    shear = math.pi * shear / 180.\n    \n    # ROTATION MATRIX\n    c1 = tf.math.cos(rotation)\n    s1 = tf.math.sin(rotation)\n    one = tf.constant([1],dtype='float32')\n    zero = tf.constant([0],dtype='float32')\n    rotation_matrix = tf.reshape( tf.concat([c1,s1,zero, -s1,c1,zero, zero,zero,one],axis=0),[3,3] )\n        \n    # SHEAR MATRIX\n    c2 = tf.math.cos(shear)\n    s2 = tf.math.sin(shear)\n    shear_matrix = tf.reshape( tf.concat([one,s2,zero, zero,c2,zero, zero,zero,one],axis=0),[3,3] )    \n    \n    # ZOOM MATRIX\n    zoom_matrix = tf.reshape( tf.concat([one/height_zoom,zero,zero, zero,one/width_zoom,zero, zero,zero,one],axis=0),[3,3] )\n    \n    # SHIFT MATRIX\n    shift_matrix = tf.reshape( tf.concat([one,zero,height_shift, zero,one,width_shift, zero,zero,one],axis=0),[3,3] )\n    \n    return K.dot(K.dot(rotation_matrix, shear_matrix), K.dot(zoom_matrix, shift_matrix))\n\n\n@tf.function\ndef transform(image):\n    # input image - is one image of size [dim,dim,3] not a batch of [b,dim,dim,3]\n    # output - image randomly rotated, sheared, zoomed, and shifted\n    DIM = IMAGE_SIZE[0]\n    XDIM = DIM%2 #fix for size 331\n    \n    # phase 1\n    if tf.random.uniform(shape=[], minval=0, maxval=2, dtype=tf.int32, seed=SEED) == 0: # 50% chance\n        rot = 17. * tf.random.normal([1],dtype='float32')\n    else:\n        rot = tf.constant([0],dtype='float32')\n    \n    if tf.random.uniform(shape=[], minval=0, maxval=2, dtype=tf.int32, seed=SEED) == 0: # 50% chance\n        shr = 5.5 * tf.random.normal([1],dtype='float32') \n    else:\n        shr = tf.constant([0],dtype='float32')\n    \n    if tf.random.uniform(shape=[], minval=0, maxval=3, dtype=tf.int32, seed=SEED) == 0: # 33% chance\n        h_zoom = tf.random.normal([1],dtype='float32')/8.5\n        if h_zoom > 0:\n            h_zoom = 1.0 + h_zoom * -1\n        else:\n            h_zoom = 1.0 + h_zoom\n    else:\n        h_zoom = tf.constant([1],dtype='float32')\n    \n    if tf.random.uniform(shape=[], minval=0, maxval=3, dtype=tf.int32, seed=SEED) == 0: # 33% chance\n        w_zoom = tf.random.normal([1],dtype='float32')/8.5\n        if w_zoom > 0:\n            w_zoom = 1.0 + w_zoom * -1\n        else:\n            w_zoom = 1.0 + w_zoom\n    else:\n        w_zoom = tf.constant([1],dtype='float32')\n    \n    if tf.random.uniform(shape=[], minval=0, maxval=3, dtype=tf.int32, seed=SEED) == 0: # 33% chance\n        h_shift = 18. * tf.random.normal([1],dtype='float32') \n    else:\n        h_shift = tf.constant([0],dtype='float32')\n    \n    if tf.random.uniform(shape=[], minval=0, maxval=3, dtype=tf.int32, seed=SEED) == 0: # 33% chance\n        w_shift = 18. * tf.random.normal([1],dtype='float32') \n    else:\n        w_shift = tf.constant([0],dtype='float32')\n\n    # phase 2\n#     if tf.random.uniform(shape=[], minval=0, maxval=10, dtype=tf.int32, seed=SEED) < 6: # 60% chance\n#         rot = 20. * tf.random.normal([1],dtype='float32')\n#     else:\n#         rot = tf.constant([0],dtype='float32')\n    \n#     if tf.random.uniform(shape=[], minval=0, maxval=10, dtype=tf.int32, seed=SEED) < 5: # 60% chance\n#         shr = 6 * tf.random.normal([1],dtype='float32') \n#     else:\n#         shr = tf.constant([0],dtype='float32')\n    \n#     if tf.random.uniform(shape=[], minval=0, maxval=10, dtype=tf.int32, seed=SEED) < 4: # 40% chance\n#         h_zoom = tf.random.normal([1],dtype='float32')/8\n#         if h_zoom > 0:\n#             h_zoom = 1.0 + h_zoom * -1\n#         else:\n#             h_zoom = 1.0 + h_zoom\n#     else:\n#         h_zoom = tf.constant([1],dtype='float32')\n    \n#     if tf.random.uniform(shape=[], minval=0, maxval=10, dtype=tf.int32, seed=SEED) < 4: # 40% chance\n#         w_zoom = tf.random.normal([1],dtype='float32')/8\n#         if w_zoom > 0:\n#             w_zoom = 1.0 + w_zoom * -1\n#         else:\n#             w_zoom = 1.0 + w_zoom\n#     else:\n#         w_zoom = tf.constant([1],dtype='float32')\n    \n#     if tf.random.uniform(shape=[], minval=0, maxval=10, dtype=tf.int32, seed=SEED) < 4: # 40% chance\n#         h_shift = 20. * tf.random.normal([1],dtype='float32') \n#     else:\n#         h_shift = tf.constant([0],dtype='float32')\n    \n#     if tf.random.uniform(shape=[], minval=0, maxval=10, dtype=tf.int32, seed=SEED) < 4: # 40% chance\n#         w_shift = 20. * tf.random.normal([1],dtype='float32') \n#     else:\n#         w_shift = tf.constant([0],dtype='float32')\n  \n    # GET TRANSFORMATION MATRIX\n    m = get_mat(rot,shr,h_zoom,w_zoom,h_shift,w_shift) \n\n    # LIST DESTINATION PIXEL INDICES\n    x = tf.repeat( tf.range(DIM//2,-DIM//2,-1), DIM )\n    y = tf.tile( tf.range(-DIM//2,DIM//2),[DIM] )\n    z = tf.ones([DIM*DIM],dtype='int32')\n    idx = tf.stack( [x,y,z] )\n    \n    # ROTATE DESTINATION PIXELS ONTO ORIGIN PIXELS\n    idx2 = K.dot(m,tf.cast(idx,dtype='float32'))\n    idx2 = K.cast(idx2,dtype='int32')\n    idx2 = K.clip(idx2,-DIM//2+XDIM+1,DIM//2)\n    \n    # FIND ORIGIN PIXEL VALUES           \n    idx3 = tf.stack( [DIM//2-idx2[0,], DIM//2-1+idx2[1,]] )\n    d = tf.gather_nd(image,tf.transpose(idx3))\n        \n    return tf.reshape(d,[DIM,DIM,3])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## GridMask","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"@tf.function\ndef transform_grid_mark(image, inv_mat, image_shape):\n    h, w, c = image_shape\n    \n    cx, cy = w//2, h//2\n\n    new_xs = tf.repeat( tf.range(-cx, cx, 1), h)\n    new_ys = tf.tile( tf.range(-cy, cy, 1), [w])\n    new_zs = tf.ones([h*w], dtype=tf.int32)\n\n    old_coords = tf.matmul(inv_mat, tf.cast(tf.stack([new_xs, new_ys, new_zs]), tf.float32))\n    old_coords_x, old_coords_y = tf.round(old_coords[0, :] + tf.cast(w, tf.float32)//2.), tf.round(old_coords[1, :] + tf.cast(h, tf.float32)//2.)\n    old_coords_x = tf.cast(old_coords_x, tf.int32)\n    old_coords_y = tf.cast(old_coords_y, tf.int32)    \n\n    clip_mask_x = tf.logical_or(old_coords_x<0, old_coords_x>w-1)\n    clip_mask_y = tf.logical_or(old_coords_y<0, old_coords_y>h-1)\n    clip_mask = tf.logical_or(clip_mask_x, clip_mask_y)\n\n    old_coords_x = tf.boolean_mask(old_coords_x, tf.logical_not(clip_mask))\n    old_coords_y = tf.boolean_mask(old_coords_y, tf.logical_not(clip_mask))\n    new_coords_x = tf.boolean_mask(new_xs+cx, tf.logical_not(clip_mask))\n    new_coords_y = tf.boolean_mask(new_ys+cy, tf.logical_not(clip_mask))\n\n    old_coords = tf.cast(tf.stack([old_coords_y, old_coords_x]), tf.int32)\n    new_coords = tf.cast(tf.stack([new_coords_y, new_coords_x]), tf.int64)\n    rotated_image_values = tf.gather_nd(image, tf.transpose(old_coords))\n    rotated_image_channel = list()\n    for i in range(c):\n        vals = rotated_image_values[:,i]\n        sparse_channel = tf.SparseTensor(tf.transpose(new_coords), vals, [h, w])\n        rotated_image_channel.append(tf.sparse.to_dense(sparse_channel, default_value=0, validate_indices=False))\n\n    return tf.transpose(tf.stack(rotated_image_channel), [1,2,0])\n\n\n@tf.function\ndef random_rotate(image, angle, image_shape):\n    def get_rotation_mat_inv(angle):\n          #transform to radian\n        angle = math.pi * angle / 180\n\n        cos_val = tf.math.cos(angle)\n        sin_val = tf.math.sin(angle)\n        one = tf.constant([1], tf.float32)\n        zero = tf.constant([0], tf.float32)\n\n        rot_mat_inv = tf.concat([cos_val, sin_val, zero,\n                                     -sin_val, cos_val, zero,\n                                     zero, zero, one], axis=0)\n        rot_mat_inv = tf.reshape(rot_mat_inv, [3,3])\n\n        return rot_mat_inv\n    angle = float(angle) * tf.random.normal([1],dtype='float32')\n    rot_mat_inv = get_rotation_mat_inv(angle)\n    return transform_grid_mark(image, rot_mat_inv, image_shape)\n\n\n@tf.function\ndef GridMask():\n    h = tf.constant(IMAGE_SIZE[0], dtype=tf.float32)\n    w = tf.constant(IMAGE_SIZE[1], dtype=tf.float32)\n    \n    image_height, image_width = (h, w)\n    # phase 1\n#     d1 = 84 # 100\n#     d2 = 168 # 160\n#     rotate_angle = 45 # 1\n#     ratio = 0.5\n    # phase 2\n    d1 = 105\n    d2 = 210\n    rotate_angle = tf.random.uniform(shape=[], minval=30, maxval=60, dtype=tf.int32)\n    ratio = 0.55\n\n    hh = tf.math.ceil(tf.math.sqrt(h*h+w*w))\n    hh = tf.cast(hh, tf.int32)\n    hh = hh+1 if hh%2==1 else hh\n    d = tf.random.uniform(shape=[], minval=d1, maxval=d2, dtype=tf.int32)\n    l = tf.cast(tf.cast(d,tf.float32)*ratio+0.5, tf.int32)\n\n    st_h = tf.random.uniform(shape=[], minval=0, maxval=d, dtype=tf.int32)\n    st_w = tf.random.uniform(shape=[], minval=0, maxval=d, dtype=tf.int32)\n\n    y_ranges = tf.range(-1 * d + st_h, -1 * d + st_h + l)\n    x_ranges = tf.range(-1 * d + st_w, -1 * d + st_w + l)\n\n    for i in range(0, hh//d+1):\n        s1 = i * d + st_h\n        s2 = i * d + st_w\n        y_ranges = tf.concat([y_ranges, tf.range(s1,s1+l)], axis=0)\n        x_ranges = tf.concat([x_ranges, tf.range(s2,s2+l)], axis=0)\n\n    x_clip_mask = tf.logical_or(x_ranges <0 , x_ranges > hh-1)\n    y_clip_mask = tf.logical_or(y_ranges <0 , y_ranges > hh-1)\n    clip_mask = tf.logical_or(x_clip_mask, y_clip_mask)\n\n    x_ranges = tf.boolean_mask(x_ranges, tf.logical_not(clip_mask))\n    y_ranges = tf.boolean_mask(y_ranges, tf.logical_not(clip_mask))\n\n    hh_ranges = tf.tile(tf.range(0,hh), [tf.cast(tf.reduce_sum(tf.ones_like(x_ranges)), tf.int32)])\n    x_ranges = tf.repeat(x_ranges, hh)\n    y_ranges = tf.repeat(y_ranges, hh)\n\n    y_hh_indices = tf.transpose(tf.stack([y_ranges, hh_ranges]))\n    x_hh_indices = tf.transpose(tf.stack([hh_ranges, x_ranges]))\n\n    y_mask_sparse = tf.SparseTensor(tf.cast(y_hh_indices, tf.int64),  tf.zeros_like(y_ranges), [hh, hh])\n    y_mask = tf.sparse.to_dense(y_mask_sparse, 1, False)\n\n    x_mask_sparse = tf.SparseTensor(tf.cast(x_hh_indices, tf.int64), tf.zeros_like(x_ranges), [hh, hh])\n    x_mask = tf.sparse.to_dense(x_mask_sparse, 1, False)\n\n    mask = tf.expand_dims( tf.clip_by_value(x_mask + y_mask, 0, 1), axis=-1)\n\n    mask = random_rotate(mask, rotate_angle, [hh, hh, 1])\n    mask = tf.image.crop_to_bounding_box(mask, (hh-tf.cast(h, tf.int32))//2, (hh-tf.cast(w, tf.int32))//2, tf.cast(image_height, tf.int32), tf.cast(image_width, tf.int32))\n\n    return mask\n\n\n@tf.function\ndef apply_grid_mask(image):\n    mask = GridMask()\n    mask = tf.concat([mask, mask, mask], axis=-1)\n\n    return image * tf.cast(mask, 'float32')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Augmentation function & tf.image functions (flip, brightness, contrast, saturation, hue)","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"@tf.function\ndef image_augmentation(iw, label):\n    global current_epoch\n    global chance\n    \n    image, words = iw\n    \n    # phase 1\n    if tf.random.uniform(shape=[], minval=0, maxval=11, dtype=tf.int32, seed=SEED) < chance:\n        if tf.random.uniform(shape=[], minval=0, maxval=2, dtype=tf.int32, seed=SEED) == 0:\n            image = tf.image.flip_left_right(image)\n        else:            \n            image = tf.image.flip_up_down(image)\n    if tf.random.uniform(shape=[], minval=0, maxval=11, dtype=tf.int32, seed=SEED) < chance:\n        image = tf.image.random_brightness(image, 0.1)\n    if tf.random.uniform(shape=[], minval=0, maxval=11, dtype=tf.int32, seed=SEED) < chance:\n        image = tf.image.random_contrast(image, 0.9, 1.1)\n    if tf.random.uniform(shape=[], minval=0, maxval=11, dtype=tf.int32, seed=SEED) < chance:\n        image = tf.image.random_saturation(image, 0.95, 1.05)\n    if tf.random.uniform(shape=[], minval=0, maxval=11, dtype=tf.int32, seed=SEED) < chance:\n        image = tf.image.random_hue(image, 0.05)\n        \n    # phase 2\n#     if tf.random.uniform(shape=[], minval=0, maxval=10, dtype=tf.int32, seed=SEED) < 6: # 60% chance\n#         image = tf.image.random_brightness(image, 0.15)\n#     if tf.random.uniform(shape=[], minval=0, maxval=10, dtype=tf.int32, seed=SEED) < 6: # 60% chance\n#         image = tf.image.random_contrast(image, 0.85, 1.15)\n#     if tf.random.uniform(shape=[], minval=0, maxval=2, dtype=tf.int32, seed=SEED) == 0: # 50% chance\n#         image = tf.image.random_saturation(image, 0.9, 1.1)\n#     if tf.random.uniform(shape=[], minval=0, maxval=2, dtype=tf.int32, seed=SEED) == 0: # 50% chance\n#         image = tf.image.random_hue(image, 0.05)\n\n    # phase 1\n    if tf.random.uniform(shape=[], minval=0, maxval=11, dtype=tf.int32, seed=SEED) < chance:\n        image = transform(image)\n\n    # phase 1\n    if tf.random.uniform(shape=[], minval=0, maxval=11, dtype=tf.int32, seed=SEED) < chance:\n        image = apply_grid_mask(image)\n\n    # phase 2\n#     if tf.random.uniform(shape=[], minval=0, maxval=11, dtype=tf.int32, seed=SEED) < chance:\n#         image = apply_grid_mask(image)\n\n    return ((image, words), label)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1eff3b87-cea0-4659-a2b5-a16b291c9629","_cell_guid":"376e507c-1201-4bc1-a90a-ad3a821d7bd8","trusted":true},"cell_type":"markdown","source":"# Show augmentated image","execution_count":null},{"metadata":{"_uuid":"242e542b-89e5-4c4d-a6b5-dc0bb8421695","_cell_guid":"48e61c4c-31ad-4979-a540-92dc306c636c","trusted":true},"cell_type":"code","source":"def show_augmented_image(same_image=True, pba=False):\n    row, col = 3, 5\n    if same_image and not pba:\n        all_elements = get_training_dataset(do_aug=False).unbatch()\n        one_element = tf.data.Dataset.from_tensors( next(iter(all_elements)) )\n        augmented_element = one_element.repeat().map(image_augmentation).batch(row*col)\n        for iw, label in augmented_element:\n            image, words = iw\n            plt.figure(figsize=(15,int(15*row/col)))\n            for j in range(row*col):\n                plt.subplot(row,col,j+1)\n                plt.axis('off')\n                plt.imshow(image[j,])\n            plt.suptitle(CLASSES[label[0]])\n            plt.show()\n            break\n    else:\n        all_elements = get_training_dataset(do_aug=True).unbatch()\n        augmented_element = all_elements.batch(row*col)\n\n        for iw, label in augmented_element:\n            image, words = iw\n            plt.figure(figsize=(15,int(15*row/col)))\n            for j in range(row*col):\n                plt.subplot(row,col,j+1)\n                plt.title(CLASSES[label[j]])\n                plt.axis('off')\n                plt.imshow(image[j,])\n            plt.show()\n            break","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7bd5915e-ee0e-4993-9662-3b50c7ec7911","_cell_guid":"de535aa0-e757-49ad-b4e0-4d31b48dec69","trusted":true},"cell_type":"code","source":"# run again to see different batch of image\nshow_augmented_image()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# run again to see different image\nshow_augmented_image(same_image=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a75f451d-4205-4cce-bd9c-13f0e55f8daf","_cell_guid":"1b4398f1-19d6-454d-a67a-839179a63d67","trusted":true},"cell_type":"markdown","source":"# Functions for model training","execution_count":null},{"metadata":{"_uuid":"9d6e4962-d6b1-4a1f-9ceb-5835b2d3a28f","_cell_guid":"91c40909-1c30-4b55-a5ab-7ffc789eab6f","trusted":true},"cell_type":"code","source":"from tensorflow.keras.models import Model, Sequential\nfrom tensorflow.keras.layers import Input, Lambda, Flatten, Dense, Dropout, AveragePooling2D, GlobalAveragePooling2D, SpatialDropout2D, BatchNormalization, Activation, Concatenate","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plt_lr(epoch_count):\n    if epoch_count > 50:\n        epoch_count = 50\n    \n    rng = [i for i in range(epoch_count)]\n\n    plt.figure()\n    y = [lrfn(x) for x in rng]\n    plt.title(f'Learning rate schedule: {y[0]} to {y[epoch_count-1]}')\n    plt.plot(rng, y)\n\ndef plt_acc(h):\n    plt.figure()\n    plt.plot(h.history[\"sparse_categorical_accuracy\"])\n    if 'val_sparse_categorical_accuracy' in h.history:\n        plt.plot(h.history[\"val_sparse_categorical_accuracy\"]) \n        plt.legend([\"training\",\"validation\"])       \n    else:\n        plt.legend([\"training\"])\n    plt.xlabel(\"epoch\")\n    plt.title(\"Sparse Categorical Accuracy\")\n    plt.show()\n\ndef plt_loss(h):\n    plt.figure()\n    plt.plot(h.history[\"loss\"])\n    if 'val_loss' in h.history:\n        plt.plot(h.history[\"val_loss\"]) \n        plt.legend([\"training\",\"validation\"])       \n    else:\n        plt.legend([\"training\"])\n    plt.legend([\"training\",\"validation\"])\n    plt.xlabel(\"epoch\")\n    plt.title(\"Loss\")\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class EpochCallback(tf.keras.callbacks.Callback):  \n    def on_epoch_begin(self, epoch, logs=None):\n        global current_epoch\n        global chance\n\n        current_epoch = epoch       \n        if current_epoch < 2:\n            chance = 0\n        elif current_epoch < 9:\n            chance = current_epoch - 1\n        else:\n            chance = 8\n        print(f'Epoch #{current_epoch} begin!')\n        print(datetime.now())\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0b0943e9-1066-438e-ad1a-ce2dc41fb4d4","_cell_guid":"838b477a-e39b-4403-b5bf-e4314f77d88a","trusted":true},"cell_type":"code","source":"es_val_acc = tf.keras.callbacks.EarlyStopping(\n    monitor='val_sparse_categorical_accuracy', min_delta=0.001, patience=5, verbose=1, mode='auto',\n    baseline=None, restore_best_weights=True\n)\n\nes_val_loss = tf.keras.callbacks.EarlyStopping(\n    monitor='val_loss', min_delta=0.001, patience=5, verbose=1, mode='auto',\n    baseline=None, restore_best_weights=True\n)\n\nes_acc = tf.keras.callbacks.EarlyStopping(\n    monitor='sparse_categorical_accuracy', min_delta=0.001, patience=5, verbose=1, mode='auto',\n    baseline=None, restore_best_weights=False\n)\n\nes_loss = tf.keras.callbacks.EarlyStopping(\n    monitor='loss', min_delta=0.001, patience=5, verbose=1, mode='auto',\n    baseline=None, restore_best_weights=False\n)\n\nepoch_cb = EpochCallback()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"57f6b8cc-ed53-4ca1-acef-a281f81e974d","_cell_guid":"7f497bc4-6949-431a-9fbf-e92a7a346ec5","trusted":true},"cell_type":"markdown","source":"# Create model","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"## EfficientNetB7 model\n\n| Layer     | Layer Type                   |\n| --------- | ---------------------------- |\n| 0         | input_1 (InputLayer)         |\n| 1         | stem_conv (Conv2D)           |\n| 2         | stem_bn (BatchNormalization) |\n| 3         | stem_activation (Activation) |\n| 4-49      | block1*                      |\n| 50 - 152  | block2*                      |\n| 153 - 255 | block3*                      |\n| 256 - 403 | block4*                      |\n| 404 - 551 | block5*                      |\n| 552 - 744 | block6*                      |\n| 745 - 802 | block7*                      |\n| 803       | top_conv (Conv2D)            |\n| 804       | top_bn (BatchNormalization)  |\n| 805       | top_activation (Activation)  |","execution_count":null},{"metadata":{"_uuid":"9962e8eb-1ccf-4879-977f-f684f1cd797e","_cell_guid":"c4c4b445-3b0a-4cf4-a272-0094c8698f01","trusted":true},"cell_type":"code","source":"with strategy.scope():\n    # phase 1\n    efn7 = efn.EfficientNetB7(weights='noisy-student', include_top=False, input_shape=(IMAGE_SIZE[0], IMAGE_SIZE[1], 3))\n    for layer in efn7.layers:\n        layer.trainable = True\n    efn0 = efn.EfficientNetB0(weights='noisy-student', include_top=False, input_shape=(IMAGE_SIZE[0], IMAGE_SIZE[1], 3))\n    for layer in efn0.layers:\n        layer.trainable = True\n        \n    # image input\n    image_input = Input((IMAGE_SIZE[0], IMAGE_SIZE[1], 3), name='image_input')\n    lambda_input = Lambda(lambda x:x)(image_input) # used when 1 input used by multiple model\n\n    model_image_1 = Sequential([\n        image_input,\n        efn7,\n        GlobalAveragePooling2D(name='efficientnet-b7_gap'),\n    ], name='b7-image')\n    model_image_2 = Sequential([\n        image_input,\n        efn0,\n        GlobalAveragePooling2D(name='efficientnet-b0_gap'),\n    ], name='b0-image')\n\n    model_words = Sequential([\n        Input((6633, ), name='mlp-words_input'),\n\n        Dense(331, name='mlp-words_dense_1'),\n        BatchNormalization(name='mlp-words_bn_1'),\n        Activation('relu', name='mlp-words_act_1'),\n\n        Dense(110, name='mlp-words_dense_2'),\n        BatchNormalization(name='mlp-words_bn_2'),\n        Activation('relu', name='mlp-words_act_2'),\n    ], name='mlp-words')\n\n    concatenate = Concatenate(name='concatenate')([model_image_1.output, model_image_2.output, model_words.output])\n    output = Dense(len(CLASSES), activation='softmax', name='output')(concatenate)\n\n    model = Model(inputs=[image_input, model_words.input], outputs=output)\n\n    # phase 2\n#     model = tf.keras.models.load_model('/kaggle/input/train-phase-1-085009/model.h5')\n#     model.load_weights('/kaggle/input/train-phase-1-085009/model_weights.h5')\n\n    model.compile(optimizer=tfa.optimizers.LAMB(0.01), loss='sparse_categorical_crossentropy', metrics=['sparse_categorical_accuracy'])\n\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f'Pre training time : {(datetime.now() - PRE_TRAINING_TIME_START).total_seconds()} seconds')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Training model","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"LR_START = 0.0005\nLR_MAX = 0.001\nLR_MIN = 0.0001\nLR_RAMPUP_EPOCHS = 2\nLR_SUSTAIN_EPOCHS = 0\nLR_EXP_DECAY = 0.77 #0.91\n\ndef lrfn(epoch):\n    if epoch < LR_RAMPUP_EPOCHS:\n        lr = (LR_MAX - LR_START) / LR_RAMPUP_EPOCHS * epoch + LR_START\n    elif epoch < LR_RAMPUP_EPOCHS + LR_SUSTAIN_EPOCHS:\n        lr = LR_MAX\n    else:\n        lr = (LR_MAX - LR_MIN) * LR_EXP_DECAY**(epoch - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS) + LR_MIN\n    return lr\nlr_schedule = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose = True)\n\nplt_lr(EPOCHS)\n# plt_lr(EPOCHS+EPOCHS)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.fit(\n    get_training_dataset(do_aug=DO_AUG), steps_per_epoch=STEPS_PER_EPOCH,\n    epochs=EPOCHS,\n    callbacks=[es_acc, epoch_cb, lr_schedule], verbose=1\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"h = model.history\nplt_acc(h)\nplt_loss(h)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"POST_TRAINING_TIME_START = datetime.now()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"231b8bbe-b450-4289-add7-94959a8dbc1a","_cell_guid":"ad997638-35d4-4eb7-a578-7551d026cde2","trusted":true},"cell_type":"markdown","source":"# Submit Result","execution_count":null},{"metadata":{"_uuid":"f0290e52-599b-46cc-9bce-52578c388915","_cell_guid":"b080a816-6493-4c78-acf8-44efb66fac86","trusted":true},"cell_type":"code","source":"test_ds = get_test_dataset(ordered=True) # since we are splitting the dataset and iterating separately on images and ids, order matters.\n\nprint('Computing predictions...')\ntest_images_ds = test_ds.map(lambda iw, filename: [iw])\nmodel_pred = model.predict(test_images_ds)\n\npredictions = np.argmax(model_pred, axis=-1)\nprint(predictions)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_ds = get_test_dataset(ordered=True) # since we are splitting the dataset and iterating separately on images and ids, order matters.\n\ntest_ids_ds = test_ds.map(lambda iw, filename: filename).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(predictions.shape[0]))).numpy().astype('U') # all in one batch\n\ndf_submission = pd.DataFrame({'filename': test_ids, 'category': predictions})\ndf_submission = df_submission.drop_duplicates()\ndf_submission['category'] = df_submission['category'].apply(lambda c: str(c).zfill(2))\ndf_submission","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Generating submission.csv file...')\ndf_submission.to_csv('submission.csv', index=False)\n!head submission.csv","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.save('model.h5')\nmodel.save_weights('model_weights.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f'Post training time : {(datetime.now() - POST_TRAINING_TIME_START).total_seconds()} seconds')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}