{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfrom kaggle_datasets import KaggleDatasets\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# tensorflow stuff\nimport tensorflow as tf\nimport tensorflow.keras as keras\nimport tensorflow.keras.layers as layers\n\n# visualization\nimport matplotlib.pyplot as plt\nplt.style.use('ggplot')\n\nAUTO = tf.data.experimental.AUTOTUNE","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-10-31T18:01:56.901140Z","iopub.execute_input":"2021-10-31T18:01:56.901650Z","iopub.status.idle":"2021-10-31T18:02:02.866792Z","shell.execute_reply.started":"2021-10-31T18:01:56.901539Z","shell.execute_reply":"2021-10-31T18:02:02.865997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\n    tpu_strategy = tf.distribute.TPUStrategy(tpu)\n    ignore_order = tf.data.Options()\n    ignore_order.experimental_deterministic = False\nexcept ValueError:\n    strategy = tf.distribute.MirroredStrategy()\n    \nprint(\"Number of accelerators: \", tpu_strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2021-10-31T18:12:38.768523Z","iopub.execute_input":"2021-10-31T18:12:38.769377Z","iopub.status.idle":"2021-10-31T18:12:44.415439Z","shell.execute_reply.started":"2021-10-31T18:12:38.769325Z","shell.execute_reply":"2021-10-31T18:12:44.414606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GCS_PATH = KaggleDatasets().get_gcs_path()\nGCS_PATH","metadata":{"execution":{"iopub.status.busy":"2021-10-31T18:12:44.416634Z","iopub.execute_input":"2021-10-31T18:12:44.416907Z","iopub.status.idle":"2021-10-31T18:12:44.980846Z","shell.execute_reply.started":"2021-10-31T18:12:44.416882Z","shell.execute_reply":"2021-10-31T18:12:44.980187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!gsutil ls $GCS_PATH","metadata":{"execution":{"iopub.status.busy":"2021-10-31T18:12:44.982836Z","iopub.execute_input":"2021-10-31T18:12:44.983144Z","iopub.status.idle":"2021-10-31T18:12:49.959181Z","shell.execute_reply.started":"2021-10-31T18:12:44.983114Z","shell.execute_reply":"2021-10-31T18:12:49.957960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Main dir on GCS\nmain_dir = os.path.join(GCS_PATH, \"tfrecords-jpeg-224x224/\")\n\n# creating paths dictionary\npaths = dict()\npaths['train'], paths['val'], paths['test'] = None, None, None\n\nfor k, v in paths.items():\n    paths[k] = tf.io.gfile.glob(main_dir + k + '/*.tfrec')\n# paths","metadata":{"execution":{"iopub.status.busy":"2021-10-31T18:12:49.961421Z","iopub.execute_input":"2021-10-31T18:12:49.961977Z","iopub.status.idle":"2021-10-31T18:12:50.487884Z","shell.execute_reply.started":"2021-10-31T18:12:49.961926Z","shell.execute_reply":"2021-10-31T18:12:50.487076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Constants**","metadata":{}},{"cell_type":"code","source":"IMAGE_SHAPE = [224, 224, 3]\nBATCH_SIZE = 16 * tpu_strategy.num_replicas_in_sync\nNUM_CLASSES = 104\nTRAIN_SIZE = 12753\nVALIDATION_SIZE = 16 * 232\nTRAIN_STEPS = TRAIN_SIZE // BATCH_SIZE\nVALIDATION_STEPS = VALIDATION_SIZE // BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2021-10-31T18:20:48.766213Z","iopub.execute_input":"2021-10-31T18:20:48.766496Z","iopub.status.idle":"2021-10-31T18:20:48.771612Z","shell.execute_reply.started":"2021-10-31T18:20:48.766468Z","shell.execute_reply":"2021-10-31T18:20:48.770609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data Retrieval**","metadata":{}},{"cell_type":"code","source":"def augment_image(image, label):\n    aug_image = tf.image.random_brightness(image, 0.6)\n    aug_image = tf.image.random_flip_left_right(aug_image)\n    aug_image = tf.image.random_flip_up_down(aug_image)\n    return aug_image, label\n\ndef load_training_data(apply_augmentation=True):\n    dataset = load_dataset(paths['train'], labeled=True)\n    if apply_augmentation:\n        dataset = dataset = dataset.map(augment_image, num_parallel_calls=AUTO)\n    dataset = dataset.repeat()\n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    return dataset\n\ndef load_validation_data():\n    dataset = load_dataset(paths['val'], labeled=True)\n    dataset = dataset.repeat()\n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    return dataset\n\ndef load_test_data():\n    dataset = load_dataset(paths['test'], labeled=False)\n    dataset = dataset.batch(BATCH_SIZE)\n    \n    return dataset\n\ndef parse_image(image_bytes):\n    image = tf.image.decode_jpeg(image_bytes, channels=3)\n    image = tf.reshape(image, IMAGE_SHAPE)\n    \n    return image\n\ndef load_dataset(paths, labeled):\n    # read from files\n    dataset = tf.data.TFRecordDataset(paths, num_parallel_reads=AUTO)\n    \n    dataset = dataset.with_options(ignore_order)\n    # map tfrecord to normal data\n    dataset = dataset.map(\n        parse_labeled_data if labeled else parse_unlabeled_data)\n    \n    return dataset\n\n    \ndef parse_labeled_data(example):\n    IMAGE_MESSAGE = {\n    'image': tf.io.FixedLenFeature([], tf.string),\n    'class': tf.io.FixedLenFeature([], tf.int64),\n    }\n    image_data = tf.io.parse_single_example(example, IMAGE_MESSAGE) \n    image = parse_image(image_data['image'])\n    label = tf.cast(image_data['class'], tf.int32)\n    \n    return image, label\n\ndef parse_unlabeled_data(example):\n    IMAGE_MESSAGE = {\n    'image': tf.io.FixedLenFeature([], tf.string),\n    'id': tf.io.FixedLenFeature([], tf.int64), \n    }\n    image_data = tf.io.parse_single_example(example, IMAGE_MESSAGE)\n    image_id = tf.cast(image_data['id'], tf.int32)\n    \n    return image, image_id\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2021-10-31T18:25:27.873113Z","iopub.execute_input":"2021-10-31T18:25:27.873681Z","iopub.status.idle":"2021-10-31T18:25:27.890792Z","shell.execute_reply.started":"2021-10-31T18:25:27.873637Z","shell.execute_reply":"2021-10-31T18:25:27.889910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for (x,y) in load_training_data().take(1):\n#     plt.imshow(x[0])\n#     plt.imshow(tf.image.random_brightness(x[0], 0.6))\n#     break","metadata":{"execution":{"iopub.status.busy":"2021-10-31T18:25:14.643314Z","iopub.execute_input":"2021-10-31T18:25:14.643626Z","iopub.status.idle":"2021-10-31T18:25:16.920908Z","shell.execute_reply.started":"2021-10-31T18:25:14.643593Z","shell.execute_reply":"2021-10-31T18:25:16.920023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(load_training_data())","metadata":{"execution":{"iopub.status.busy":"2021-10-31T18:13:00.273284Z","iopub.execute_input":"2021-10-31T18:13:00.274217Z","iopub.status.idle":"2021-10-31T18:13:00.548337Z","shell.execute_reply.started":"2021-10-31T18:13:00.274161Z","shell.execute_reply":"2021-10-31T18:13:00.547357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data dump\nprint(\"Training data shapes:\")\nfor image, label in load_training_data().take(3):\n    print(image.numpy().shape, label.numpy().shape)\nprint(\"Training data label examples:\", label.numpy()[:15])","metadata":{"execution":{"iopub.status.busy":"2021-10-31T18:13:02.564823Z","iopub.execute_input":"2021-10-31T18:13:02.565148Z","iopub.status.idle":"2021-10-31T18:13:05.698041Z","shell.execute_reply.started":"2021-10-31T18:13:02.565117Z","shell.execute_reply":"2021-10-31T18:13:05.697007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# labels, counts = np.unique(train_y, return_counts=True)\n# max_count = max(counts)\n# normalized_counts = counts / max_count\n# my_cmap = plt.get_cmap('viridis_r')\n# colors = my_cmap(normalized_counts)\n# fig, ax = plt.subplots(figsize=(20,30))\n# count_graph = ax.barh(labels, counts, color=colors)\n# ax.set_title('Class Count Histogram')\n# ax.set_yticks(labels);","metadata":{"execution":{"iopub.status.busy":"2021-10-31T18:19:09.222926Z","iopub.execute_input":"2021-10-31T18:19:09.223628Z","iopub.status.idle":"2021-10-31T18:19:09.228489Z","shell.execute_reply.started":"2021-10-31T18:19:09.223585Z","shell.execute_reply":"2021-10-31T18:19:09.227606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# img = train_x[0]\n# plt.axis('off')\n# plt.imshow(img)\n# print(f'Label : {train_y[0]}')\n# print(f'Image shape : {img.shape}')","metadata":{"execution":{"iopub.status.busy":"2021-10-31T18:19:12.426782Z","iopub.execute_input":"2021-10-31T18:19:12.427080Z","iopub.status.idle":"2021-10-31T18:19:12.430769Z","shell.execute_reply.started":"2021-10-31T18:19:12.427049Z","shell.execute_reply":"2021-10-31T18:19:12.429963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nwith tpu_strategy.scope():\n    # defining the layers\n    dense = layers.Dense(NUM_CLASSES, activation='softmax', name='Last-Layer')\n    core = keras.applications.EfficientNetB6(weights='imagenet', include_top=False, pooling='avg')\n    preproc_img = layers.Lambda(lambda data: \n                                keras.applications.efficientnet.\n                                    preprocess_input(tf.cast(data, tf.float32)), input_shape=IMAGE_SHAPE)\n    # model process\n    i = layers.Input(shape=IMAGE_SHAPE)\n    x = preproc_img(i)\n    x = core(x)\n    x = dense(x)\n\n    # creating model\n    model = keras.Model(inputs=[i], outputs=[x])\n    \n    # create an optimizer\n    optimizer = keras.optimizers.Adam()\n    # create a loss function\n    loss_fn = keras.losses.SparseCategoricalCrossentropy()\n    # create metrics\n    metrics = [keras.metrics.SparseCategoricalAccuracy()]\n\n    model.compile(\n        optimizer=optimizer,\n        loss =loss_fn,\n        metrics=metrics,\n        steps_per_execution=8\n        )\n\n\n","metadata":{"execution":{"iopub.status.busy":"2021-10-31T18:43:11.994353Z","iopub.execute_input":"2021-10-31T18:43:11.994704Z","iopub.status.idle":"2021-10-31T18:43:49.656756Z","shell.execute_reply.started":"2021-10-31T18:43:11.994671Z","shell.execute_reply":"2021-10-31T18:43:49.655566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-10-31T18:43:49.658908Z","iopub.execute_input":"2021-10-31T18:43:49.659572Z","iopub.status.idle":"2021-10-31T18:43:49.727877Z","shell.execute_reply.started":"2021-10-31T18:43:49.659526Z","shell.execute_reply":"2021-10-31T18:43:49.726921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    load_training_data(),\n    validation_data=load_validation_data(),\n    epochs=30,\n    steps_per_epoch=TRAIN_STEPS,\n    validation_steps=VALIDATION_STEPS\n    )","metadata":{"execution":{"iopub.status.busy":"2021-10-31T18:43:49.729410Z","iopub.execute_input":"2021-10-31T18:43:49.729756Z","iopub.status.idle":"2021-10-31T18:57:18.406410Z","shell.execute_reply.started":"2021-10-31T18:43:49.729713Z","shell.execute_reply":"2021-10-31T18:57:18.405618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history[\"loss\"])\nplt.plot(history.history[\"val_loss\"])\nplt.legend(['train loss', 'val loss'], loc='upper right');","metadata":{"execution":{"iopub.status.busy":"2021-10-31T18:57:18.408512Z","iopub.execute_input":"2021-10-31T18:57:18.408883Z","iopub.status.idle":"2021-10-31T18:57:18.675214Z","shell.execute_reply.started":"2021-10-31T18:57:18.408852Z","shell.execute_reply":"2021-10-31T18:57:18.674569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history[\"sparse_categorical_accuracy\"])\nplt.plot(history.history[\"val_sparse_categorical_accuracy\"])\nplt.legend(['train acc', 'val acc'], loc='upper left');","metadata":{"execution":{"iopub.status.busy":"2021-10-31T18:57:18.676163Z","iopub.execute_input":"2021-10-31T18:57:18.677001Z","iopub.status.idle":"2021-10-31T18:57:18.933989Z","shell.execute_reply.started":"2021-10-31T18:57:18.676968Z","shell.execute_reply":"2021-10-31T18:57:18.933055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}