{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import tensorflow as tf\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport numpy as np\nimport os","metadata":{"execution":{"iopub.status.busy":"2021-07-07T03:09:11.607992Z","iopub.execute_input":"2021-07-07T03:09:11.608641Z","iopub.status.idle":"2021-07-07T03:09:16.622687Z","shell.execute_reply.started":"2021-07-07T03:09:11.608529Z","shell.execute_reply":"2021-07-07T03:09:16.621825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"basedir = \"../input/tpu-getting-started\"\ntfrecordsdir = os.path.join(basedir, \"tfrecords-jpeg-224x224\")\ntraindir = os.path.join(tfrecordsdir, \"train\")\ntestdir = os.path.join(tfrecordsdir, \"test\")\nvaldir = os.path.join(tfrecordsdir, \"val\")\nsubmission_file = os.path.join(basedir, \"sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-07-07T03:09:16.624023Z","iopub.execute_input":"2021-07-07T03:09:16.624358Z","iopub.status.idle":"2021-07-07T03:09:16.629881Z","shell.execute_reply.started":"2021-07-07T03:09:16.624313Z","shell.execute_reply":"2021-07-07T03:09:16.627935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_SIZE = (224, 224)\nIMAGE_SHAPE = IMAGE_SIZE + (3, )\nBATCH_SIZE = 32\nEPOCHS_INIT = 10\nEPOCHS_FINE = 10","metadata":{"execution":{"iopub.status.busy":"2021-07-07T03:09:16.631722Z","iopub.execute_input":"2021-07-07T03:09:16.632055Z","iopub.status.idle":"2021-07-07T03:09:16.646282Z","shell.execute_reply.started":"2021-07-07T03:09:16.632021Z","shell.execute_reply":"2021-07-07T03:09:16.645575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_tfrecord_ds(dir):\n    filenames = tf.io.gfile.glob(os.path.join(dir, \"*\")) \n    return tf.data.TFRecordDataset(filenames, num_parallel_reads=tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2021-07-07T03:09:16.647884Z","iopub.execute_input":"2021-07-07T03:09:16.648413Z","iopub.status.idle":"2021-07-07T03:09:16.657015Z","shell.execute_reply.started":"2021-07-07T03:09:16.648377Z","shell.execute_reply":"2021-07-07T03:09:16.656286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_feature_description_train = {\n    'class': tf.io.FixedLenFeature([], tf.int64),\n    'image': tf.io.FixedLenFeature([], tf.string),\n}\n\nimage_feature_description_test = {\n    'id': tf.io.FixedLenFeature([], tf.string),\n    'image': tf.io.FixedLenFeature([], tf.string),\n}\n\ndef parse_image_train(proto):\n    example = tf.io.parse_single_example(proto, image_feature_description_train)\n    image = tf.image.decode_jpeg(example[\"image\"], channels=3)\n    label = example[\"class\"]\n    return image, label\n\ndef parse_image_test(proto):\n    example = tf.io.parse_single_example(proto, image_feature_description_test)\n    image = tf.image.decode_jpeg(example[\"image\"], channels=3)\n    return image, example[\"id\"]","metadata":{"execution":{"iopub.status.busy":"2021-07-07T03:09:16.658234Z","iopub.execute_input":"2021-07-07T03:09:16.658829Z","iopub.status.idle":"2021-07-07T03:09:16.668028Z","shell.execute_reply.started":"2021-07-07T03:09:16.658794Z","shell.execute_reply":"2021-07-07T03:09:16.667272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_train = get_tfrecord_ds(traindir).map(parse_image_train)\nds_val = get_tfrecord_ds(valdir).map(parse_image_train)\nds_test = get_tfrecord_ds(testdir).map(parse_image_test)\n\nds_train = ds_train.cache().shuffle(1000).batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\nds_val = ds_val.batch(BATCH_SIZE).cache().prefetch(tf.data.AUTOTUNE)\nds_test = ds_test.batch(BATCH_SIZE).cache().prefetch(tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2021-07-07T03:09:16.669244Z","iopub.execute_input":"2021-07-07T03:09:16.669676Z","iopub.status.idle":"2021-07-07T03:09:18.569722Z","shell.execute_reply.started":"2021-07-07T03:09:16.669641Z","shell.execute_reply":"2021-07-07T03:09:18.568902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 10))\nfor ds in ds_train.take(1):\n    for i in range(9):\n        plt.subplot(3, 3, i + 1)\n        plt.axis(\"off\")\n        plt.imshow(ds[0][i])\n        plt.title(ds[1][i].numpy())\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-07T03:09:18.574025Z","iopub.execute_input":"2021-07-07T03:09:18.576017Z","iopub.status.idle":"2021-07-07T03:09:21.199994Z","shell.execute_reply.started":"2021-07-07T03:09:18.575976Z","shell.execute_reply":"2021-07-07T03:09:21.199237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_augmentation = tf.keras.Sequential([\n    tf.keras.layers.experimental.preprocessing.RandomFlip('horizontal'),\n    tf.keras.layers.experimental.preprocessing.RandomRotation(0.2),\n    tf.keras.layers.experimental.preprocessing.RandomZoom(0.2),\n])","metadata":{"execution":{"iopub.status.busy":"2021-07-07T03:09:21.201026Z","iopub.execute_input":"2021-07-07T03:09:21.201340Z","iopub.status.idle":"2021-07-07T03:09:21.230202Z","shell.execute_reply.started":"2021-07-07T03:09:21.201293Z","shell.execute_reply":"2021-07-07T03:09:21.229486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"plt.figure(figsize=(10, 10))\nfor ds in ds_train.take(1):\n    plt.subplot(3, 3, 1)\n    plt.axis(\"off\")\n    plt.imshow(ds[0][0])\n    for i in range(2, 10):\n        plt.subplot(3, 3, i)\n        plt.axis(\"off\")\n        img = tf.expand_dims(ds[0][0], 0)\n        img_aug = data_augmentation(img)\n        plt.imshow(img_aug[0])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-02T09:07:47.502302Z","iopub.execute_input":"2021-07-02T09:07:47.502675Z","iopub.status.idle":"2021-07-02T09:07:49.175308Z","shell.execute_reply.started":"2021-07-02T09:07:47.502635Z","shell.execute_reply":"2021-07-02T09:07:49.17438Z"}}},{"cell_type":"code","source":"preprocess_input = tf.keras.applications.xception.preprocess_input\nbase_model = tf.keras.applications.Xception(\n    input_shape=IMAGE_SHAPE,\n    include_top=False,\n    weights='imagenet'\n)\nbase_model.trainable = False","metadata":{"execution":{"iopub.status.busy":"2021-07-07T03:09:21.233002Z","iopub.execute_input":"2021-07-07T03:09:21.233278Z","iopub.status.idle":"2021-07-07T03:09:23.184692Z","shell.execute_reply.started":"2021-07-07T03:09:21.233250Z","shell.execute_reply":"2021-07-07T03:09:23.183699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs = tf.keras.Input(shape=IMAGE_SHAPE)\nx = data_augmentation(inputs)\nx = preprocess_input(x)\nx = base_model(x, training=False)\nx = tf.keras.layers.GlobalAveragePooling2D()(x)\nx = tf.keras.layers.Dropout(0.2)(x)\noutputs = tf.keras.layers.Dense(104)(x)\nmodel = tf.keras.Model(inputs, outputs)\n\nbase_learning_rate = 3e-4\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=base_learning_rate),\n    loss=tf.keras.losses.SparseCategoricalCrossentropy(from_logits=True),\n    metrics=['accuracy'],\n)","metadata":{"execution":{"iopub.status.busy":"2021-07-07T03:09:23.186428Z","iopub.execute_input":"2021-07-07T03:09:23.186765Z","iopub.status.idle":"2021-07-07T03:09:23.684784Z","shell.execute_reply.started":"2021-07-07T03:09:23.186730Z","shell.execute_reply":"2021-07-07T03:09:23.683971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    ds_train, \n    epochs=EPOCHS_INIT, \n    validation_data=ds_val,\n)","metadata":{"execution":{"iopub.status.busy":"2021-07-07T03:09:23.686044Z","iopub.execute_input":"2021-07-07T03:09:23.686379Z","iopub.status.idle":"2021-07-07T03:15:51.334903Z","shell.execute_reply.started":"2021-07-07T03:09:23.686344Z","shell.execute_reply":"2021-07-07T03:15:51.334010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\n\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs_range = range(EPOCHS_INIT)\n\nplt.figure(figsize=(8, 8))\nplt.subplot(1, 2, 1)\nplt.plot(epochs_range, acc, label='Training Accuracy')\nplt.plot(epochs_range, val_acc, label='Validation Accuracy')\nplt.legend(loc='lower right')\nplt.title('Training and Validation Accuracy')\n\nplt.subplot(1, 2, 2)\nplt.plot(epochs_range, loss, label='Training Loss')\nplt.plot(epochs_range, val_loss, label='Validation Loss')\nplt.legend(loc='upper right')\nplt.title('Training and Validation Loss')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-07T03:15:51.336564Z","iopub.execute_input":"2021-07-07T03:15:51.336918Z","iopub.status.idle":"2021-07-07T03:15:51.583860Z","shell.execute_reply.started":"2021-07-07T03:15:51.336880Z","shell.execute_reply":"2021-07-07T03:15:51.583099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model.trainable = True\nfine_tune_at = -30\nfor layer in base_model.layers[:fine_tune_at]:\n    layer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2021-07-07T03:15:51.585074Z","iopub.execute_input":"2021-07-07T03:15:51.585413Z","iopub.status.idle":"2021-07-07T03:15:51.597994Z","shell.execute_reply.started":"2021-07-07T03:15:51.585377Z","shell.execute_reply":"2021-07-07T03:15:51.597076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(\n    optimizer=tf.keras.optimizers.RMSprop(learning_rate=base_learning_rate/10),\n    loss=tf.keras.losses.SparseCategoricalCrossentropy(from_logits=True),\n    metrics=['accuracy'],\n)","metadata":{"execution":{"iopub.status.busy":"2021-07-07T03:15:51.599416Z","iopub.execute_input":"2021-07-07T03:15:51.599803Z","iopub.status.idle":"2021-07-07T03:15:51.618965Z","shell.execute_reply.started":"2021-07-07T03:15:51.599767Z","shell.execute_reply":"2021-07-07T03:15:51.618181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_fine = model.fit(\n    ds_train,\n    epochs=EPOCHS_INIT+EPOCHS_FINE,\n    initial_epoch=EPOCHS_INIT,\n    validation_data=ds_val,\n)","metadata":{"execution":{"iopub.status.busy":"2021-07-07T03:15:51.621809Z","iopub.execute_input":"2021-07-07T03:15:51.622074Z","iopub.status.idle":"2021-07-07T03:24:41.837344Z","shell.execute_reply.started":"2021-07-07T03:15:51.622050Z","shell.execute_reply":"2021-07-07T03:24:41.836537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc += history_fine.history['accuracy']\nval_acc += history_fine.history['val_accuracy']\n\nloss += history_fine.history['loss']\nval_loss += history_fine.history['val_loss']","metadata":{"execution":{"iopub.status.busy":"2021-07-07T03:24:41.840109Z","iopub.execute_input":"2021-07-07T03:24:41.840374Z","iopub.status.idle":"2021-07-07T03:24:41.846683Z","shell.execute_reply.started":"2021-07-07T03:24:41.840348Z","shell.execute_reply":"2021-07-07T03:24:41.845950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 8))\nplt.subplot(2, 1, 1)\nplt.plot(acc, label='Training Accuracy')\nplt.plot(val_acc, label='Validation Accuracy')\nplt.plot([EPOCHS_INIT-1,EPOCHS_INIT-1], plt.ylim(), label='Start Fine Tuning')\nplt.legend(loc='lower right')\nplt.title('Training and Validation Accuracy')\n\nplt.subplot(2, 1, 2)\nplt.plot(loss, label='Training Loss')\nplt.plot(val_loss, label='Validation Loss')\nplt.plot([EPOCHS_INIT-1,EPOCHS_INIT-1], plt.ylim(), label='Start Fine Tuning')\nplt.legend(loc='upper right')\nplt.title('Training and Validation Loss')\nplt.xlabel('epoch')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-07T03:24:41.849437Z","iopub.execute_input":"2021-07-07T03:24:41.849690Z","iopub.status.idle":"2021-07-07T03:24:42.137685Z","shell.execute_reply.started":"2021-07-07T03:24:41.849666Z","shell.execute_reply":"2021-07-07T03:24:42.136906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save(\"model\")","metadata":{"execution":{"iopub.status.busy":"2021-07-07T03:24:42.138913Z","iopub.execute_input":"2021-07-07T03:24:42.139237Z","iopub.status.idle":"2021-07-07T03:25:03.519267Z","shell.execute_reply.started":"2021-07-07T03:24:42.139203Z","shell.execute_reply":"2021-07-07T03:25:03.518384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(ds_test)\npred_label = tf.math.argmax(pred, 1)","metadata":{"execution":{"iopub.status.busy":"2021-07-07T03:25:03.554377Z","iopub.execute_input":"2021-07-07T03:25:03.554685Z","iopub.status.idle":"2021-07-07T03:25:20.576891Z","shell.execute_reply.started":"2021-07-07T03:25:03.554653Z","shell.execute_reply":"2021-07-07T03:25:20.575969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_test_id = ds_test.map(lambda image, iid: iid).unbatch()\nids = [str(x, \"utf-8\") for x in ds_test_id.as_numpy_iterator()]","metadata":{"execution":{"iopub.status.busy":"2021-07-07T03:25:20.578107Z","iopub.execute_input":"2021-07-07T03:25:20.578468Z","iopub.status.idle":"2021-07-07T03:25:21.077447Z","shell.execute_reply.started":"2021-07-07T03:25:20.578424Z","shell.execute_reply":"2021-07-07T03:25:21.076537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(submission_file)\ndf[\"label\"] = pred_label\ndf[\"id\"] = ids\ndf.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2021-07-07T03:29:04.792200Z","iopub.execute_input":"2021-07-07T03:29:04.792572Z","iopub.status.idle":"2021-07-07T03:29:04.824892Z","shell.execute_reply.started":"2021-07-07T03:29:04.792540Z","shell.execute_reply":"2021-07-07T03:29:04.824157Z"},"trusted":true},"execution_count":null,"outputs":[]}]}