{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-25T20:36:52.415051Z","iopub.execute_input":"2023-10-25T20:36:52.415790Z","iopub.status.idle":"2023-10-25T20:36:52.443737Z","shell.execute_reply.started":"2023-10-25T20:36:52.415757Z","shell.execute_reply":"2023-10-25T20:36:52.442774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_hub as hub","metadata":{"execution":{"iopub.status.busy":"2023-10-25T20:36:52.445360Z","iopub.execute_input":"2023-10-25T20:36:52.445650Z","iopub.status.idle":"2023-10-25T20:36:52.449939Z","shell.execute_reply.started":"2023-10-25T20:36:52.445625Z","shell.execute_reply":"2023-10-25T20:36:52.448986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check if GPUs are available\ngpus = tf.config.experimental.list_physical_devices('GPU')\n\nif gpus:\n    try:\n        # Set GPU memory growth to True to allocate GPU memory dynamically\n        for gpu in gpus:\n            tf.config.experimental.set_memory_growth(gpu, True)\n        strategy = tf.distribute.MirroredStrategy()  # Use MirroredStrategy for multi-GPU training\n        print(\"Running on \", len(gpus), \" GPU(s)\")\n    except RuntimeError as e:\n        print(e)\n        strategy = tf.distribute.get_strategy()  # Default strategy if GPUs are not available\nelse:\n    strategy = tf.distribute.get_strategy()  # Default strategy if GPUs are not available\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-10-25T20:36:52.451061Z","iopub.execute_input":"2023-10-25T20:36:52.451356Z","iopub.status.idle":"2023-10-25T20:36:52.461910Z","shell.execute_reply.started":"2023-10-25T20:36:52.451332Z","shell.execute_reply":"2023-10-25T20:36:52.460886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = 30\nimage_size = 224\nbatch_size = 64\nsteps_per_epoch = 12753 / batch_size\nlr = 3e-5\nAuto = tf.data.experimental.AUTOTUNE","metadata":{"execution":{"iopub.status.busy":"2023-10-25T20:36:52.463855Z","iopub.execute_input":"2023-10-25T20:36:52.464148Z","iopub.status.idle":"2023-10-25T20:36:52.473398Z","shell.execute_reply.started":"2023-10-25T20:36:52.464123Z","shell.execute_reply":"2023-10-25T20:36:52.472484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32)/255\n    image = tf.reshape(image, [image_size, image_size, 3])\n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFRECORD_FORMAT = {\n        \"image\" : tf.io.FixedLenFeature([], tf.string),\n        \"class\" : tf.io.FixedLenFeature([], tf.int64)\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFRECORD_FORMAT)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFRECORD_FORMAT = {\n        \"image\" : tf.io.FixedLenFeature([], tf.string),\n        \"id\" : tf.io.FixedLenFeature([], tf.string)\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFRECORD_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    \n    if not ordered:\n        ignore_order.experimental_deterministic = False\n    \n    dataset = tf.data.TFRecordDataset(filenames)\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord)\n    \n    return dataset\n\ndef data_augumentation(image, label):\n    image_aug = tf.image.random_brightness(image, 0.3)\n    image_aug = tf.image.random_contrast(image_aug, 0.3, 0.7)\n    image_aug = tf.image.random_flip_left_right(image_aug)\n    image_aug = tf.image.random_flip_up_down(image_aug)\n    image_aug = tf.image.random_hue(image_aug, 0.3)\n    image_aug = tf.image.adjust_gamma(image_aug, 0.3)\n    image_aug = tf.image.random_saturation(image_aug, 0.3, 0.7)\n    return image_aug, label\n\ndef get_training_dataset():\n    dataset = load_dataset(tf.io.gfile.glob('/kaggle/input/tpu-getting-started/tfrecords-jpeg-224x224/train/*.tfrec'), labeled=True)\n    dataset = dataset.repeat() # the training dataset must repeat for several epochs\n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(batch_size)\n    return dataset\n\ndef get_training_dataset_aug():\n    dataset = load_dataset(tf.io.gfile.glob('/kaggle/input/tpu-getting-started/tfrecords-jpeg-224x224/train/*.tfrec'), labeled=True)\n    dataset = dataset.map(data_augumentation, num_parallel_calls=Auto)\n    dataset = dataset.repeat() # the training dataset must repeat for several epochs\n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(batch_size)\n    return dataset\n\ndef get_validation_dataset(ordered=False):\n    dataset = load_dataset(tf.io.gfile.glob('/kaggle/input/tpu-getting-started/tfrecords-jpeg-224x224/val/*.tfrec'), labeled=True, ordered=False)\n    dataset = dataset.batch(batch_size)\n    dataset = dataset.cache()\n    return dataset\n\ndef get_test_dataset(ordered=False):\n    dataset = load_dataset(tf.io.gfile.glob('/kaggle/input/tpu-getting-started/tfrecords-jpeg-224x224/test/*.tfrec'), labeled=False, ordered=ordered)\n    dataset = dataset.batch(batch_size)\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2023-10-25T20:36:52.474563Z","iopub.execute_input":"2023-10-25T20:36:52.474846Z","iopub.status.idle":"2023-10-25T20:36:52.493315Z","shell.execute_reply.started":"2023-10-25T20:36:52.474801Z","shell.execute_reply":"2023-10-25T20:36:52.492287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_dataset = get_training_dataset()\naug_training_dataset = get_training_dataset_aug()\ntrain_dataset_augmented = training_dataset.concatenate(aug_training_dataset)\nvalidation_dataset = get_validation_dataset()\ntest_dataset = get_test_dataset(ordered=True)","metadata":{"execution":{"iopub.status.busy":"2023-10-25T20:36:52.494551Z","iopub.execute_input":"2023-10-25T20:36:52.494943Z","iopub.status.idle":"2023-10-25T20:36:52.805334Z","shell.execute_reply.started":"2023-10-25T20:36:52.494910Z","shell.execute_reply":"2023-10-25T20:36:52.804494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def scheduler(epoch):\n    global lr\n    if epoch<10:\n        lr += 1e-5\n        return lr\n    else:\n        lr = lr * tf.math.exp(-0.1)\n        return lr\n\ncallback = tf.keras.callbacks.LearningRateScheduler(\n    scheduler,\n    verbose = 1\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-25T20:36:52.807955Z","iopub.execute_input":"2023-10-25T20:36:52.808643Z","iopub.status.idle":"2023-10-25T20:36:52.814136Z","shell.execute_reply.started":"2023-10-25T20:36:52.808612Z","shell.execute_reply":"2023-10-25T20:36:52.813118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from matplotlib import pyplot as plt\n\nrng = [i for i in range(epochs)]\ny = [scheduler(x) for x in rng]\nplt.plot(rng, y)\nprint(\"Learning rate schedule: {:.3g} to {:.3g} to {:.3g}\".format(y[0], max(y), y[-1]))","metadata":{"execution":{"iopub.status.busy":"2023-10-25T20:36:52.815497Z","iopub.execute_input":"2023-10-25T20:36:52.815861Z","iopub.status.idle":"2023-10-25T20:36:53.109523Z","shell.execute_reply.started":"2023-10-25T20:36:52.815826Z","shell.execute_reply":"2023-10-25T20:36:53.108423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_extractor = \"https://tfhub.dev/google/imagenet/mobilenet_v2_140_224/feature_vector/5\"\npretrained_model = hub.KerasLayer(\n    feature_extractor,\n    input_shape = (image_size, image_size, 3),\n    trainable = True\n)\n\nmodel = tf.keras.Sequential([\n    pretrained_model,\n    tf.keras.layers.Dense(512, activation='relu'),\n    tf.keras.layers.Dense(104, activation='softmax')\n])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-10-25T20:36:53.110904Z","iopub.execute_input":"2023-10-25T20:36:53.111232Z","iopub.status.idle":"2023-10-25T20:36:55.138421Z","shell.execute_reply.started":"2023-10-25T20:36:53.111203Z","shell.execute_reply":"2023-10-25T20:36:55.137180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(\n    optimizer='adam',\n    loss = 'sparse_categorical_crossentropy',\n    metrics = ['sparse_categorical_accuracy']\n)\n\nhistory = model.fit(\n    train_dataset_augmented,\n    steps_per_epoch = steps_per_epoch,\n    callbacks = [callback],\n    epochs = epochs,\n    validation_data = validation_dataset\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-25T20:36:55.139694Z","iopub.execute_input":"2023-10-25T20:36:55.139985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nacc = history.history['sparse_categorical_accuracy']\nval_acc = history.history['val_sparse_categorical_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs = range(len(acc))\n\nplt.plot(epochs, acc, 'r', label='Training accuracy')\nplt.plot(epochs, val_acc, 'b', label='Validation accuracy')\nplt.title('Training and validation accuracy')\nplt.legend(loc=0)\nplt.figure()\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_images_ds = test_dataset.map(lambda image, idnum: image)\nprobabilities = model.predict(test_images_ds)\npredictions = np.argmax(probabilities, axis=-1)\nprint(predictions)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ids_ds = test_dataset.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(7382))).numpy().astype('U') # all in one batch\nnp.savetxt('submission.csv', np.rec.fromarrays([test_ids, predictions]), fmt=['%s', '%d'], delimiter=',', header='id,label', comments='')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}