{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport numpy as np\nimport tensorflow as tf\nimport tensorflow_hub as hub\nimport matplotlib.pyplot as plt\nfrom kaggle_datasets import KaggleDatasets","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tf.test.is_gpu_available()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()  # TPU detection. No parameters necessary if TPU_NAME environment variable is set. On Kaggle this is always the case.\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy() # default distribution strategy in Tensorflow. Works on CPU and single GPU.\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Reading TFRecord Files","execution_count":null},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"GCS_DS_PATH = KaggleDatasets().get_gcs_path()\ndataset_base_path = GCS_DS_PATH+\"/tfrecords-jpeg-224x224\"\ndef get_records(split_name):\n    filenames = tf.io.gfile.glob(dataset_base_path+\"/\"+ split_name+\"/\"+ \"*.tfrec\")\n    return tf.data.TFRecordDataset(filenames)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_records = get_records(\"train\")\nval_records = get_records(\"val\")\n\noptions = tf.data.Options()\noptions.experimental_deterministic = False\n\ntest_records = get_records(\"test\").with_options(options)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_feature_description = {\n    'id': tf.io.FixedLenFeature([], tf.string),\n    'class': tf.io.FixedLenFeature([], tf.int64, default_value=0),\n    'image': tf.io.FixedLenFeature([], tf.string),\n}\n\ndef decode_image(image):\n    image = tf.io.decode_image(image)\n    image = tf.cast(image, tf.float32)\n    image = tf.reshape(image, (image_size, image_size, 3))\n    image /= 255\n    return image\n\ndef parse_image_function(example_proto):\n    example = tf.io.parse_single_example(example_proto, image_feature_description)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'], tf.int64)\n    id = example['id']\n    return image, label, id\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_size = 224\n\ntraining_size = sum(1 for record in train_records.map(parse_image_function))\nprint(\"Training Size:\", training_size)\n\ntrain_records\n\nfor image, label, id in train_records.map(parse_image_function):\n    plt.imshow(image)\n    plt.title(label.numpy())\n    break","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Input Pipeline","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"batch_size = 32\n\ndef label_data_map(image, label, id):\n    return image, label\n\ndef unlabel_data_map(image, label, id):\n    return image\n\ntraining_batch = train_records.map(parse_image_function).shuffle(training_size//4).map(label_data_map).batch(batch_size).prefetch(1)\nvalidation_batch = val_records.map(parse_image_function).map(label_data_map).batch(batch_size).prefetch(1)\ntest_batch = test_records.map(parse_image_function).map(unlabel_data_map).batch(batch_size).prefetch(1)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Model","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"num_classes = 104\n\nwith strategy.scope():\n    feature_extractor = tf.keras.applications.MobileNetV2(input_shape=(image_size, image_size, 3), weights='imagenet')\n    feature_extractor.trainable = False\n    \n    model = tf.keras.Sequential([\n        feature_extractor,\n        tf.keras.layers.Dense(128, activation='relu'),\n        tf.keras.layers.Dense(num_classes, activation='softmax')\n    ])\n    \n    model.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\nmodel.summary()\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"EPOCHS = 50\n\nhistory = model.fit(training_batch, epochs=EPOCHS, validation_data=validation_batch)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Validation","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"model.evaluate(validation_batch)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Prediction","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"ids = []\nfor img, label, id in test_records.map(parse_image_function):\n    ids.append(id.numpy().decode(\"utf-8\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"probs = model.predict(test_batch)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = np.argmax(probs, axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.savetxt('submission.csv', np.rec.fromarrays([ids, preds]), fmt=['%s', '%d'], header='id,label', delimiter=',', comments='')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}