{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<br>\n<h2 style = \"font-size:50px; font-family:Monaco ; font-weight : normal; background-color: #f6f5f5 ; color : #031cfc; text-align: center; border-radius: 100px 100px;\">[Tensorflow] Creating TFRecords</h2>\n<br>\n\nFor more information on **TFRecord Creation**, refer to the [Quick Keras Recipe](https://keras.io/examples/keras_recipes/creating_tfrecords/) by [Dimitre Oliveira](https://www.linkedin.com/in/dimitre-oliveira-7a1a0113a/).","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-06T09:33:35.270208Z","iopub.execute_input":"2022-02-06T09:33:35.270574Z","iopub.status.idle":"2022-02-06T09:33:40.250562Z","shell.execute_reply.started":"2022-02-06T09:33:35.270462Z","shell.execute_reply":"2022-02-06T09:33:40.24973Z"}}},{"cell_type":"code","source":"import os\nimport wandb\nimport numpy as np\nimport pandas as pd\nfrom glob import glob\nimport tensorflow as tf\nfrom tqdm.notebook import tqdm\nimport matplotlib.pyplot as plt\nfrom kaggle_secrets import UserSecretsClient","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CONFIGS = {\n    \"data_dir\": \"../input/happy-whale-and-dolphin\",\n    \"tfrecord_dir\": \"happy_whale_tfrecords\",\n    \"target_size\": 512,\n    \"anit_aliasing\": True\n}\n\nif not os.path.exists(os.path.join(CONFIGS[\"tfrecord_dir\"], \"train\")):\n    os.makedirs(os.path.join(CONFIGS[\"tfrecord_dir\"], \"train\"))\n\nif not os.path.exists(os.path.join(CONFIGS[\"tfrecord_dir\"], \"test\")):\n    os.makedirs(os.path.join(CONFIGS[\"tfrecord_dir\"], \"test\"))","metadata":{"execution":{"iopub.status.busy":"2022-02-06T09:35:48.909292Z","iopub.execute_input":"2022-02-06T09:35:48.90962Z","iopub.status.idle":"2022-02-06T09:35:48.915336Z","shell.execute_reply.started":"2022-02-06T09:35:48.909585Z","shell.execute_reply":"2022-02-06T09:35:48.91451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tree {CONFIGS[\"tfrecord_dir\"]}","metadata":{"execution":{"iopub.status.busy":"2022-02-06T09:35:49.133898Z","iopub.execute_input":"2022-02-06T09:35:49.134442Z","iopub.status.idle":"2022-02-06T09:35:49.899736Z","shell.execute_reply.started":"2022-02-06T09:35:49.134402Z","shell.execute_reply":"2022-02-06T09:35:49.898755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv(os.path.join(CONFIGS[\"data_dir\"], \"train.csv\"))\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-06T09:35:50.144844Z","iopub.execute_input":"2022-02-06T09:35:50.14515Z","iopub.status.idle":"2022-02-06T09:35:50.213733Z","shell.execute_reply.started":"2022-02-06T09:35:50.145116Z","shell.execute_reply":"2022-02-06T09:35:50.212984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_labels = list(train_data.species.unique())\nCONFIGS[\"label_map\"] = {label: idx for idx, label in enumerate(unique_labels)}\nCONFIGS[\"shard_size\"] = 1024\n\n# train_data = train_data.head(1000) # For Testing\ntest_image_list = glob(os.path.join(CONFIGS[\"data_dir\"], \"test_images\", \"*\"))\n\nnum_train_samples = len(train_data)\nnum_test_samples = len(test_image_list)\n\nCONFIGS[\"num_train_tfrecords\"] = num_train_samples // CONFIGS[\"shard_size\"]\nif num_train_samples % CONFIGS[\"shard_size\"]:\n    CONFIGS[\"num_train_tfrecords\"] += 1  # add one record if there are any remaining samples\n\nCONFIGS[\"num_test_tfrecords\"] = num_train_samples // CONFIGS[\"shard_size\"]\nif num_train_samples % CONFIGS[\"shard_size\"]:\n    CONFIGS[\"num_test_tfrecords\"] += 1  # add one record if there are any remaining samples","metadata":{"execution":{"iopub.status.busy":"2022-02-06T09:35:52.069035Z","iopub.execute_input":"2022-02-06T09:35:52.069309Z","iopub.status.idle":"2022-02-06T09:35:52.148222Z","shell.execute_reply.started":"2022-02-06T09:35:52.069278Z","shell.execute_reply":"2022-02-06T09:35:52.147508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def image_feature(value):\n    \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n    return tf.train.Feature(\n        bytes_list=tf.train.BytesList(value=[tf.io.encode_jpeg(value).numpy()])\n    )\n\n\ndef bytes_feature(value):\n    \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n    return tf.train.Feature(bytes_list=tf.train.BytesList(value=[value.encode()]))\n\n\ndef int64_feature(value):\n    \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n    return tf.train.Feature(int64_list=tf.train.Int64List(value=[value]))\n\n\ndef create_example_train(image, species, individual_id):\n    feature = {\n        \"image\": image_feature(image),\n        \"species\": bytes_feature(species),\n        \"species_label\": int64_feature(CONFIGS[\"label_map\"][species]),\n        \"individual_id\": bytes_feature(individual_id)\n    }\n    return tf.train.Example(features=tf.train.Features(feature=feature))\n\n\ndef create_example_test(image):\n    return tf.train.Example(\n        features=tf.train.Features(feature={\n            \"image\": image_feature(image)\n        })\n    )\n\n\ndef get_samples(data_list, tfrec_num):\n    return data_list[\n        (tfrec_num * CONFIGS[\"shard_size\"]) : ((tfrec_num + 1) * CONFIGS[\"shard_size\"])\n    ]","metadata":{"execution":{"iopub.status.busy":"2022-02-06T09:35:53.138707Z","iopub.execute_input":"2022-02-06T09:35:53.138991Z","iopub.status.idle":"2022-02-06T09:35:53.14852Z","shell.execute_reply.started":"2022-02-06T09:35:53.138963Z","shell.execute_reply":"2022-02-06T09:35:53.147558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_list = list(train_data[\"image\"])\nspecies_list = list(train_data[\"species\"])\nindividual_id_list = list(train_data[\"individual_id\"])\nfor tfrec_num in range(CONFIGS[\"num_train_tfrecords\"]):\n    image_samples = get_samples(image_list, tfrec_num)\n    species_samples = get_samples(species_list, tfrec_num)\n    individual_id_samples = get_samples(individual_id_list, tfrec_num)\n    num_samples = len(image_samples)\n    tfrecord_path = os.path.join(\n        CONFIGS[\"tfrecord_dir\"], \"train\",\n        \"file_%.2i-%i.tfrec\" % (tfrec_num, num_samples)\n    )\n    print(f\"\\nWriting Data to {tfrecord_path}, tfrecord ({tfrec_num + 1}/{CONFIGS['num_train_tfrecords']})...\\n\")\n    with tf.io.TFRecordWriter(tfrecord_path) as writer:\n        for idx in tqdm(range(num_samples)):\n            image_path = os.path.join(\n                CONFIGS[\"data_dir\"], \"train_images\", image_samples[idx]\n            )\n            image = tf.io.decode_jpeg(tf.io.read_file(image_path))\n            image = tf.image.resize(\n                image, (CONFIGS[\"target_size\"], CONFIGS[\"target_size\"]),\n                antialias=CONFIGS[\"anit_aliasing\"]\n            )\n            image = tf.cast(image, dtype=tf.uint8)\n            species = species_list[idx]\n            individual_id = individual_id_list[idx]\n            example = create_example_train(image, species, individual_id)\n            writer.write(example.SerializeToString())","metadata":{"execution":{"iopub.status.busy":"2022-02-06T09:37:19.669118Z","iopub.execute_input":"2022-02-06T09:37:19.669693Z","iopub.status.idle":"2022-02-06T11:09:34.624073Z","shell.execute_reply.started":"2022-02-06T09:37:19.66965Z","shell.execute_reply":"2022-02-06T11:09:34.622128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for tfrec_num in range(CONFIGS[\"num_test_tfrecords\"]):\n    image_samples = get_samples(test_image_list, tfrec_num)\n    num_samples = len(image_samples)\n    tfrecord_path = os.path.join(\n        CONFIGS[\"tfrecord_dir\"], \"test\",\n        \"file_%.2i-%i.tfrec\" % (tfrec_num, num_samples)\n    )\n    print(f\"\\nWriting Data to {tfrecord_path}, tfrecord ({tfrec_num + 1}/{CONFIGS['num_test_tfrecords']})...\\n\")\n    with tf.io.TFRecordWriter(tfrecord_path) as writer:\n        for idx in tqdm(range(num_samples)):\n            image = tf.io.decode_jpeg(tf.io.read_file(image_samples[idx]))\n            image = tf.image.resize(\n                image, (CONFIGS[\"target_size\"], CONFIGS[\"target_size\"]),\n                antialias=CONFIGS[\"anit_aliasing\"]\n            )\n            image = tf.cast(image, dtype=tf.uint8)\n            example = create_example_test(image)\n            writer.write(example.SerializeToString())","metadata":{"execution":{"iopub.status.busy":"2022-02-06T07:58:10.521439Z","iopub.status.idle":"2022-02-06T07:58:10.522112Z","shell.execute_reply.started":"2022-02-06T07:58:10.521831Z","shell.execute_reply":"2022-02-06T07:58:10.521859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"user_secrets = UserSecretsClient()\nwandb_api_key = user_secrets.get_secret(\"wandb_api_key\")\nos.environ[\"WANDB_API_KEY\"] = wandb_api_key\n\nwith wandb.init(project=\"happy-whale\", entity=\"geekyrakshit\", config=CONFIGS):\n    artifact = wandb.Artifact('happy-whale-tfrecords', type='dataset')\n    artifact.add_dir(CONFIGS[\"tfrecord_dir\"])\n    wandb.log_artifact(artifact)","metadata":{"execution":{"iopub.status.busy":"2022-02-06T07:58:10.523689Z","iopub.status.idle":"2022-02-06T07:58:10.52448Z","shell.execute_reply.started":"2022-02-06T07:58:10.52427Z","shell.execute_reply":"2022-02-06T07:58:10.524295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}