{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":117682,"databundleVersionId":14443416,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# About\n\n- Convert `.tif` format to **TFRecord** format for faster data loading.\n- Create and upload each shared `.tfrec` file to Kaggle dataset.\n\n# Issue\n\n- Kaggle output disk space is limited. Which is why we remove shared `.tfrec` file after pushing to Kaggle dataset.\n- Though it works but, each push creates a new version.\n\n# Workaround\n\n- Open kaggle notebook to collab.\n- Upload whole dataset at once.\n- Remove raw `.tif` file after being processed.\n\n# Final TFRecord Dataset\n\n- [Vesuvius Challenge - Surface Detection [TFRecord]](https://www.kaggle.com/datasets/ipythonx/vesuvius-tfrecords/data)","metadata":{}},{"cell_type":"code","source":"!pip install imagecodecs -q","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport kagglehub\n\nimport numpy as np\nimport pandas as pd\nimport tifffile\nfrom tqdm import tqdm\nfrom matplotlib import pyplot as plt\n\nimport tensorflow as tf\ntf.__version__","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"root_dir = \"/kaggle/input/vesuvius-challenge-surface-detection\"\nimages_dir = f\"{root_dir}/train_images\"\nlabels_dir = f\"{root_dir}/train_labels\"\nall_image_files = sorted(tf.io.gfile.glob(os.path.join(images_dir, \"*.tif\")))\nall_label_files = sorted(tf.io.gfile.glob(os.path.join(labels_dir, \"*.tif\")))\n\nlen(all_image_files), len(all_label_files)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## TFRecord Encode","metadata":{}},{"cell_type":"code","source":"def serialize_example(image_path, label_path):\n    image = tifffile.imread(image_path).astype(np.uint8)\n    label = tifffile.imread(label_path).astype(np.uint8)\n\n    image_shape = np.array(image.shape, dtype=np.int64)\n    label_shape = np.array(label.shape, dtype=np.int64)\n    \n    image = image.tobytes()\n    label = label.tobytes()\n    \n    feature = {\n        \"image\": tf.train.Feature(\n            bytes_list=tf.train.BytesList(value=[image])\n        ),\n        \"label\": tf.train.Feature(\n            bytes_list=tf.train.BytesList(value=[label])\n        ),\n        \"image_shape\": tf.train.Feature(\n            int64_list=tf.train.Int64List(value=image_shape)\n        ),\n        \"label_shape\": tf.train.Feature(\n            int64_list=tf.train.Int64List(value=label_shape)\n        ),\n    }\n    example = tf.train.Example(features=tf.train.Features(feature=feature))\n    return example.SerializeToString()","metadata":{"trusted":true,"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"handle = \"ipythonx/vesuvius-tfrecords\"\noutput_dir = \"tfrecords\"\nos.makedirs(output_dir, exist_ok=True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_and_upload_tfrecords(dataset, shard_size=100, set_name=\"train\"):\n    num_shards = (len(dataset) + shard_size - 1) // shard_size\n    \n    for shard_idx in range(num_shards):\n        print(f\"\\nCreating shard {shard_idx}/{num_shards-1}\")\n        shard_path = os.path.join(output_dir, f\"{set_name}_shard_{shard_idx}.tfrec\")\n        start_idx = shard_idx * shard_size\n        end_idx = min(start_idx + shard_size, len(dataset))\n\n        with tf.io.TFRecordWriter(shard_path) as writer:\n            for i in tqdm(range(start_idx, end_idx)):\n                sample = dataset[i]\n                tf_example = serialize_example(sample[\"image\"], sample[\"label\"])\n                writer.write(tf_example)\n        print(f\"Local TFRecord: {shard_path}\")\n        \n        print(\"Uploading to Kaggle via kagglehub…\")\n        kagglehub.dataset_upload(\n            handle, \n            output_dir,\n            version_notes=f\"Added {set_name} shard {shard_idx}\"\n        )\n        print(\"Upload complete.\")\n\n        if shard_idx < num_shards - 1:\n            os.remove(shard_path)\n            print(\"Deleted local shard to free space.\")\n        else:\n            print(\"Keeping last shard for testing.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset = [\n    {\"image\": img, \"label\": lbl}\n    for img, lbl in zip(all_image_files, all_label_files)\n]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"create_and_upload_tfrecords(dataset, shard_size=5, set_name='training')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## TFRecord Decord","metadata":{}},{"cell_type":"code","source":"def parse_tfrecord_fn(example_proto):\n    feature_description = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"label\": tf.io.FixedLenFeature([], tf.string),\n        \"image_shape\": tf.io.FixedLenFeature([3], tf.int64),\n        \"label_shape\": tf.io.FixedLenFeature([3], tf.int64),\n    }\n    example = tf.io.parse_single_example(\n        example_proto, feature_description\n    )\n    \n    # Decode image and label data\n    image = tf.io.decode_raw(example[\"image\"], tf.uint8)\n    label = tf.io.decode_raw(example[\"label\"], tf.uint8)\n    \n    # Reshape to original dimensions\n    image = tf.reshape(image, example[\"image_shape\"])\n    label = tf.reshape(label, example[\"label_shape\"])\n    \n    return image, label","metadata":{"trusted":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Quick Look**","metadata":{}},{"cell_type":"code","source":"tfrecord_path = \"tfrecords/training_shard_0.tfrec\"\ndataset = tf.data.TFRecordDataset(tfrecord_path)\ndataset = dataset.map(parse_tfrecord_fn)\n\nfor image, label in dataset:\n    print(f\"Image shape {image.shape} and type {image.dtype}:\")\n    print(f\"Label shape {label.shape} and type {label.dtype}:\")\n    break","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## TFRecord DataLoader","metadata":{}},{"cell_type":"code","source":"def parse_tfrecord_fn(example):\n    feature_description = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"label\": tf.io.FixedLenFeature([], tf.string),\n        \"image_shape\": tf.io.FixedLenFeature([3], tf.int64),\n        \"label_shape\": tf.io.FixedLenFeature([3], tf.int64),\n    }\n    parsed_example = tf.io.parse_single_example(example, feature_description)\n    image = tf.io.decode_raw(parsed_example[\"image\"], tf.uint8)\n    label = tf.io.decode_raw(parsed_example[\"label\"], tf.uint8)\n    image_shape = tf.cast(parsed_example[\"image_shape\"], tf.int64)\n    label_shape = tf.cast(parsed_example[\"label_shape\"], tf.int64)\n    image = tf.reshape(image, image_shape)\n    label = tf.reshape(label, label_shape)\n    return image, label","metadata":{"trusted":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_tfrecord_dataset(tfrecord_pattern, batch_size=1):\n    dataset = tf.data.TFRecordDataset(tf.io.gfile.glob(tfrecord_pattern))\n    dataset = dataset.map(parse_tfrecord_fn, num_parallel_calls=tf.data.AUTOTUNE)\n    dataset = dataset.batch(batch_size).prefetch(tf.data.AUTOTUNE)\n    return dataset\n\ntfrecord_pattern = \"tfrecords/{}_shard_*.tfrec\"\ntrain_ds = load_tfrecord_dataset(tfrecord_pattern.format(\"training\"), batch_size=1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x, y = next(iter(train_ds))\nx.shape, y.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_temp = x.numpy().squeeze()\ny_temp = y.numpy().squeeze()\nx_temp.shape, y_temp.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_image = x_temp\ntest_mask = y_temp\nprint(np.unique(test_mask))\n\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 6))\nax1.imshow(test_image[test_image.shape[0]//10], cmap='gray')\nax1.set_title(f'Image shape: {test_image.shape}')\nax2.imshow(test_mask[test_mask.shape[0]//10], cmap='gray')\nax2.set_title(f'Label shape: {test_mask.shape}')\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}