{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Dataset conversion\n\nWork with images instead of tfrecords - Useful for Pytorch 🔥 users","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom PIL import Image\nfrom tqdm import tqdm\nimport tensorflow as tf\n\n# change in_path according to input folders (e.g. /kaggle/input/tpu-getting-started/tfrecords-jpeg-331x331)\nin_path = \"/kaggle/input/tpu-getting-started/tfrecords-jpeg-512x512\"\n# out_path folder\nout_path = \"images-png-512x512\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-05T14:16:24.892829Z","iopub.execute_input":"2023-02-05T14:16:24.894393Z","iopub.status.idle":"2023-02-05T14:16:24.900085Z","shell.execute_reply.started":"2023-02-05T14:16:24.894346Z","shell.execute_reply":"2023-02-05T14:16:24.899163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Convert from tfrecords to png","metadata":{}},{"cell_type":"code","source":"def read_tfrecord(example, test=False):\n    # Test tfrecord has only image and id keys\n    if test:\n        tfrec_format = {\n            \"image\": tf.io.FixedLenFeature([], tf.string),\n            \"id\": tf.io.FixedLenFeature([], tf.string),\n        }\n        example = tf.io.parse_single_example(example, tfrec_format)\n        ids_or_label = example['id']\n    # Train/Val tfrecord has image and class keys\n    else:\n        tfrec_format = {\n            \"image\": tf.io.FixedLenFeature([], tf.string),\n            \"class\": tf.io.FixedLenFeature([], tf.int64),\n        }\n        example = tf.io.parse_single_example(example, tfrec_format)\n        ids_or_label = tf.cast(example['class'], tf.int32)\n    # example[\"image\"] is the binary encoding of the jpeg image, the following line will dencode it to raw image:\n    image = tf.image.decode_jpeg(example['image'], channels=3)\n    return image, ids_or_label\n\ndef read_test_tfrecord(example):\n    return read_tfrecord(example, test=True)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:04:14.242611Z","iopub.execute_input":"2023-02-05T14:04:14.242939Z","iopub.status.idle":"2023-02-05T14:04:14.253346Z","shell.execute_reply.started":"2023-02-05T14:04:14.242908Z","shell.execute_reply":"2023-02-05T14:04:14.251610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"k = 0\nfor folder_split in [\"train\", \"val\", \"test\"]:\n    root = os.path.join(in_path, folder_split)\n    \n    # Form the list of .tfrec files for the local folder (train, val, test)\n    filenames = [os.path.join(root, f) for f in os.listdir(root)]\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=None)\n    dataset = dataset.map(read_tfrecord if folder_split != \"test\" else read_test_tfrecord)\n    \n    # Count approxmately the number of element of the tfrecord (use batch to iterate chunks of 64 elements to speedup)\n    n_tot = sum([1 for i in dataset.batch(64)])*64\n    \n    out_folder = os.path.join(out_path, folder_split)\n    os.makedirs(out_folder, exist_ok=True)\n    # Iterate over dataset and convert in jpeg file using PIL Image\n    print(\"Converting split\", folder_split)\n    for iter, data in enumerate(tqdm(dataset, total=n_tot)):\n        image, ids_or_label = data\n        image = Image.fromarray(image.numpy())\n        if folder_split == \"test\":\n            ids = str(ids_or_label.numpy().decode())\n            image.save(os.path.join(out_folder, f\"{ids}.png\"))\n        else:\n            label = str(ids_or_label.numpy())\n            out_label = os.path.join(out_folder, label)\n            if not os.path.isdir(out_label):\n                os.makedirs(out_label, exist_ok=True)\n            # Save the image\n            image.save(os.path.join(out_folder, label, f\"{k:04d}.png\"))\n            k += 1\n","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:20:24.596377Z","iopub.execute_input":"2023-02-05T14:20:24.597589Z","iopub.status.idle":"2023-02-05T14:20:33.191012Z","shell.execute_reply.started":"2023-02-05T14:20:24.597535Z","shell.execute_reply":"2023-02-05T14:20:33.189438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create CSV dataset","metadata":{}},{"cell_type":"code","source":"file_list = []\nclass_list = []\nsplit_list = []\nfor split in [\"train\", \"val\"]:\n    out_folder = os.path.join(out_path, split)\n    classes = os.listdir(out_folder)\n    \n    for c in classes:\n        class_folder = os.path.join(out_folder, c)\n        png_files = os.listdir(class_folder)\n        for png in png_files:\n            file_list.append(png)\n            class_list.append(c)\n            split_list.append(split)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:23:58.727687Z","iopub.execute_input":"2023-02-05T14:23:58.728162Z","iopub.status.idle":"2023-02-05T14:23:58.740621Z","shell.execute_reply.started":"2023-02-05T14:23:58.728126Z","shell.execute_reply":"2023-02-05T14:23:58.739245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame()\ndf[\"class\"] = class_list\ndf[\"file\"] = file_list\ndf[\"split\"] = split_list\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:24:02.406739Z","iopub.execute_input":"2023-02-05T14:24:02.407235Z","iopub.status.idle":"2023-02-05T14:24:02.426354Z","shell.execute_reply.started":"2023-02-05T14:24:02.407196Z","shell.execute_reply":"2023-02-05T14:24:02.424516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save ot csv\ndf.to_csv(\"train_val.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-02-04T22:53:54.411009Z","iopub.execute_input":"2023-02-04T22:53:54.411490Z","iopub.status.idle":"2023-02-04T22:53:54.421224Z","shell.execute_reply.started":"2023-02-04T22:53:54.411438Z","shell.execute_reply":"2023-02-04T22:53:54.420145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}