{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-29T10:05:19.573857Z","iopub.execute_input":"2023-08-29T10:05:19.574257Z","iopub.status.idle":"2023-08-29T10:05:19.581362Z","shell.execute_reply.started":"2023-08-29T10:05:19.574226Z","shell.execute_reply":"2023-08-29T10:05:19.580140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tqdm import tqdm\nimport os\nimport IPython.display as display\nfrom PIL import Image\nimport shutil\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:05:19.583274Z","iopub.execute_input":"2023-08-29T10:05:19.583684Z","iopub.status.idle":"2023-08-29T10:05:19.600076Z","shell.execute_reply.started":"2023-08-29T10:05:19.583651Z","shell.execute_reply":"2023-08-29T10:05:19.598903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Specify path for tfrecord files\nTRAIN_TFRECORD_PATH_192 = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-192x192/train/' \nTRAIN_TFRECORD_PATH_224 = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-224x224/train/' \nTRAIN_TFRECORD_PATH_331 = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-331x331/train/' \nTRAIN_TFRECORD_PATH_512 = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-512x512/train/'\n\nVAL_TFRECORD_PATH_192 = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-192x192/val/' \nVAL_TFRECORD_PATH_224 = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-224x224/val/' \nVAL_TFRECORD_PATH_331 = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-331x331/val/' \nVAL_TFRECORD_PATH_512 = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-512x512/val/'\n\nTEST_TFRECORD_PATH_192 = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-192x192/test/' \nTEST_TFRECORD_PATH_224 = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-224x224/test/' \nTEST_TFRECORD_PATH_331 = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-331x331/test/' \nTEST_TFRECORD_PATH_512 = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-512x512/test/'\n# Specify path where you want to extract images from tfrecord\nTRAIN_IMAGES_FROM_TFRECORD_PATH_192 = './tpu-getting-started-new/tfrecords-jpeg-192x192/train/'           \nTRAIN_IMAGES_FROM_TFRECORD_PATH_224 = './tpu-getting-started-new/tfrecords-jpeg-224x224/train/'           \nTRAIN_IMAGES_FROM_TFRECORD_PATH_331 = './tpu-getting-started-new/tfrecords-jpeg-331x331/train/'           \nTRAIN_IMAGES_FROM_TFRECORD_PATH_512 = './tpu-getting-started-new/tfrecords-jpeg-512x512/train/'       \n\nVAL_IMAGES_FROM_TFRECORD_PATH_192 = './tpu-getting-started-new/tfrecords-jpeg-192x192/val/'           \nVAL_IMAGES_FROM_TFRECORD_PATH_224 = './tpu-getting-started-new/tfrecords-jpeg-224x224/val/'           \nVAL_IMAGES_FROM_TFRECORD_PATH_331 = './tpu-getting-started-new/tfrecords-jpeg-331x331/val/'           \nVAL_IMAGES_FROM_TFRECORD_PATH_512 = './tpu-getting-started-new/tfrecords-jpeg-512x512/val/' \n\nTEST_IMAGES_FROM_TFRECORD_PATH_192 = './tpu-getting-started-new/tfrecords-jpeg-192x192/test/'           \nTEST_IMAGES_FROM_TFRECORD_PATH_224 = './tpu-getting-started-new/tfrecords-jpeg-224x224/test/'           \nTEST_IMAGES_FROM_TFRECORD_PATH_331 = './tpu-getting-started-new/tfrecords-jpeg-331x331/test/'           \nTEST_IMAGES_FROM_TFRECORD_PATH_512 = './tpu-getting-started-new/tfrecords-jpeg-512x512/test/' ","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:05:19.602040Z","iopub.execute_input":"2023-08-29T10:05:19.603012Z","iopub.status.idle":"2023-08-29T10:05:19.615951Z","shell.execute_reply.started":"2023-08-29T10:05:19.602976Z","shell.execute_reply":"2023-08-29T10:05:19.614746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TFRECORD_PATH_TRAIN_VAL = [TRAIN_TFRECORD_PATH_192, TRAIN_TFRECORD_PATH_224, \n                           TRAIN_TFRECORD_PATH_331, TRAIN_TFRECORD_PATH_512, \n                           VAL_TFRECORD_PATH_192, VAL_TFRECORD_PATH_224, \n                           VAL_TFRECORD_PATH_331, VAL_TFRECORD_PATH_512]\n\nTFRECORD_PATH_TEST = [TEST_TFRECORD_PATH_192, TEST_TFRECORD_PATH_224, \n                      TEST_TFRECORD_PATH_331, TEST_TFRECORD_PATH_512]","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:05:19.617693Z","iopub.execute_input":"2023-08-29T10:05:19.618110Z","iopub.status.idle":"2023-08-29T10:05:19.634839Z","shell.execute_reply.started":"2023-08-29T10:05:19.618069Z","shell.execute_reply":"2023-08-29T10:05:19.633288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGES_FROM_TFRECORD_PATH_TRAIN_VAL = [TRAIN_IMAGES_FROM_TFRECORD_PATH_192, TRAIN_IMAGES_FROM_TFRECORD_PATH_224,\n                                       TRAIN_IMAGES_FROM_TFRECORD_PATH_331, TRAIN_IMAGES_FROM_TFRECORD_PATH_512, \n                                       VAL_IMAGES_FROM_TFRECORD_PATH_192, VAL_IMAGES_FROM_TFRECORD_PATH_224, \n                                       VAL_IMAGES_FROM_TFRECORD_PATH_331, VAL_IMAGES_FROM_TFRECORD_PATH_512]\n\nIMAGES_FROM_TFRECORD_PATH_TEST = [TEST_IMAGES_FROM_TFRECORD_PATH_192, TEST_IMAGES_FROM_TFRECORD_PATH_224, \n                                  TEST_IMAGES_FROM_TFRECORD_PATH_331, TEST_IMAGES_FROM_TFRECORD_PATH_512]","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:05:19.636896Z","iopub.execute_input":"2023-08-29T10:05:19.637558Z","iopub.status.idle":"2023-08-29T10:05:19.648869Z","shell.execute_reply.started":"2023-08-29T10:05:19.637523Z","shell.execute_reply":"2023-08-29T10:05:19.647568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's create required path for TRAIN images\nfor tfr_path in IMAGES_FROM_TFRECORD_PATH_TRAIN_VAL:\n    for label in range(104):\n        if os.path.exists(tfr_path + str(label)):\n            print('Looks like already exists')\n        else:\n            os.makedirs(tfr_path + str(label))\n            print(f\"Great! You have created required folder at {tfr_path + str(label)}\")\n        \n# Let's create required path for TRAIN images\nfor tfr_path in IMAGES_FROM_TFRECORD_PATH_TEST:\n    if os.path.exists(tfr_path):\n        print('Looks like already exists')\n    else:\n        os.makedirs(tfr_path)\n        print(f\"Great! You have created required folder at {tfr_path}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:05:19.708711Z","iopub.execute_input":"2023-08-29T10:05:19.709396Z","iopub.status.idle":"2023-08-29T10:05:19.729109Z","shell.execute_reply.started":"2023-08-29T10:05:19.709350Z","shell.execute_reply":"2023-08-29T10:05:19.727881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filenames = ['/kaggle/input/tpu-getting-started/tfrecords-jpeg-192x192/train/00-192x192-798.tfrec']\nraw_dataset = tf.data.TFRecordDataset(filenames)\nraw_dataset","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:05:19.731967Z","iopub.execute_input":"2023-08-29T10:05:19.732732Z","iopub.status.idle":"2023-08-29T10:05:19.754087Z","shell.execute_reply.started":"2023-08-29T10:05:19.732687Z","shell.execute_reply":"2023-08-29T10:05:19.752869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for raw_record in raw_dataset.take(1):\n    example = tf.train.Example()\n    example.ParseFromString(raw_record.numpy())\n    print(example)","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:05:19.756680Z","iopub.execute_input":"2023-08-29T10:05:19.757056Z","iopub.status.idle":"2023-08-29T10:05:19.787923Z","shell.execute_reply.started":"2023-08-29T10:05:19.757026Z","shell.execute_reply":"2023-08-29T10:05:19.786983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train and validation datasets","metadata":{}},{"cell_type":"code","source":"# # Create a dictionary describing the features. This \nimage_feature_description = {\n    'class': tf.io.FixedLenFeature([], tf.int64),\n    'id': tf.io.FixedLenFeature([], tf.string),\n    'image': tf.io.FixedLenFeature([], tf.string)\n}\n\ndef _parse_image_function(example_proto):\n  # Parse the input tf.train.Example proto using the dictionary above.\n  return tf.io.parse_single_example(example_proto, image_feature_description)","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:05:19.789098Z","iopub.execute_input":"2023-08-29T10:05:19.789735Z","iopub.status.idle":"2023-08-29T10:05:19.796282Z","shell.execute_reply.started":"2023-08-29T10:05:19.789699Z","shell.execute_reply":"2023-08-29T10:05:19.795033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I want to store tfrecord filename in variable, filename\nfilename = {}\nfor tfr_path in TFRECORD_PATH_TRAIN_VAL:\n    print(tfr_path)\n    filename[tfr_path[34:]] = [tfr_path + i for i in os.listdir(tfr_path)]\n    \ndf_to_store_imagename_and_info = {}\nimage_name_to_df = {}\ntarget_name_to_df = {}\nfor tfr_path in IMAGES_FROM_TFRECORD_PATH_TRAIN_VAL:\n    print(tfr_path[26:])\n    df_to_store_imagename_and_info[tfr_path[26:]] = pd.DataFrame()\n    image_name_to_df[tfr_path[26:]] = []\n    target_name_to_df[tfr_path[26:]] = []\n\n    # Here i am converting only first two tfrecord file, so if you want to convert all tfrecord files, then remove square bracket\n    # from parsed parameter to the TFRecordDataset(filename = filename)\n\n    raw_image_dataset = tf.data.TFRecordDataset(filenames=filename[tfr_path[26:]]) # parsing tfrecords path to the TFrecordDataset function from tensorflow\n    parsed_image_dataset = raw_image_dataset.map(_parse_image_function) # Basically our tfrecorf file is in the shape, where it contains\n                                                                  # dict with (array of image, image_name, target variable)\n\n\n    for image_features in tqdm(parsed_image_dataset):\n        image_raw = image_features['image'].numpy()        # accessing image array  \n    #     display.display(display.Image(data=image_raw))\n        array = tf.io.decode_image(image_raw, dtype=tf.dtypes.uint8, expand_animations=True).numpy() # Decoding image\n        im = Image.fromarray(array)                        # converting as a image\n        image_path = tfr_path + str(image_features['class'].numpy()) + '/' + str(image_features['id'].numpy(), 'utf8') + '.jpg'#+ '_' +str(image_features['class'].numpy())+'.jpg'  # specifying image path with image name\n        im.save(image_path)\n\n        # we are storing image path and target variable for later use \n        image_name_to_df[tfr_path[26:]].append(str(image_features['id'].numpy(), 'utf8'))\n        target_name_to_df[tfr_path[26:]].append(image_features['class'].numpy())\n\n    df_to_store_imagename_and_info[tfr_path[26:]]['Id'] = image_name_to_df[tfr_path[26:]]\n    df_to_store_imagename_and_info[tfr_path[26:]]['Class'] = target_name_to_df[tfr_path[26:]]\n\nprint('Great you have converted from tfrecord to jpg format :)')","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:05:19.799570Z","iopub.execute_input":"2023-08-29T10:05:19.800573Z","iopub.status.idle":"2023-08-29T10:09:34.131542Z","shell.execute_reply.started":"2023-08-29T10:05:19.800532Z","shell.execute_reply":"2023-08-29T10:09:34.130224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Test dataset","metadata":{}},{"cell_type":"code","source":"# # Create a dictionary describing the features. This \nimage_feature_description = {\n    'id': tf.io.FixedLenFeature([], tf.string),\n    'image': tf.io.FixedLenFeature([], tf.string)\n}\n\ndef _parse_image_function(example_proto):\n  # Parse the input tf.train.Example proto using the dictionary above.\n  return tf.io.parse_single_example(example_proto, image_feature_description)","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:09:34.133571Z","iopub.execute_input":"2023-08-29T10:09:34.133968Z","iopub.status.idle":"2023-08-29T10:09:34.141328Z","shell.execute_reply.started":"2023-08-29T10:09:34.133935Z","shell.execute_reply":"2023-08-29T10:09:34.139844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I want to store tfrecord filename in variable, filename\nfilename = {}\nfor tfr_path in TFRECORD_PATH_TEST:\n    print(tfr_path)\n    filename[tfr_path[34:]] = [tfr_path + i for i in os.listdir(tfr_path)]\n    \ndf_to_store_imagename_and_info = {}\nimage_name_to_df = {}\ntarget_name_to_df = {}\nfor tfr_path in IMAGES_FROM_TFRECORD_PATH_TEST:\n    print(tfr_path[26:])\n    df_to_store_imagename_and_info[tfr_path[26:]] = pd.DataFrame()\n    image_name_to_df[tfr_path[26:]] = []\n    target_name_to_df[tfr_path[26:]] = []\n\n    # Here i am converting only first two tfrecord file, so if you want to convert all tfrecord files, then remove square bracket\n    # from parsed parameter to the TFRecordDataset(filename = filename)\n\n    raw_image_dataset = tf.data.TFRecordDataset(filenames=filename[tfr_path[26:]]) # parsing tfrecords path to the TFrecordDataset function from tensorflow\n    parsed_image_dataset = raw_image_dataset.map(_parse_image_function) # Basically our tfrecorf file is in the shape, where it contains\n                                                                  # dict with (array of image, image_name, target variable)\n\n\n    for image_features in tqdm(parsed_image_dataset):\n        image_raw = image_features['image'].numpy()        # accessing image array  \n    #     display.display(display.Image(data=image_raw))\n        array = tf.io.decode_image(image_raw, dtype=tf.dtypes.uint8, expand_animations=True).numpy() # Decoding image\n        im = Image.fromarray(array)                        # converting as a image\n        image_path = tfr_path +str(image_features['id'].numpy(), 'utf8')+'.jpg'  # specifying image path with image name\n        im.save(image_path)\n\n        # we are storing image path and target variable for later use \n        image_name_to_df[tfr_path[26:]].append(str(image_features['id'].numpy(), 'utf8'))\n\n    df_to_store_imagename_and_info[tfr_path[26:]]['Id'] = image_name_to_df[tfr_path[26:]]\n\nprint('Great you have converted from tfrecord to jpg format :)')","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:09:34.142773Z","iopub.execute_input":"2023-08-29T10:09:34.143526Z","iopub.status.idle":"2023-08-29T10:11:58.955325Z","shell.execute_reply.started":"2023-08-29T10:09:34.143482Z","shell.execute_reply":"2023-08-29T10:11:58.953889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### If you found this notebook useful do like and comment. Thank you","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}