{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n## Importing libraries","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tqdm import tqdm\nimport numpy as np\nimport os\nimport IPython.display as display\nfrom PIL import Image\nimport pandas as pd\nimport shutil\nimport matplotlib.pyplot as plt","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Path Initialization","metadata":{}},{"cell_type":"code","source":"TRAIN_TFRECORD_PATH = '/kaggle/input/cassava-leaf-disease-classification/train_tfrecords/' # Specify path for tfrecord files\nTRAIN_IMAGES_FROM_TFRECORD_PATH = '/kaggle/working/Train_Images/'                     # Specify path where you want to extract images from tfrecord","metadata":{"execution":{"iopub.status.busy":"2021-09-13T09:30:52.724448Z","iopub.execute_input":"2021-09-13T09:30:52.725374Z","iopub.status.idle":"2021-09-13T09:30:52.729101Z","shell.execute_reply.started":"2021-09-13T09:30:52.725333Z","shell.execute_reply":"2021-09-13T09:30:52.728254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's create required path for TRAIN images\n\nif os.path.exists(TRAIN_IMAGES_FROM_TFRECORD_PATH):\n    print('Looks like already exists')\nelse:\n    os.makedirs(TRAIN_IMAGES_FROM_TFRECORD_PATH)\n    print(f\"Great , You have created required folder at {TRAIN_IMAGES_FROM_TFRECORD_PATH}\")","metadata":{"execution":{"iopub.status.busy":"2021-09-13T09:44:33.770554Z","iopub.execute_input":"2021-09-13T09:44:33.770917Z","iopub.status.idle":"2021-09-13T09:44:33.777457Z","shell.execute_reply.started":"2021-09-13T09:44:33.77088Z","shell.execute_reply":"2021-09-13T09:44:33.776818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Let's make hand dirty by converting tfrecord to jpg","metadata":{}},{"cell_type":"code","source":"# # Create a dictionary describing the features. This \nimage_feature_description = {\n    'image': tf.io.FixedLenFeature([], tf.string),  \n    'image_name': tf.io.FixedLenFeature([], tf.string),\n    'target': tf.io.FixedLenFeature([], tf.int64),\n}\n\ndef _parse_image_function(example_proto):\n  # Parse the input tf.train.Example proto using the dictionary above.\n  return tf.io.parse_single_example(example_proto, image_feature_description)","metadata":{"execution":{"iopub.status.busy":"2021-09-13T09:44:36.226103Z","iopub.execute_input":"2021-09-13T09:44:36.227007Z","iopub.status.idle":"2021-09-13T09:44:36.233104Z","shell.execute_reply.started":"2021-09-13T09:44:36.226967Z","shell.execute_reply":"2021-09-13T09:44:36.232169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I want to store tfrecord filename in variable, filename\nfilename = [TRAIN_TFRECORD_PATH + i for i in os.listdir(TRAIN_TFRECORD_PATH)]\nfilename[:2]","metadata":{"execution":{"iopub.status.busy":"2021-09-13T09:44:38.376556Z","iopub.execute_input":"2021-09-13T09:44:38.376891Z","iopub.status.idle":"2021-09-13T09:44:38.387452Z","shell.execute_reply.started":"2021-09-13T09:44:38.376831Z","shell.execute_reply":"2021-09-13T09:44:38.386423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Code without using multiprocessing take appox = 50 iter/s, We have around 21,397 images which will cost us 21397/50 = 427 sec =7.13 minutes,","metadata":{}},{"cell_type":"code","source":"\ndf_to_store_imagename_and_info = pd.DataFrame()\nimage_name_to_df = []\ntarget_name_to_df = []\n\n# Here i am converting only first two tfrecord file, so if you want to convert all tfrecord files, then remove square bracket\n# from parsed parameter to the TFRecordDataset(filename = filename)\n\nraw_image_dataset = tf.data.TFRecordDataset(filenames=filename[:2]) # parsing tfrecords path to the TFrecordDataset function from tensorflow\nparsed_image_dataset = raw_image_dataset.map(_parse_image_function) # Basically our tfrecorf file is in the shape, where it contains\n                                                              # dict with (array of image, image_name, target variable)\n\n\nfor image_features in tqdm(parsed_image_dataset):\n    image_raw = image_features['image'].numpy()        # accessing image array  \n    # display.display(display.Image(data=image_raw))\n    array = tf.io.decode_image(image_raw, dtype=tf.dtypes.uint8, expand_animations=True).numpy() # Decoding image\n    im = Image.fromarray(array)                        # converting as a image\n    image_path = TRAIN_IMAGES_FROM_TFRECORD_PATH +str(image_features['image_name'].numpy(), 'utf8')  # specifying image path with image name\n    im.save(image_path)\n    \n    # we are storing image path and target variable for later use \n    image_name_to_df.append(str(image_features['image_name'].numpy(), 'utf8'))\n    target_name_to_df.append(image_features['target'].numpy())\n\ndf_to_store_imagename_and_info['Image_name'] = image_name_to_df\ndf_to_store_imagename_and_info['Target'] = target_name_to_df\n\nprint('Great you have converted from tfrecord to jpg format :)')","metadata":{"execution":{"iopub.status.busy":"2021-09-13T09:45:40.768854Z","iopub.execute_input":"2021-09-13T09:45:40.76927Z","iopub.status.idle":"2021-09-13T09:46:25.279401Z","shell.execute_reply.started":"2021-09-13T09:45:40.769235Z","shell.execute_reply":"2021-09-13T09:46:25.278443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_to_store_imagename_and_info.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-09-13T09:55:11.722271Z","iopub.execute_input":"2021-09-13T09:55:11.722555Z","iopub.status.idle":"2021-09-13T09:55:11.748326Z","shell.execute_reply.started":"2021-09-13T09:55:11.722528Z","shell.execute_reply":"2021-09-13T09:55:11.747358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we have extracted tfrecrd to jpg format, We have extracted 2676 out of 21397 images, Coooool\nlen(os.listdir(TRAIN_IMAGES_FROM_TFRECORD_PATH))","metadata":{"execution":{"iopub.status.busy":"2021-09-13T09:54:00.982152Z","iopub.execute_input":"2021-09-13T09:54:00.982645Z","iopub.status.idle":"2021-09-13T09:54:00.994435Z","shell.execute_reply.started":"2021-09-13T09:54:00.982595Z","shell.execute_reply":"2021-09-13T09:54:00.993444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualizing image","metadata":{}},{"cell_type":"code","source":"img = tf.io.read_file(TRAIN_IMAGES_FROM_TFRECORD_PATH + os.listdir(TRAIN_IMAGES_FROM_TFRECORD_PATH)[0] )\nimg = tf.image.decode_image(img).numpy()\nplt.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2021-09-13T09:57:52.695615Z","iopub.execute_input":"2021-09-13T09:57:52.696466Z","iopub.status.idle":"2021-09-13T09:57:53.008034Z","shell.execute_reply.started":"2021-09-13T09:57:52.696417Z","shell.execute_reply":"2021-09-13T09:57:53.006919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I am removing the extracted files from output directory , Please comment this line so that you will have images in output directory\nshutil.rmtree(TRAIN_IMAGES_FROM_TFRECORD_PATH)","metadata":{"execution":{"iopub.status.busy":"2021-09-13T09:57:59.289168Z","iopub.execute_input":"2021-09-13T09:57:59.289728Z","iopub.status.idle":"2021-09-13T09:57:59.408881Z","shell.execute_reply.started":"2021-09-13T09:57:59.28969Z","shell.execute_reply":"2021-09-13T09:57:59.407573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Congratulations,  Finally you have reached end of this notebook, If you found useful do like and comment, If I can improve this in anyway\n# Please let me know :)\n","metadata":{},"execution_count":null,"outputs":[]}]}