{"cells": [{"source": ["My other kernel can be used to transform BSON files to TFRecord. This one shows how to read created TFRecord files."], "metadata": {}, "cell_type": "markdown"}, {"execution_count": null, "outputs": [], "source": ["# This Python 3 environment comes with many helpful analytics libraries installed\n", "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n", "# For example, here's several helpful packages to load in \n", "\n", "import io\n", "import bson                       # this is installed with the pymongo package\n", "import matplotlib.pyplot as plt\n", "from skimage.data import imread   # or, whatever image library you prefer\n", "import multiprocessing as mp      # will come in handy due to the size of the data\n", "import tensorflow as tf\n", "#print(check_output([\"ls\", \"input\"]).decode(\"utf8\"))\n", "from time import time \n", "\n", "%matplotlib inline"], "metadata": {"_cell_guid": "82d5ad62-6da1-44dc-b4e8-59d16eb325ed", "_uuid": "cb745c17fa4638dcf37701a5d109e686bad5241d", "collapsed": true}, "cell_type": "code"}, {"execution_count": null, "outputs": [], "source": ["tfrecords_filename = 'input/tfrecord/img_{}.tfrecords'\n", "opts = tf.python_io.TFRecordOptions(tf.python_io.TFRecordCompressionType.ZLIB)"], "metadata": {"collapsed": true}, "cell_type": "code"}, {"execution_count": null, "outputs": [], "source": ["record_iterator = tf.python_io.tf_record_iterator(path=tfrecords_filename.format(0), options=opts)\n", "img_string = \"\"\n", "for string_record in record_iterator:\n", "    example = tf.train.Example()\n", "    example.ParseFromString(string_record)\n", "    height = int(example.features.feature['height']\n", "                                 .int64_list\n", "                                 .value[0])\n", "    \n", "    width = int(example.features.feature['width']\n", "                                .int64_list\n", "                                .value[0])\n", "    \n", "    img_string = (example.features.feature['img_raw'].bytes_list.value[0])\n", "    \n", "    category_id = (example.features.feature['category_id']\n", "                                .int64_list\n", "                                .value[0])\n", "\n", "    product_id = (example.features.feature['product_id']\n", "                                .int64_list\n", "                                .value[0])\n", "    \n", "    print(height, width, category_id, product_id)\n", "    if img_string != \"\":\n", "        break\n", "    "], "metadata": {"collapsed": true}, "cell_type": "code"}, {"execution_count": null, "outputs": [], "source": ["plt.imshow(imread(io.BytesIO(img_string)))"], "metadata": {"collapsed": true}, "cell_type": "code"}], "nbformat_minor": 1, "nbformat": 4, "metadata": {"kernelspec": {"language": "python", "display_name": "Python 3", "name": "python3"}, "language_info": {"mimetype": "text/x-python", "pygments_lexer": "ipython3", "file_extension": ".py", "version": "3.6.1", "nbconvert_exporter": "python", "name": "python", "codemirror_mode": {"name": "ipython", "version": 3}}}}