{"cells":[{"metadata":{},"cell_type":"markdown","source":"TODO:  \nBreak the chunks into manageable tfrec sizes  \n5 GB local disk space  \nAnd 9 hour exec time  ","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.model_selection import train_test_split\nfrom matplotlib import pyplot as plt\n\nfrom kaggle_datasets import KaggleDatasets\n\n#import efficientnet.tfkeras as efn\nimport tensorflow as tf\nimport tensorflow.keras.layers as L\nprint(\"Tensorflow version \" + tf.__version__)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\nimport IPython.display as display","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"GCS_DS_PATH = KaggleDatasets().get_gcs_path();GCS_DS_PATH","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"! ls /kaggle/input/imet-2020-fgvc7","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/imet-2020-fgvc7/train.csv')\nsubmissions_df = pd.read_csv('/kaggle/input/imet-2020-fgvc7/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df[\"attribute_ids\"] = df[\"attribute_ids\"].apply(lambda x:x.split(\" \"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Fit the multi-label binarizer on the training set\nprint(\"Labels:\")\nmlb = MultiLabelBinarizer()\nmlb.fit(df.attribute_ids.values)\n\n# Loop over all labels and show them\nN_LABELS = len(mlb.classes_)\n# for (i, label) in enumerate(mlb.classes_):\n#     print(\"{}. {}\".format(i, label))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"N_LABELS #3471","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_bin = mlb.transform(df.attribute_ids.values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head(2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.sum(y_bin[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.where(y_bin[0] == 1)[0] #so 4588628","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"mlb.classes_[1588],mlb.classes_[2860],mlb.classes_[3230] # and that is how you get them values back","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def format_path_train(st):\n    return GCS_DS_PATH + '/train/' + st + '.png'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def format_path_test(st):\n    return GCS_DS_PATH + '/test/' + st + '.png'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_image_paths = df['id'].apply(format_path_train).values #apply(format_path).values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(train_image_paths),df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_image_paths_chunks = list(np.array_split(train_image_paths, 64)) #so 14406567\ntrain_image_labels_chunks = list(np.array_split(y_bin, 64))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(train_image_paths_chunks)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(train_image_paths_chunks[0])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Save to disk","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# The following functions can be used to convert a value to a type compatible\n# with tf.Example.\n\ndef _bytes_feature(value):\n  \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n  if isinstance(value, type(tf.constant(0))):\n    value = value.numpy() # BytesList won't unpack a string from an EagerTensor.\n  return tf.train.Feature(bytes_list=tf.train.BytesList(value=[value]))\n\ndef _bytes_featureList(value):\n  \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n  if isinstance(value, type(tf.constant(0))):\n    value = value.numpy() # BytesList won't unpack a string from an EagerTensor.\n  return tf.train.FeatureList(bytes_list=tf.train.BytesList(value=[value]))\n\ndef _float_feature(value):\n  \"\"\"Returns a float_list from a float / double.\"\"\"\n  return tf.train.Feature(float_list=tf.train.FloatList(value=[value]))\n\ndef _int64_feature(value):\n  \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n  #return tf.train.Feature(int64_list=tf.train.Int64List(value=[value]))\n  return tf.train.Feature(int64_list=tf.train.Int64List(value=value)) # becaus ethis is already a list..","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def _bytestring_feature(list_of_bytestrings):\n    return tf.train.Feature(bytes_list=tf.train.BytesList(value=list_of_bytestrings))\n\ndef _int_feature(list_of_ints): # int64\n    return tf.train.Feature(int64_list=tf.train.Int64List(value=list_of_ints))\n\ndef _float_feature(list_of_floats): # float32\n    return tf.train.Feature(float_list=tf.train.FloatList(value=list_of_floats))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def image_example(filename, label):\n  bits = tf.io.read_file(filename) # returns a tensor of type string\n  #image = tf.image.decode_jpeg(bits, channels=3)\n  #image_shape = tf.image.decode_jpeg(image_string).shape\n  feature = {\n      #'label': _float_feature(label),\n      'label': _int_feature(label), # SO #45427637 #47861084\n      'image_raw': _bytes_feature(bits),\n  }\n\n  return tf.train.Example(features=tf.train.Features(feature=feature))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(train_image_paths_chunks[1]) # 1111 when 128 split\n# each file then is 190 mb # 8 mins per file\n#len(train_image_paths_chunks[1]) # 2221 when 64 split\n# each file then is 382  mb // 11 mins per file","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(train_image_paths_chunks)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\nfor i in range(len(train_image_paths_chunks)):\n#for i in range(1):\n  record_file = 'imet_tfrecord_images_'+str(i)+'.tfrec'\n  with tf.io.TFRecordWriter(record_file) as writer:  # the writer context is used in the last/end\n    #for a,b in zip(train_image_paths_chunks[i][:5], train_image_labels_chunks[i][:5]):# start with 5 of them\n    for a,b in zip(train_image_paths_chunks[i], train_image_labels_chunks[i]):\n      tf_example = image_example(a, b)\n      writer.write(tf_example.SerializeToString())\n  writer.close()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# https://github.com/tensorflow/tensorflow/issues/4467\n# Slow and storage consming indeed","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#!find /kaggle/working/ -name '*.tfrec' -delete","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## try reading them back...","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"image_feature_description = {\n      'label': tf.io.FixedLenFeature([N_LABELS], tf.int64), \n      'image_raw': tf.io.FixedLenFeature([], tf.string),\n}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def _parse_image_function(example_proto):\n  # Parse the input tf.Example proto using the dictionary above.\n  return tf.io.parse_single_example(example_proto, image_feature_description)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_dataset_2 = tf.data.TFRecordDataset('imet_tfrecord_images_0.tfrec', num_parallel_reads=AUTO) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"parsed_image_dataset_2 = test_dataset_2.map(_parse_image_function)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for image_features in parsed_image_dataset_2.take(5):\n  image_raw = image_features['image_raw'].numpy()\n  #print(image_raw)\n  #image = Image.open(io.BytesIO(image_raw))\n  #pilimg = Image.fromarray(image_raw)\n  display.display(display.Image(data=image_raw))\n  print(image_features['label'].numpy())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# move the tf recs to a data set.. and use the gcs id of that  dataset to feed in the tpu training..\n# similar to the flowers tpu challenge","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}