{"cells":[{"metadata":{},"cell_type":"markdown","source":"# How To Create TFRecords\nIn this notebook, we learn how to create TFRecords to train TensorFlow models. We will create TFRecords from the Kaggle dataset of 512x512x3 jpegs [here][1]. This dataset contains the Melanoma Classification competition data (train 30,000 and test 10,000 ) and an additional 30,000 external images. It was published by [Alex Shonenkov][2]\n\nThere is a discussion post about these TFRecords [here][3] and Alex discusses where these images came from [here][4]\n\n[1]: https://www.kaggle.com/shonenkov/melanoma-merged-external-data-512x512-jpeg\n[2]: https://www.kaggle.com/shonenkov\n[3]: https://www.kaggle.com/c/siim-isic-melanoma-classification/discussion/156245\n[4]: https://www.kaggle.com/c/siim-isic-melanoma-classification/discussion/155859","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# Load Meta Data","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# LOAD LIBRARIES\nimport numpy as np, pandas as pd, os\nimport matplotlib.pyplot as plt, cv2\nimport tensorflow as tf, re, math","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# PATHS TO IMAGES\nPATH = '../input/melanoma-merged-external-data-512x512-jpeg/512x512-dataset-melanoma/512x512-dataset-melanoma/'\nPATH2 = '../input/melanoma-merged-external-data-512x512-jpeg/512x512-test/512x512-test/'\nIMGS = os.listdir(PATH); IMGS2 = os.listdir(PATH2)\nprint('There are %i train images and %i test images'%(len(IMGS),len(IMGS2)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# LOAD TRAIN META DATA\ndf = pd.read_csv('../input/melanoma-merged-external-data-512x512-jpeg/marking.csv')\ndf.rename({'image_id':'image_name'},axis=1,inplace=True)\ndf['target']=df['target'].astype(int)\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# LOAD TEST META DATA\ntest = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv')\npsevdolabel =  pd.read_csv(\"../input/subkaggle2/psevdo_label.csv\")\ntest['target'] = psevdolabel['target'].astype(int)\ndel psevdolabel                             \ntest.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Label Encode Meta Data\nIt is more efficient to store this meta data as integers instead of strings. We will impute the Age NaNs to Age mean. Then all other NaNs will be convert to `-1` and the other strings will be converted to `0, 1, 2, 3, ...` in the order they appear in the printed lists below.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# COMBINE TRAIN AND TEST TO ENCODE TOGETHER\ncols = test.columns\ncomb = pd.concat([df[cols],test[cols]],ignore_index=True,axis=0).reset_index(drop=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# LABEL ENCODE ALL STRINGS\ncats = ['patient_id','sex','anatom_site_general_challenge'] \nfor c in cats:\n    comb[c],mp = comb[c].factorize()\n    print(mp)\nprint('Imputing Age NaN count =',comb.age_approx.isnull().sum())\ncomb.age_approx.fillna(comb.age_approx.mean(),inplace=True)\ncomb['age_approx'] = comb.age_approx.astype('int')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# REWRITE DATA TO DATAFRAMES\ndf[cols] = comb.loc[:df.shape[0]-1,cols].values\ntest[cols] = comb.loc[df.shape[0]:,cols].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# LABEL ENCODE TRAIN SOURCE\ndf.source,mp = df.source.factorize()\nprint(mp)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"> # Write TFRecords - Train\nAll the code below comes from TensorFlow's docs [here][1]\n\n[1]: https://www.tensorflow.org/tutorials/load_data/tfrecord","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def _bytes_feature(value):\n  \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n  if isinstance(value, type(tf.constant(0))):\n    value = value.numpy() # BytesList won't unpack a string from an EagerTensor.\n  return tf.train.Feature(bytes_list=tf.train.BytesList(value=[value]))\n\ndef _float_feature(value):\n  \"\"\"Returns a float_list from a float / double.\"\"\"\n  return tf.train.Feature(float_list=tf.train.FloatList(value=[value]))\n\ndef _int64_feature(value):\n  \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n  return tf.train.Feature(int64_list=tf.train.Int64List(value=[value]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def serialize_example(feature0, feature1, feature2, feature3, feature4, feature5, feature6, feature7):\n  feature = {\n      'image': _bytes_feature(feature0),\n      'image_name': _bytes_feature(feature1),\n      'patient_id': _int64_feature(feature2),\n      'sex': _int64_feature(feature3),\n      'age_approx': _int64_feature(feature4),\n      'anatom_site_general_challenge': _int64_feature(feature5),\n      'source': _int64_feature(feature6),\n      'target': _int64_feature(feature7)\n  }\n  example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n  return example_proto.SerializeToString()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"SIZE = 2071\nCT = len(IMGS)//SIZE + int(len(IMGS)%SIZE!=0)\nfor j in range(CT):\n    print(); print('Writing TFRecord %i of %i...'%(j,CT))\n    CT2 = min(SIZE,len(IMGS)-j*SIZE)\n    with tf.io.TFRecordWriter('train%.2i-%i.tfrec'%(j,CT2)) as writer:\n        for k in range(CT2):\n            img = cv2.imread(PATH+IMGS[SIZE*j+k])\n            img = cv2.resize(img,(384,384))\n            img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR) # Fix incorrect colors\n            img = cv2.imencode('.jpg', img, (cv2.IMWRITE_JPEG_QUALITY, 94))[1].tostring()\n            name = IMGS[SIZE*j+k].split('.')[0]\n            row = df.loc[df.image_name==name]\n            example = serialize_example(\n                img, str.encode(name),\n                row.patient_id.values[0],\n                row.sex.values[0],\n                row.age_approx.values[0],                        \n                row.anatom_site_general_challenge.values[0],\n                row.source.values[0],\n                row.target.values[0])\n            writer.write(example)\n            if k%100==0: print(k,', ',end='')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"! ls -l","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Write TFRecords - Test","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def serialize_example2(feature0, feature1, feature2, feature3, feature4, feature5,feature6): \n  feature = {\n      'image': _bytes_feature(feature0),\n      'image_name': _bytes_feature(feature1),\n      'patient_id': _int64_feature(feature2),\n      'sex': _int64_feature(feature3),\n      'age_approx': _int64_feature(feature4),\n      'anatom_site_general_challenge': _int64_feature(feature5),\n      'target': _int64_feature(feature6)\n  }\n  example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n  return example_proto.SerializeToString()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"SIZE = 687\nCT = len(IMGS2)//SIZE + int(len(IMGS2)%SIZE!=0)\nfor j in range(CT):\n    print(); print('Writing TFRecord %i of %i...'%(j,CT))\n    CT2 = min(SIZE,len(IMGS2)-j*SIZE)\n    with tf.io.TFRecordWriter('test%.2i-%i.tfrec'%(j,CT2)) as writer:\n        for k in range(CT2):\n            img = cv2.imread(PATH2+IMGS2[SIZE*j+k])\n            img = cv2.resize(img,(384,384))\n            img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR) # Fix incorrect colors\n            img = cv2.imencode('.jpg', img, (cv2.IMWRITE_JPEG_QUALITY, 94))[1].tostring()\n            name = IMGS2[SIZE*j+k].split('.')[0]\n            row = test.loc[test.image_name==name]\n            example = serialize_example2(\n                img, str.encode(name),\n                row.patient_id.values[0],\n                row.sex.values[0],\n                row.age_approx.values[0],                        \n                row.anatom_site_general_challenge.values[0],\n                row.target.values[0])\n            writer.write(example)\n            if k%100==0: print(k,', ',end='')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}