{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# How To Create TFRecords\nIn this notebook, we learn how to create TFRecords to train TensorFlow models. We will create TFRecords from the Kaggle dataset of 512x512x3 jpegs [here][1]. This dataset contains the Melanoma Classification competition data (train 30,000 and test 10,000 ) and an additional 30,000 external images. It was published by [Alex Shonenkov][2]\n\nThere is a discussion post about these TFRecords [here][3] and Alex discusses where these images came from [here][4]\n\n[1]: https://www.kaggle.com/shonenkov/melanoma-merged-external-data-512x512-jpeg\n[2]: https://www.kaggle.com/shonenkov\n[3]: https://www.kaggle.com/c/siim-isic-melanoma-classification/discussion/156245\n[4]: https://www.kaggle.com/c/siim-isic-melanoma-classification/discussion/155859","metadata":{}},{"cell_type":"markdown","source":"# Load Meta Data","metadata":{}},{"cell_type":"code","source":"# LOAD LIBRARIES\nimport numpy as np, pandas as pd, os\nimport matplotlib.pyplot as plt, cv2\nimport tensorflow as tf, re, math\nimport tensorflow.keras.backend as K","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-10-17T03:24:59.198012Z","iopub.execute_input":"2021-10-17T03:24:59.198374Z","iopub.status.idle":"2021-10-17T03:24:59.204154Z","shell.execute_reply.started":"2021-10-17T03:24:59.198343Z","shell.execute_reply":"2021-10-17T03:24:59.202965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfrec_shape = 256\ncrop_size = {256: 256, 384: 370, 512: 500, 768: 750}\nnet_size = {256: 256, 384: 370, 512: 500, 768: 750}\n\nCFG = dict(\n    read_size=tfrec_shape,\n    crop_size=crop_size[tfrec_shape],\n    net_size=net_size[tfrec_shape],\n\n    # DATA AUGMENTATION\n    rot=180.0,\n    shr=1.5,\n    hzoom=6.0,\n    wzoom=6.0,\n    hshift=6.0,\n    wshift=6.0,\n    num_aug=55,\n\n    # HAIR AUGMENTATION:\n    # hair_augm = hair_augm[tfrec_shape],\n)","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:24:59.206346Z","iopub.execute_input":"2021-10-17T03:24:59.206813Z","iopub.status.idle":"2021-10-17T03:24:59.220016Z","shell.execute_reply.started":"2021-10-17T03:24:59.206769Z","shell.execute_reply":"2021-10-17T03:24:59.219192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PATHS TO IMAGES\nPATH = '../input/siim-isic-melanoma-classification/jpeg/train/'\nIMGS = os.listdir(PATH)\nprint('There are %i train images'%(len(IMGS)))","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:24:59.221565Z","iopub.execute_input":"2021-10-17T03:24:59.221981Z","iopub.status.idle":"2021-10-17T03:24:59.250399Z","shell.execute_reply.started":"2021-10-17T03:24:59.221940Z","shell.execute_reply":"2021-10-17T03:24:59.249283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LOAD TRAIN META DATA\ndf = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:24:59.253898Z","iopub.execute_input":"2021-10-17T03:24:59.254268Z","iopub.status.idle":"2021-10-17T03:24:59.319862Z","shell.execute_reply.started":"2021-10-17T03:24:59.254233Z","shell.execute_reply":"2021-10-17T03:24:59.318892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LOAD TEST META DATA\n# test = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv')\n# test.head()","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:24:59.321851Z","iopub.execute_input":"2021-10-17T03:24:59.322286Z","iopub.status.idle":"2021-10-17T03:24:59.326520Z","shell.execute_reply.started":"2021-10-17T03:24:59.322242Z","shell.execute_reply":"2021-10-17T03:24:59.325442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Label Encode Meta Data\nIt is more efficient to store this meta data as integers instead of strings. We will impute the Age NaNs to Age mean. Then all other NaNs will be convert to `-1` and the other strings will be converted to `0, 1, 2, 3, ...` in the order they appear in the printed lists below.","metadata":{}},{"cell_type":"code","source":"# COMBINE TRAIN AND TEST TO ENCODE TOGETHER\n# cols = test.columns\n# print([df[cols],test[cols]])\n# comb = pd.concat([df[cols],test[cols]],ignore_index=True,axis=0).reset_index(drop=True)\n# print(comb)","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:24:59.328240Z","iopub.execute_input":"2021-10-17T03:24:59.328640Z","iopub.status.idle":"2021-10-17T03:24:59.337792Z","shell.execute_reply.started":"2021-10-17T03:24:59.328596Z","shell.execute_reply":"2021-10-17T03:24:59.336562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LABEL ENCODE ALL STRINGS\n# cats = ['patient_id','sex','anatom_site_general_challenge'] \n# for c in cats:\n#     comb[c],mp = comb[c].factorize()\n#     print(mp)\n# print('Imputing Age NaN count =',comb.age_approx.isnull().sum())\n# comb.age_approx.fillna(comb.age_approx.mean(),inplace=True)\n# comb['age_approx'] = comb.age_approx.astype('int')","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:24:59.339032Z","iopub.execute_input":"2021-10-17T03:24:59.339393Z","iopub.status.idle":"2021-10-17T03:24:59.350340Z","shell.execute_reply.started":"2021-10-17T03:24:59.339315Z","shell.execute_reply":"2021-10-17T03:24:59.349096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# REWRITE DATA TO DATAFRAMES\n# df[cols] = comb.loc[:df.shape[0]-1,cols].values\n# test[cols] = comb.loc[df.shape[0]:,cols].values","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:24:59.352111Z","iopub.execute_input":"2021-10-17T03:24:59.352631Z","iopub.status.idle":"2021-10-17T03:24:59.361249Z","shell.execute_reply.started":"2021-10-17T03:24:59.352583Z","shell.execute_reply":"2021-10-17T03:24:59.360103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LABEL ENCODE TRAIN SOURCE\n# df.source,mp = df.source.factorize()\n# print(mp)\n\n# LABEL ENCODE ALL STRINGS\ncats = ['patient_id','sex','anatom_site_general_challenge', 'diagnosis'] \nfor c in cats:\n    df[c],mp = df[c].factorize()\n    print(mp)\nprint('Imputing Age NaN count =',df.age_approx.isnull().sum())\ndf.age_approx.fillna(df.age_approx.mean(),inplace=True)\ndf['age_approx'] = df.age_approx.astype('int')","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:24:59.363795Z","iopub.execute_input":"2021-10-17T03:24:59.364328Z","iopub.status.idle":"2021-10-17T03:24:59.406866Z","shell.execute_reply.started":"2021-10-17T03:24:59.364279Z","shell.execute_reply":"2021-10-17T03:24:59.405787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['benign_malignant'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:24:59.408487Z","iopub.execute_input":"2021-10-17T03:24:59.408907Z","iopub.status.idle":"2021-10-17T03:24:59.417605Z","shell.execute_reply.started":"2021-10-17T03:24:59.408862Z","shell.execute_reply":"2021-10-17T03:24:59.416720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:24:59.422177Z","iopub.execute_input":"2021-10-17T03:24:59.422820Z","iopub.status.idle":"2021-10-17T03:24:59.436114Z","shell.execute_reply.started":"2021-10-17T03:24:59.422773Z","shell.execute_reply":"2021-10-17T03:24:59.434885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_0 = df[df.target==0]\ndf_1 = df[df.target==1]\n\ndf_0_80 = df_0[0:int(0.8*len(df_0))]\ndf_0_20 = df_0[int(0.8*len(df_0)):len(df_0)]\ndf_1_80 = df_1[0:int(0.8*len(df_1))]\ndf_1_20 = df_1[int(0.8*len(df_1)):len(df_1)]\n\ndf_80 = pd.concat([df_0_80,df_1_80],ignore_index=True,axis=0).reset_index(drop=True)\nval_20 = pd.concat([df_0_20,df_1_20],ignore_index=True,axis=0).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:24:59.438882Z","iopub.execute_input":"2021-10-17T03:24:59.439339Z","iopub.status.idle":"2021-10-17T03:24:59.458926Z","shell.execute_reply.started":"2021-10-17T03:24:59.439294Z","shell.execute_reply":"2021-10-17T03:24:59.458089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(df_0_80))\nprint(len(df_0_20))\nprint(len(df_1_80))\nprint(len(df_1_20))\nprint('')\nprint(len(df_80))\nprint(len(val_20))","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:24:59.460363Z","iopub.execute_input":"2021-10-17T03:24:59.460768Z","iopub.status.idle":"2021-10-17T03:24:59.467045Z","shell.execute_reply.started":"2021-10-17T03:24:59.460735Z","shell.execute_reply":"2021-10-17T03:24:59.465686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 16))\nfor i in range(9):\n    plt.subplot(3, 3, i + 1)\n    img_path = PATH + df_80.iloc[i].image_name + '.jpg'\n    img = plt.imread(img_path)\n    plt.imshow(img, cmap='gray')\n    plt.axis('off')\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:24:59.469223Z","iopub.execute_input":"2021-10-17T03:24:59.469982Z","iopub.status.idle":"2021-10-17T03:25:18.693594Z","shell.execute_reply.started":"2021-10-17T03:24:59.469937Z","shell.execute_reply":"2021-10-17T03:25:18.692658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_SIZE = (256, 256)\n\ndef center_crop(img, new_size=IMG_SIZE):\n    (height, width, _) = img.shape\n\n    new_width = min(width, height)\n    new_height = min(width, height)\n\n    left = int(np.ceil((width - new_width) / 2))\n    right = width - int(np.floor((width - new_width) / 2))\n\n    top = int(np.ceil((height - new_height) / 2))\n    bottom = height - int(np.floor((height - new_height) / 2))\n\n    center_cropped_img = img[top:bottom, left:right]\n\n    center_cropped_img = cv2.resize(center_cropped_img, IMG_SIZE, interpolation=cv2.INTER_AREA)\n    return center_cropped_img\n\n\nplt.figure(figsize=(16, 16))\nfor i in range(9):\n    plt.subplot(3, 3, i + 1)\n    img_path = PATH + df_80.iloc[i].image_name + '.jpg'\n    img = cv2.imread(img_path)\n    img = center_crop(img)\n    img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR)\n    plt.imshow(img, cmap='gray')\n    plt.axis('off')\nplt.tight_layout() ","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:25:18.695150Z","iopub.execute_input":"2021-10-17T03:25:18.695443Z","iopub.status.idle":"2021-10-17T03:25:22.906113Z","shell.execute_reply.started":"2021-10-17T03:25:18.695414Z","shell.execute_reply":"2021-10-17T03:25:22.905252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> # Write TFRecords - Train\nAll the code below comes from TensorFlow's docs [here][1]\n\n[1]: https://www.tensorflow.org/tutorials/load_data/tfrecord","metadata":{}},{"cell_type":"code","source":"def _bytes_feature(value):\n  \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n  if isinstance(value, type(tf.constant(0))):\n    value = value.numpy() # BytesList won't unpack a string from an EagerTensor.\n  return tf.train.Feature(bytes_list=tf.train.BytesList(value=[value]))\n\ndef _float_feature(value):\n  \"\"\"Returns a float_list from a float / double.\"\"\"\n  return tf.train.Feature(float_list=tf.train.FloatList(value=[value]))\n\ndef _int64_feature(value):\n  \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n  return tf.train.Feature(int64_list=tf.train.Int64List(value=[value]))","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:25:22.907181Z","iopub.execute_input":"2021-10-17T03:25:22.907584Z","iopub.status.idle":"2021-10-17T03:25:22.915450Z","shell.execute_reply.started":"2021-10-17T03:25:22.907545Z","shell.execute_reply":"2021-10-17T03:25:22.914382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def serialize_example(feature0, feature1, feature2, feature3, feature4, feature5, feature6, feature7):\n  feature = {\n      'image': _bytes_feature(feature0),\n      'image_name': _bytes_feature(feature1),\n      'patient_id': _int64_feature(feature2),\n      'sex': _int64_feature(feature3),\n      'age_approx': _int64_feature(feature4),\n      'anatom_site_general_challenge': _int64_feature(feature5),\n      'diagnosis': _int64_feature(feature6),\n      'target': _int64_feature(feature7)\n  }\n  example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n  return example_proto.SerializeToString()","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:25:22.916569Z","iopub.execute_input":"2021-10-17T03:25:22.916949Z","iopub.status.idle":"2021-10-17T03:25:22.928413Z","shell.execute_reply.started":"2021-10-17T03:25:22.916917Z","shell.execute_reply":"2021-10-17T03:25:22.927581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(df_80))\nprint(len(val_20))","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:25:22.930007Z","iopub.execute_input":"2021-10-17T03:25:22.930323Z","iopub.status.idle":"2021-10-17T03:25:22.943516Z","shell.execute_reply.started":"2021-10-17T03:25:22.930294Z","shell.execute_reply":"2021-10-17T03:25:22.942733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Augmentation","metadata":{}},{"cell_type":"code","source":"def get_mat(rotation, shear, height_zoom, width_zoom, height_shift, width_shift):\n    # returns 3x3 transformmatrix which transforms indicies\n\n    # CONVERT DEGREES TO RADIANS\n    rotation = math.pi * rotation / 180.\n    shear = math.pi * shear / 180.\n\n    def get_3x3_mat(lst):\n        return tf.reshape(tf.concat([lst], axis=0), [3, 3])\n\n    # ROTATION MATRIX\n    c1 = tf.math.cos(rotation)\n    s1 = tf.math.sin(rotation)\n    one = tf.constant([1], dtype='float32')\n    zero = tf.constant([0], dtype='float32')\n\n    rotation_matrix = get_3x3_mat([c1, s1, zero,\n                                   -s1, c1, zero,\n                                   zero, zero, one])\n    # SHEAR MATRIX\n    c2 = tf.math.cos(shear)\n    s2 = tf.math.sin(shear)\n\n    shear_matrix = get_3x3_mat([one, s2, zero,\n                                zero, c2, zero,\n                                zero, zero, one])\n    # ZOOM MATRIX\n    zoom_matrix = get_3x3_mat([one / height_zoom, zero, zero,\n                               zero, one / width_zoom, zero,\n                               zero, zero, one])\n    # SHIFT MATRIX\n    shift_matrix = get_3x3_mat([one, zero, height_shift,\n                                zero, one, width_shift,\n                                zero, zero, one])\n\n    return K.dot(K.dot(rotation_matrix, shear_matrix),\n                 K.dot(zoom_matrix, shift_matrix))\n\n\ndef transform(image, cfg):\n    # input image - is one image of size [dim,dim,3] not a batch of [b,dim,dim,3]\n    # output - image randomly rotated, sheared, zoomed, and shifted\n    DIM = cfg[\"read_size\"]\n    XDIM = DIM % 2  # fix for size 331\n\n    rot = cfg['rot'] * tf.random.normal([1], dtype='float32')\n    shr = cfg['shr'] * tf.random.normal([1], dtype='float32')\n    h_zoom = 1.0 + tf.random.normal([1], dtype='float32') / cfg['hzoom']\n    w_zoom = 1.0 + tf.random.normal([1], dtype='float32') / cfg['wzoom']\n    h_shift = cfg['hshift'] * tf.random.normal([1], dtype='float32')\n    w_shift = cfg['wshift'] * tf.random.normal([1], dtype='float32')\n\n    # GET TRANSFORMATION MATRIX\n    m = get_mat(rot, shr, h_zoom, w_zoom, h_shift, w_shift)\n\n    # LIST DESTINATION PIXEL INDICES\n    x = tf.repeat(tf.range(DIM // 2, -DIM // 2, -1), DIM)\n    y = tf.tile(tf.range(-DIM // 2, DIM // 2), [DIM])\n    z = tf.ones([DIM * DIM], dtype='int32')\n    idx = tf.stack([x, y, z])\n\n    # ROTATE DESTINATION PIXELS ONTO ORIGIN PIXELS\n    idx2 = K.dot(m, tf.cast(idx, dtype='float32'))\n    idx2 = K.cast(idx2, dtype='int32')\n    idx2 = K.clip(idx2, -DIM // 2 + XDIM + 1, DIM // 2)\n\n    # FIND ORIGIN PIXEL VALUES\n    idx3 = tf.stack([DIM // 2 - idx2[0,], DIM // 2 - 1 + idx2[1,]])\n    d = tf.gather_nd(image, tf.transpose(idx3))\n\n    return tf.reshape(d, [DIM, DIM, 3])\n\n\ndef prepare_image(img, cfg=None, augment=True):\n#     img = tf.image.decode_jpeg(img, channels=3)\n#     img = tf.image.resize(img, [cfg['read_size'], cfg['read_size']])\n#     img = tf.cast(img, tf.float32) / 255.0  # # Cast and normalize the image to [0,1]\n\n    if augment:\n        # Data augmentation\n        img = transform(img, cfg)\n#         img = tf.image.random_crop(img, [cfg['crop_size'], cfg['crop_size'], 3])\n        # Coarse dropout\n        # img = dropout(img, DIM=cfg['crop_size'], PROBABILITY=cfg['DROP_FREQ'], CT=cfg['DROP_CT'], SZ=cfg['DROP_SIZE'])\n        # Other augmentations\n        img = tf.image.random_flip_left_right(img)\n        img = tf.image.random_hue(img, 0.01)\n        img = tf.image.random_saturation(img, 0.7, 1.3)\n        img = tf.image.random_contrast(img, 0.8, 1.2)\n        img = tf.image.random_brightness(img, 0.1)\n        # Hair augmentation\n        # img = hair_aug_tf(img, augment=cfg['hair_augm'])\n    else:\n        img = tf.image.central_crop(img, cfg['crop_size'] / cfg['read_size'])\n\n#     img = tf.image.resize(img, [cfg['net_size'], cfg['net_size']])\n#     img = tf.reshape(img, [cfg['net_size'], cfg['net_size'], 3])\n    return img","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:25:22.944834Z","iopub.execute_input":"2021-10-17T03:25:22.945438Z","iopub.status.idle":"2021-10-17T03:25:22.977457Z","shell.execute_reply.started":"2021-10-17T03:25:22.945391Z","shell.execute_reply":"2021-10-17T03:25:22.976358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def write_tfrecord_aug(SIZE, df_w, tfrecord_name):\n    LEN = len(df_w)\n    NUM_AUG_IMAGES = CFG['num_aug']\n    SIZE = SIZE // NUM_AUG_IMAGES\n    CT = LEN // SIZE + int(LEN % SIZE != 0)\n    for j in range(CT):\n        print()\n        print('Writing TFRecord %i of %i...' % (j, CT))\n\n        CT2 = min(SIZE, LEN - j * SIZE)\n        with tf.io.TFRecordWriter(tfrecord_name + '%.2i-%i.tfrec' % (j, CT2 * NUM_AUG_IMAGES)) as writer:\n            for k in range(CT2):\n                img = cv2.imread(PATH + df_w.iloc[SIZE * j + k].image_name + '.jpg')\n                img = center_crop(img)\n                #                 img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR) # Fix incorrect colors\n                imgNoAug = cv2.imencode('.jpg', img, (cv2.IMWRITE_JPEG_QUALITY, 94))[1].tostring()\n                name = df_w.iloc[SIZE * j + k].image_name\n                row = df_w.loc[df_w.image_name == name]\n                example = serialize_example(\n                    imgNoAug, str.encode(name),\n                    row.patient_id.values[0],\n                    row.sex.values[0],\n                    row.age_approx.values[0],\n                    row.anatom_site_general_challenge.values[0],\n                    row.diagnosis.values[0],\n                    row.target.values[0])\n                writer.write(example)\n\n                if tfrecord_name == 'trainb' and row.target.values[0] == 1:\n                    for l in range(NUM_AUG_IMAGES - 1):\n                        imgAug = prepare_image(img, cfg=CFG).numpy()\n                        imgAug = cv2.imencode('.jpg', imgAug, (cv2.IMWRITE_JPEG_QUALITY, 94))[1].tostring()\n                        exname = name + \"_\" + str(l)\n                        example = serialize_example(\n                            imgAug, str.encode(exname),\n                            row.patient_id.values[0],\n                            row.sex.values[0],\n                            row.age_approx.values[0],\n                            row.anatom_site_general_challenge.values[0],\n                            row.diagnosis.values[0],\n                            row.target.values[0])\n                        writer.write(example)\n\n                if k*NUM_AUG_IMAGES % 100 == 0:\n                    print(k*NUM_AUG_IMAGES, ', ', end='')\n\n\ndef write_tfrecord(SIZE, df_w, tfrecord_name):\n    LEN = len(df_w)\n    CT = LEN // SIZE + int(LEN % SIZE != 0)\n    for j in range(CT):\n        print()\n        print('Writing TFRecord %i of %i...' % (j, CT))\n        CT2 = min(SIZE, LEN - j * SIZE)\n        with tf.io.TFRecordWriter(tfrecord_name + '%.2i-%i.tfrec' % (j, CT2)) as writer:\n            for k in range(CT2):\n                img = cv2.imread(PATH + df_w.iloc[SIZE * j + k].image_name + '.jpg')\n                img = center_crop(img)\n                #                 img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR) # Fix incorrect colors\n                imgNoAug = cv2.imencode('.jpg', img, (cv2.IMWRITE_JPEG_QUALITY, 94))[1].tostring()\n                name = df_w.iloc[SIZE * j + k].image_name\n                row = df_w.loc[df_w.image_name == name]\n                example = serialize_example(\n                    imgNoAug, str.encode(name),\n                    row.patient_id.values[0],\n                    row.sex.values[0],\n                    row.age_approx.values[0],\n                    row.anatom_site_general_challenge.values[0],\n                    row.diagnosis.values[0],\n                    row.target.values[0])\n                writer.write(example)\n\n                if k % 100 == 0:\n                    print(k, ', ', end='')","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:25:22.979421Z","iopub.execute_input":"2021-10-17T03:25:22.979810Z","iopub.status.idle":"2021-10-17T03:25:23.011151Z","shell.execute_reply.started":"2021-10-17T03:25:22.979767Z","shell.execute_reply":"2021-10-17T03:25:23.010218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SIZE = 1766\n# write_tfrecord(SIZE, df_80, 'train')\n\nSIZE = len(df_1_80) * CFG['num_aug'] // 15\nwrite_tfrecord_aug(SIZE, df_1_80, 'trainb')\n\n# SIZE = 1700\n# write_tfrecord(SIZE, df_0_80, 'traina')","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:25:23.013555Z","iopub.execute_input":"2021-10-17T03:25:23.013899Z","iopub.status.idle":"2021-10-17T03:26:47.646490Z","shell.execute_reply.started":"2021-10-17T03:25:23.013869Z","shell.execute_reply":"2021-10-17T03:26:47.643366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SIZE = 441\n# write_tfrecord(SIZE, val_20, 'val')","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:26:47.647419Z","iopub.status.idle":"2021-10-17T03:26:47.647860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ! ls -l","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:26:47.648976Z","iopub.status.idle":"2021-10-17T03:26:47.649376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(IMG_SIZE[0])\n# print(IMG_SIZE[1])","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:26:47.650170Z","iopub.status.idle":"2021-10-17T03:26:47.650582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# zipFileName = f'melanoma-{IMG_SIZE[0]}x{IMG_SIZE[1]}.zip'\n# print(zipFileName)\n# !zip -r zipFileName ./","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:26:47.651755Z","iopub.status.idle":"2021-10-17T03:26:47.652210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Write TFRecords - Test","metadata":{}},{"cell_type":"code","source":"# def serialize_example2(feature0, feature1, feature2, feature3, feature4, feature5): \n#   feature = {\n#       'image': _bytes_feature(feature0),\n#       'image_name': _bytes_feature(feature1),\n#       'patient_id': _int64_feature(feature2),\n#       'sex': _int64_feature(feature3),\n#       'age_approx': _int64_feature(feature4),\n#       'anatom_site_general_challenge': _int64_feature(feature5),\n#   }\n#   example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n#   return example_proto.SerializeToString()","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:26:47.653154Z","iopub.status.idle":"2021-10-17T03:26:47.653597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SIZE = 687\n# CT = len(IMGS2)//SIZE + int(len(IMGS2)%SIZE!=0)\n# for j in range(CT):\n#     print(); print('Writing TFRecord %i of %i...'%(j,CT))\n#     CT2 = min(SIZE,len(IMGS2)-j*SIZE)\n#     with tf.io.TFRecordWriter('test%.2i-%i.tfrec'%(j,CT2)) as writer:\n#         for k in range(CT2):\n#             img = cv2.imread(PATH2+IMGS2[SIZE*j+k])\n#             img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR) # Fix incorrect colors\n#             img = cv2.imencode('.jpg', img, (cv2.IMWRITE_JPEG_QUALITY, 94))[1].tostring()\n#             name = IMGS2[SIZE*j+k].split('.')[0]\n#             row = test.loc[test.image_name==name]\n#             example = serialize_example2(\n#                 img, str.encode(name),\n#                 row.patient_id.values[0],\n#                 row.sex.values[0],\n#                 row.age_approx.values[0],                        \n#                 row.anatom_site_general_challenge.values[0])\n#             writer.write(example)\n#             if k%100==0: print(k,', ',end='')","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:26:47.654408Z","iopub.status.idle":"2021-10-17T03:26:47.654889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Verify TFRecords\nWe will verify the TFRecords we just made by using code from the Flower Comp starter notebook [here][1] to display the TFRecords below.\n\n[1]: https://www.kaggle.com/mgornergoogle/getting-started-with-100-flowers-on-tpu","metadata":{}},{"cell_type":"code","source":"# numpy and matplotlib defaults\nnp.set_printoptions(threshold=15, linewidth=80)\nCLASSES = [0,1]\n\ndef batch_to_numpy_images_and_labels(data):\n    images, labels = data\n    numpy_images = images.numpy()\n    numpy_labels = labels.numpy()\n    #if numpy_labels.dtype == object: # binary string in this case, these are image ID strings\n    #    numpy_labels = [None for _ in enumerate(numpy_images)]\n    # If no labels, only image IDs, return None for labels (this is the case for test data)\n    return numpy_images, numpy_labels\n\ndef title_from_label_and_target(label, correct_label):\n    if correct_label is None:\n        return CLASSES[label], True\n    correct = (label == correct_label)\n    return \"{} [{}{}{}]\".format(CLASSES[label], 'OK' if correct else 'NO', u\"\\u2192\" if not correct else '',\n                                CLASSES[correct_label] if not correct else ''), correct\n\ndef display_one_flower(image, title, subplot, red=False, titlesize=16):\n    plt.subplot(*subplot)\n    plt.axis('off')\n    plt.imshow(image)\n    if len(title) > 0:\n        plt.title(title, fontsize=int(titlesize) if not red else int(titlesize/1.2), color='red' if red else 'black', fontdict={'verticalalignment':'center'}, pad=int(titlesize/1.5))\n    return (subplot[0], subplot[1], subplot[2]+1)\n    \ndef display_batch_of_images(databatch, predictions=None):\n    \"\"\"This will work with:\n    display_batch_of_images(images)\n    display_batch_of_images(images, predictions)\n    display_batch_of_images((images, labels))\n    display_batch_of_images((images, labels), predictions)\n    \"\"\"\n    # data\n    images, labels = batch_to_numpy_images_and_labels(databatch)\n    if labels is None:\n        labels = [None for _ in enumerate(images)]\n        \n    # auto-squaring: this will drop data that does not fit into square or square-ish rectangle\n    rows = int(math.sqrt(len(images)))\n    cols = len(images)//rows\n        \n    # size and spacing\n    FIGSIZE = 13.0\n    SPACING = 0.1\n    subplot=(rows,cols,1)\n    if rows < cols:\n        plt.figure(figsize=(FIGSIZE,FIGSIZE/cols*rows))\n    else:\n        plt.figure(figsize=(FIGSIZE/rows*cols,FIGSIZE))\n    \n    # display\n    for i, (image, label) in enumerate(zip(images[:rows*cols], labels[:rows*cols])):\n        title = label\n        correct = True\n        if predictions is not None:\n            title, correct = title_from_label_and_target(predictions[i], label)\n        dynamic_titlesize = FIGSIZE*SPACING/max(rows,cols)*40+3 # magic formula tested to work from 1x1 to 10x10 images\n        subplot = display_one_flower(image, title, subplot, not correct, titlesize=dynamic_titlesize)\n    \n    #layout\n    plt.tight_layout()\n    if label is None and predictions is None:\n        plt.subplots_adjust(wspace=0, hspace=0)\n    else:\n        plt.subplots_adjust(wspace=SPACING, hspace=SPACING)\n    plt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-10-17T03:26:47.656045Z","iopub.status.idle":"2021-10-17T03:26:47.656531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n#     channels = tf.unstack (image, axis=-1)\n#     image = tf.stack([channels[2], channels[1], channels[0]], axis=-1)\n    image = tf.cast(image, tf.float32) / 255.0  # convert image to floats in [0, 1] range\n    image = tf.reshape(image, [*IMAGE_SIZE, 3]) # explicit size needed for TPU\n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string), # tf.string means bytestring\n        \"image_name\": tf.io.FixedLenFeature([], tf.string),  # shape [] means single element\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    label = example['image_name']\n    return image, label # returns a dataset of (image, label) pairs\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    # Read from TFRecords. For optimal performance, reading from multiple files at once and\n    # disregarding data order. Order does not matter since we will be shuffling the data anyway.\n\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False # disable order, increase speed\n\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO) # automatically interleaves reads from multiple files\n    dataset = dataset.with_options(ignore_order) # uses data as soon as it streams in, rather than in its original order\n    dataset = dataset.map(read_labeled_tfrecord)\n    # returns a dataset of (image, label) pairs if labeled=True or (image, id) pairs if labeled=False\n    return dataset\n\ndef get_training_dataset():\n    dataset = load_dataset(TRAINING_FILENAMES, labeled=True)\n#     dataset = dataset.repeat() # the training dataset must repeat for several epochs\n#     dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO) # prefetch next batch while training (autotune prefetch buffer size)\n    return dataset\n\ndef count_data_items(filenames):\n    # the number of data items is written in the name of the .tfrec files, i.e. flowers00-230.tfrec = 230 data items\n    n = [int(re.compile(r\"-([0-9]*)\\.\").search(filename).group(1)) for filename in filenames]\n    return np.sum(n)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-10-17T03:26:47.657589Z","iopub.status.idle":"2021-10-17T03:26:47.658064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# INITIALIZE VARIABLES\nIMAGE_SIZE= [256,256]; BATCH_SIZE = 32\nAUTO = tf.data.experimental.AUTOTUNE\nTRAINING_FILENAMES = tf.io.gfile.glob('train*.tfrec')\nprint('There are %i train images'%count_data_items(TRAINING_FILENAMES))","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:26:47.658932Z","iopub.status.idle":"2021-10-17T03:26:47.659336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# DISPLAY TRAIN IMAGES\ntraining_dataset = get_training_dataset()\ntraining_dataset = training_dataset.unbatch().batch(600)\ntrain_batch = iter(training_dataset)\n\ndisplay_batch_of_images(next(train_batch))","metadata":{"execution":{"iopub.status.busy":"2021-10-17T03:26:47.660258Z","iopub.status.idle":"2021-10-17T03:26:47.660664Z"},"trusted":true},"execution_count":null,"outputs":[]}]}