{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, cv2\nimport tensorflow as tf\nimport json\nimport multiprocessing as mp\n#from multiprocessing import Pool, freeze_support\n\ndef image_feature(value):\n    \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n    return tf.train.Feature(\n        bytes_list=tf.train.BytesList(value=[tf.io.encode_jpeg(value).numpy()])\n    )\n\n\ndef bytes_feature(value):\n    \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n    return tf.train.Feature(bytes_list=tf.train.BytesList(value=[value.encode()]))\n\n\ndef float_feature(value):\n    \"\"\"Returns a float_list from a float / double.\"\"\"\n    return tf.train.Feature(float_list=tf.train.FloatList(value=[value]))\n\n\ndef int64_feature(value):\n    \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n    return tf.train.Feature(int64_list=tf.train.Int64List(value=[value]))\n\n\ndef float_feature_list(value):\n    \"\"\"Returns a list of float_list from a float / double.\"\"\"\n    return tf.train.Feature(float_list=tf.train.FloatList(value=value))\n\n\ndef create_example(image, img_idx, label):\n    feature = {\n        \"image\": image_feature(image),\n        \"image_idx\": int64_feature(img_idx),\n        \"label\" : int64_feature(label)\n    }\n    #feature = {\n    #    \"image\": image_feature(image),\n    #    \"path\": bytes_feature(path),\n    #    \"area\": float_feature(example[\"area\"]),\n    #    \"bbox\": float_feature_list(example[\"bbox\"]),\n    #    \"category_id\": int64_feature(example[\"category_id\"]),\n    #    \"id\": int64_feature(example[\"id\"]),\n    #    \"image_id\": int64_feature(example[\"image_id\"]),\n    #}\n    return tf.train.Example(features=tf.train.Features(feature=feature))\n\ndef create_unlabeled_example(image, image_idx):\n    feature = {\n        \"image\": image_feature(image),\n        \"image_idx\": int64_feature(image_idx),\n        \n    }\n    #feature = {\n    #    \"image\": image_feature(image),\n    #    \"path\": bytes_feature(path),\n    #    \"area\": float_feature(example[\"area\"]),\n    #    \"bbox\": float_feature_list(example[\"bbox\"]),\n    #    \"category_id\": int64_feature(example[\"category_id\"]),\n    #    \"id\": int64_feature(example[\"id\"]),\n    #    \"image_id\": int64_feature(example[\"image_id\"]),\n    #}\n    return tf.train.Example(features=tf.train.Features(feature=feature))\n\n\n\n\ndef parse_tfrecord_fn(example):\n    feature_description = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"image_idx\": tf.io.FixedLenFeature([], tf.int64),\n        \"label\": tf.io.FixedLenFeature([], tf.int64),\n        \n        \n    }\n    example = tf.io.parse_single_example(example, feature_description)\n    example[\"image\"] = tf.io.decode_jpeg(example[\"image\"], channels=3)\n    return example\n\ndef make_tfrecords_train(tfrecords_dir,tfrec_num,samples,images,imsize=600):\n    #= args[0],args[1],args[2],args[3]\n    print( tfrecords_dir + \"/file_%.5i-%i.tfrec\" % (tfrec_num, len(samples)))\n    with tf.io.TFRecordWriter(\n        tfrecords_dir + \"/file_%.5i-%i.tfrec\" % (tfrec_num, len(samples))\n    ) as writer:\n        for sample in samples:\n            for img in images:\n                if sample['id']==img['id']:\n                    image_path = f\"{parent}/train/{img['file_name']}\"\n                    image = tf.io.decode_jpeg(tf.io.read_file(image_path))\n                    image = tf.image.resize(image,(imsize,imsize))\n                    image = tf.cast(image, tf.uint8)\n                    example = create_example(image, sample['image_id'], sample['category_id'])\n                    writer.write(example.SerializeToString())\n\ndef make_tfrecords_test(tfrecords_dir,tfrec_num,samples,imsize=600):\n        \n        with tf.io.TFRecordWriter(\n            tfrecords_dir + \"/file_%.2i-%i.tfrec\" % (tfrec_num, len(samples))\n        ) as writer:\n            for img in samples:\n                image_path = f\"{parent}/test/{img['file_name']}\"\n                image = tf.io.decode_jpeg(tf.io.read_file(image_path))\n                image = tf.image.resize(image,(imsize,imsize))\n                image = tf.cast(image, tf.uint8)\n                example = create_unlabeled_example(image, img['id'])\n                writer.write(example.SerializeToString())\n\n            \n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if __name__ == '__main__':\n\n    parent = 'Herbarium_2021_FGVC8'\n    file = 'Herbarium_2021_FGVC8/train/metadata.json'\n    with open(file,'r') as f : \n        data = json.load(f)\n    file = 'Herbarium_2021_FGVC8/test/metadata.json'\n    with open(file,'r') as f : \n        test_data = json.load(f)\n    annotations = data['annotations']\n    images = data['images']\n    tfrecords_dir = 'Herbarium_2021_FGVC8/tfrecords/train_4'\n    num_samples = 10000\n    num_tfrecords = len(annotations) // num_samples\n    if len(annotations) % num_samples:\n        num_tfrecords += 1  # add one record if there are any remaining samples\n\n    if not os.path.exists(tfrecords_dir):\n        os.makedirs(tfrecords_dir)  # creating TFRecords output folder\n    print('starting mp')\n\n    imsize = 1600\n    #pool = mp.Pool(processes=4)\n    workers = 20\n    #mp.freeze_support()\n    prcs = []\n    for tfrec_num in range(num_tfrecords):\n        samples = annotations[(tfrec_num * num_samples) : ((tfrec_num + 1) * num_samples)]\n        #pool.map(make_tfrecords,args=(tfrecords_dir,tfrec_num,samples,images))\n\n        p = mp.Process(target=make_tfrecords_train,args=(tfrecords_dir,tfrec_num,samples,images,imsize))\n        p.start()\n        prcs.append(p)\n        if tfrec_num%workers==0 and tfrec_num>0: \n            for p in prcs:\n                p.join()\n            prcs =[]\n        \n    tfrecords_dir = 'Herbarium_2021_FGVC8/tfrecords/test_4'\n    annotations = test_data['images']\n    num_samples = 10000\n    num_tfrecords = len(annotations) // num_samples\n    \n    if len(annotations) % num_samples:\n        num_tfrecords += 1  # add one record if there are any remaining samples\n\n    if not os.path.exists(tfrecords_dir):\n        os.makedirs(tfrecords_dir)  # creating TFRecords output folder  \n\n    prcs =[]\n    for tfrec_num in range(num_tfrecords):\n        samples = annotations[(tfrec_num * num_samples) : ((tfrec_num + 1) * num_samples)]\n        p = mp.Process(target=make_tfrecords_test,args=(tfrecords_dir,tfrec_num,samples,imsize))\n        p.start()\n        \n        prcs.append(p)\n        if tfrec_num%workers==0 and tfrec_num>0: \n            for p in prcs:\n                p.join()\n            prcs =[]\n","metadata":{},"execution_count":null,"outputs":[]}]}