{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"print(\"\\n... IMPORTS STARTING ...\\n\")\n\nprint(\"\\n\\tVERSION INFORMATION\")\n\n# Machine Learning and Data Science Imports\nimport tensorflow as tf; print(f\"\\t\\t– TENSORFLOW VERSION: {tf.__version__}\");\nimport tensorflow_hub as tfhub; print(f\"\\t\\t– TENSORFLOW HUB VERSION: {tfhub.__version__}\");\nimport tensorflow_addons as tfa; print(f\"\\t\\t– TENSORFLOW ADDONS VERSION: {tfa.__version__}\");\nimport pandas as pd; pd.options.mode.chained_assignment = None;\nimport numpy as np; print(f\"\\t\\t– NUMPY VERSION: {np.__version__}\");\nimport sklearn; print(f\"\\t\\t– SKLEARN VERSION: {sklearn.__version__}\");\nfrom sklearn.preprocessing import RobustScaler, PolynomialFeatures\nfrom sklearn.model_selection import GroupKFold, StratifiedKFold\n\n# Built In Imports\nfrom kaggle_datasets import KaggleDatasets\nfrom collections import Counter\nfrom datetime import datetime\nfrom zipfile import ZipFile\nfrom glob import glob\nimport warnings\nimport requests\nimport hashlib\nimport imageio\nimport IPython\nimport sklearn\nimport urllib\nimport zipfile\nimport pickle\nimport random\nimport shutil\nimport string\nimport json\nimport math\nimport time\nimport gzip\nimport ast\nimport sys\nimport io\nimport os\nimport gc\nimport re\n\n# Visualization Imports\nfrom matplotlib.colors import ListedColormap\nfrom matplotlib.patches import Rectangle\nimport matplotlib.patches as patches\nimport plotly.graph_objects as go\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm; tqdm.pandas();\nimport plotly.express as px\nimport tifffile as tif\nimport seaborn as sns\nfrom PIL import Image, ImageEnhance; Image.MAX_IMAGE_PIXELS = 5_000_000_000;\nimport matplotlib; print(f\"\\t\\t– MATPLOTLIB VERSION: {matplotlib.__version__}\");\nfrom matplotlib import animation, rc; rc('animation', html='jshtml')\nimport plotly\nimport PIL\nimport cv2\n\nimport plotly.io as pio\nprint(pio.renderers)\n\n# Basic helpers and seeding\ndef seed_it_all(seed=7):\n    \"\"\" Attempt to be Reproducible \"\"\"\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\nseed_it_all()\n\ndef flatten_l_o_l(nested_list):\n    \"\"\" Flatten a list of lists \"\"\"\n    return [item for sublist in nested_list for item in sublist]\n\ndef load_json_to_dict(json_path):\n    \"\"\" tbd \"\"\"\n    with open(json_path) as json_file:\n        data = json.load(json_file)\n    return data\n    \nprint(\"\\n\\n... IMPORTS COMPLETE ...\\n\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-30T19:02:39.721501Z","iopub.execute_input":"2022-07-30T19:02:39.722148Z","iopub.status.idle":"2022-07-30T19:02:49.338826Z","shell.execute_reply.started":"2022-07-30T19:02:39.722084Z","shell.execute_reply":"2022-07-30T19:02:49.337559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_coco2017(dir_path, style=\"captions\", combine_caption_rows=False, shuffle_train=True, max_train=60_000, max_val=6_000):\n    \n    annot_dir = os.path.join(dir_path, \"annotations\")\n    \n    if style==\"instances\":\n        coco_instances_train = load_json_to_dict(os.path.join(annot_dir, \"instances_train2017.json\"))    \n        COCO_I2S = {c_map[\"id\"]:c_map[\"name\"] for c_map in coco_instances_train[\"categories\"]}\n        COCO_S2I = {v:k for k,v in COCO_I2S.items()}\n\n        train_license_map = {_l[\"id\"]:_l[\"name\"] for _l in coco_instances_train[\"licenses\"]}\n        train_images, train_annots = coco_instances_train[\"images\"], coco_instances_train[\"annotations\"]\n        train_df = pd.merge(pd.DataFrame(train_annots), pd.DataFrame(train_images), left_on=\"image_id\", right_on=\"id\").drop(columns=[\"id_x\", \"id_y\"]).sort_values(by=\"image_id\")\n        train_df[\"license\"] = train_df[\"license\"].map(train_license_map)\n        train_df[\"img_path\"] = os.path.join(dir_path, \"train2017\")+\"/\"+train_df[\"file_name\"]\n        train_df[\"str_category\"] = train_df[\"category_id\"].map(COCO_I2S)\n        train_df = train_df[['image_id', 'file_name', 'img_path', 'category_id',\n                             'str_category', 'segmentation', 'area', 'iscrowd', 'bbox', 'license', \n                             'coco_url', 'height', 'width', 'date_captured','flickr_url']].reset_index(drop=True)\n\n        coco_instances_val = load_json_to_dict(os.path.join(annot_dir, \"instances_val2017.json\"))\n        val_license_map = {_l[\"id\"]:_l[\"name\"] for _l in coco_instances_val[\"licenses\"]}\n        val_images, val_annots = coco_instances_val[\"images\"], coco_instances_val[\"annotations\"]\n        val_df = pd.merge(pd.DataFrame(val_annots), pd.DataFrame(val_images), left_on=\"image_id\", right_on=\"id\").drop(columns=[\"id_x\", \"id_y\"]).sort_values(by=\"image_id\")\n        val_df[\"license\"] = val_df[\"license\"].map(val_license_map)\n        val_df[\"img_path\"] = os.path.join(dir_path, \"val2017\")+\"/\"+val_df[\"file_name\"]\n        val_df[\"str_category\"] = val_df[\"category_id\"].map(COCO_I2S)\n        val_df = val_df[['image_id', 'file_name', 'img_path', 'category_id', \n                         'str_category', 'segmentation', 'area', 'iscrowd', 'bbox', 'license', \n                         'coco_url', 'height', 'width', 'date_captured','flickr_url']].reset_index(drop=True)\n    \n    else:\n        coco_captions_train = load_json_to_dict(os.path.join(annot_dir, \"captions_train2017.json\"))    \n        train_license_map = {_l[\"id\"]:_l[\"name\"] for _l in coco_captions_train[\"licenses\"]}\n        train_images, train_annots = coco_captions_train[\"images\"], coco_captions_train[\"annotations\"]\n        train_df = pd.merge(pd.DataFrame(train_annots), pd.DataFrame(train_images), left_on=\"image_id\", right_on=\"id\").drop(columns=[\"id_x\", \"id_y\"]).sort_values(by=\"image_id\")\n        train_df[\"license\"] = train_df[\"license\"].map(train_license_map)\n        train_df[\"img_path\"] = os.path.join(dir_path, \"train2017\")+\"/\"+train_df[\"file_name\"]\n        if combine_caption_rows:\n            train_df = train_df.merge(\n                pd.DataFrame(train_df.groupby(\"image_id\")\\\n                                     .caption.apply(list)\\\n                                     .apply(lambda x: pd.Series(x[:5]))\\\n                                     .add_prefix(\"caption_\"))\\\n                                     .reset_index(), \n                on=\"image_id\", how=\"right\").drop_duplicates(subset=[\"image_id\",]).drop(columns=[\"caption\"])\n        \n        if shuffle_train:\n            train_df = train_df.sample(len(train_df)).reset_index(drop=True)\n        \n        if max_train is not None:\n            train_df = train_df.head(max_train)\n            \n        coco_captions_val = load_json_to_dict(os.path.join(annot_dir, \"captions_val2017.json\"))    \n        val_license_map = {_l[\"id\"]:_l[\"name\"] for _l in coco_captions_val[\"licenses\"]}\n        val_images, val_annots = coco_captions_val[\"images\"], coco_captions_val[\"annotations\"]\n        val_df = pd.merge(pd.DataFrame(val_annots), pd.DataFrame(val_images), left_on=\"image_id\", right_on=\"id\").drop(columns=[\"id_x\", \"id_y\"]).sort_values(by=\"image_id\")\n        val_df[\"license\"] = val_df[\"license\"].map(val_license_map)\n        val_df[\"img_path\"] = os.path.join(dir_path, \"val2017\")+\"/\"+val_df[\"file_name\"]\n        \n        if combine_caption_rows:\n            val_df = val_df.merge(\n                pd.DataFrame(val_df.groupby(\"image_id\")\\\n                                   .caption.apply(list)\\\n                                   .apply(lambda x: pd.Series(x[:5]))\\\n                                   .add_prefix(\"caption_\"))\\\n                                   .reset_index(), \n                on=\"image_id\", how=\"right\").drop_duplicates(subset=[\"image_id\",]).drop(columns=[\"caption\"])        \n        \n        if max_val is not None:\n            val_df = val_df.sample(max_val).reset_index(drop=True)\n        \n    # Get test dataframe images\n    test_df = pd.DataFrame({\"img_path\":glob(os.path.join(dir_path, \"test2017\", \"*.jpg\"))})\n    test_df.insert(0, \"image_id\", test_df[\"img_path\"].apply(lambda x: x.rsplit(\"/\", 1)[-1][:-4]))\n\n    # Cleanup\n    gc.collect(); gc.collect(); gc.collect();\n    \n    return train_df.reset_index(drop=True), val_df.reset_index(drop=True), test_df.reset_index(drop=True)\n    \ntrain_df, val_df, test_df = load_coco2017(\"/kaggle/input/coco-2017-dataset/coco2017\")\nN_TRAIN, N_VAL, N_TEST = len(train_df), len(val_df), len(test_df)\n\nprint(f\"\\n\\n... COCO TRAINING DATAFRAME ({N_TRAIN} EXAMPLES OVER {train_df.image_id.nunique()} IMAGES) ...\\n\")\ndisplay(train_df)\n\nprint(f\"\\n\\n\\n... COCO VALIDATION DATAFRAME ({N_VAL} EXAMPLES OVER {val_df.image_id.nunique()} IMAGES) ...\\n\")\ndisplay(val_df)\n\nprint(f\"\\n\\n\\n... COCO VALIDATION DATAFRAME ({N_TEST} EXAMPLES OVER {test_df.image_id.nunique()} IMAGES) ...\\n\")\ndisplay(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T19:02:49.345737Z","iopub.execute_input":"2022-07-30T19:02:49.346534Z","iopub.status.idle":"2022-07-30T19:03:03.490010Z","shell.execute_reply.started":"2022-07-30T19:02:49.346490Z","shell.execute_reply":"2022-07-30T19:03:03.488818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lowercase everything!\ntrain_df.caption = train_df.caption.str.lower()\nval_df.caption = val_df.caption.str.lower()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T19:03:03.491671Z","iopub.execute_input":"2022-07-30T19:03:03.491998Z","iopub.status.idle":"2022-07-30T19:03:03.618946Z","shell.execute_reply.started":"2022-07-30T19:03:03.491970Z","shell.execute_reply":"2022-07-30T19:03:03.617760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tf_load_image(img_path, reshape_to=(512,512)):\n    return tf.cast(tf.image.resize_with_pad(tf.image.decode_image(tf.io.read_file(img_path), channels=3, expand_animations=False), reshape_to[1], reshape_to[0]), tf.uint8)\n\ndef _bytes_feature(value, is_list=False):\n    \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n    if isinstance(value, type(tf.constant(0))):\n        value = value.numpy() # BytesList won't unpack a string from an EagerTensor.\n    \n    if not is_list:\n        value = [value]\n    \n    return tf.train.Feature(bytes_list=tf.train.BytesList(value=value))\n\ndef _float_feature(value, is_list=False):\n    \"\"\"Returns a float_list from a float / double.\"\"\"\n        \n    if not is_list:\n        value = [value]\n        \n    return tf.train.Feature(float_list=tf.train.FloatList(value=value))\n\ndef _int64_feature(value, is_list=False):\n    \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n        \n    if not is_list:\n        value = [value]\n        \n    return tf.train.Feature(int64_list=tf.train.Int64List(value=value))\n\ndef create_tf_dataset(paths, captions):\n    ds = tf.data.Dataset.from_tensor_slices((paths, captions))\n    ds = ds.map(lambda x,y: (tf_load_image(x), y))\n    return ds\n    \ndef serialize_raw(image, caption):\n    \"\"\"\n    Creates a tf.Example message ready to be written to a file from 2 (img and caption).\n\n    Args:\n        image (tf.constant): TBD\n        caption (str): TBD\n    \n    Returns:\n        A tf.Example Message ready to be written to file\n    \"\"\"\n    \n    # Create a dictionary mapping the feature name to the \n    # tf.Example-compatible data type.\n    features = {\n        'image': _bytes_feature(tf.io.encode_png(image), is_list=False),\n        'caption': _bytes_feature(caption, is_list=False)\n    }\n        \n    # Create a Features message using tf.train.Example.\n    example_proto = tf.train.Example(features=tf.train.Features(feature=features))\n    return example_proto.SerializeToString()\n\ndef show_all_ex_captions(_df, _image_id=None):\n    if _image_id is None: _image_id = _df[\"image_id\"].sample(1).values[0]\n    _sub_df = _df[_df[\"image_id\"]==_image_id]\n    _img_path = _sub_df.img_path.values[0]\n    _annotations = list(_sub_df.caption.values)\n    \n    plt.figure(figsize=(13,15))    \n    plt.imshow(tf_load_image(_img_path))\n    plt.title(\"'\"+\"'\\n'\".join(_annotations)+\"'\", fontweight=\"bold\")\n    plt.axis(False)\n    plt.tight_layout()\n    plt.show()\n    \nfor i in range(2):\n    show_all_ex_captions(val_df)\n\n# Create our datasets\ntrain_ds = create_tf_dataset(train_df.img_path.values, train_df.caption.values)\nval_ds = create_tf_dataset(val_df.img_path.values, val_df.caption.values)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T19:03:03.623087Z","iopub.execute_input":"2022-07-30T19:03:03.623475Z","iopub.status.idle":"2022-07-30T19:03:05.649357Z","shell.execute_reply.started":"2022-07-30T19:03:03.623441Z","shell.execute_reply":"2022-07-30T19:03:05.648219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def write_tfrecords(ds, n_ex, n_ex_per_rec=1500, serialize_fn=serialize_raw, out_dir=\"/kaggle/working/512x512\", ds_type=\"train\", fold=None):\n    \"\"\"\"\"\"\n    n_recs = int(np.ceil(n_ex/n_ex_per_rec))\n    \n    # Make dataset iterable\n    ds = iter(ds)\n        \n    out_dir = os.path.join(out_dir, ds_type)\n    # Create folder\n    if not os.path.isdir(out_dir):\n        os.makedirs(out_dir, exist_ok=True)\n        \n    # Create tfrecords\n    for i in tqdm(range(n_recs), total=n_recs):\n        print(f\"\\n... Writing {ds_type.title()} TFRecord {i+1} of {n_recs} For Fold #{fold}...\\n\")\n        if fold is not None:\n            tfrec_path = os.path.join(out_dir, f\"{ds_type}_fold_{fold}__{(i+1):02}_{n_recs:02}.tfrec\")\n        else:\n            tfrec_path = os.path.join(out_dir, f\"{ds_type}__{(i+1):02}_{n_recs:02}.tfrec\")\n        \n        # This makes the tfrecord\n        with tf.io.TFRecordWriter(tfrec_path) as writer:\n            for ex in tqdm(range(n_ex_per_rec), total=n_ex_per_rec):\n                try:\n                    example = serialize_fn(*next(ds))\n                    writer.write(example)\n                except:\n                    break\n\nwrite_tfrecords(train_ds, n_ex=N_TRAIN, n_ex_per_rec=300, ds_type=\"train\")\nwrite_tfrecords(val_ds, n_ex=N_VAL, n_ex_per_rec=300, ds_type=\"val\")","metadata":{"execution":{"iopub.status.busy":"2022-07-30T19:03:05.650794Z","iopub.execute_input":"2022-07-30T19:03:05.651125Z"},"trusted":true},"execution_count":null,"outputs":[]}]}