{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, math, glob, re\nimport numpy as np\nimport pandas as pd\nimport cv2\nfrom PIL import Image\n\nimport matplotlib.pyplot as plt\n\nfrom kaggle_datasets import KaggleDatasets\n\nfrom sklearn.model_selection import StratifiedKFold, KFold\n\nfrom tqdm.auto import tqdm\nfrom datetime import datetime\n\nfrom random import shuffle\nimport tensorflow as tf\n\nfrom collections import Counter","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-19T09:40:45.481834Z","iopub.execute_input":"2022-08-19T09:40:45.482523Z","iopub.status.idle":"2022-08-19T09:40:53.048732Z","shell.execute_reply.started":"2022-08-19T09:40:45.482396Z","shell.execute_reply":"2022-08-19T09:40:53.047390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!df -h","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:40:53.050829Z","iopub.execute_input":"2022-08-19T09:40:53.051291Z","iopub.status.idle":"2022-08-19T09:40:54.208886Z","shell.execute_reply.started":"2022-08-19T09:40:53.051237Z","shell.execute_reply":"2022-08-19T09:40:54.207561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_SIZE = 512\nNUM_IMAGES_PER_CLASS = 50\nN_GROUPS = 32        # number of tfrecord files\nOFFSET_LABEL = 1000 + 9691 # label offset (0-999: imagenet1k[1000labels], 1000-10690: products10k[9691labels]) \nprint(OFFSET_LABEL)\n\nCREATE_DATASET = True\n\nUSE_N_CLASSES = 7000\n\nUSER_NAME = \"motono0223\"\nDATASET_NAME = f'guie-glr2021mini-tfrecords-label-{OFFSET_LABEL}-{OFFSET_LABEL+USE_N_CLASSES-1}'\nprint(DATASET_NAME)\n\nlocal_directory = \"../input/landmark-recognition-2021\"\nlocal_label_path = local_directory+\"/train.csv\"","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:52:30.285804Z","iopub.execute_input":"2022-08-19T09:52:30.286162Z","iopub.status.idle":"2022-08-19T09:52:30.294081Z","shell.execute_reply.started":"2022-08-19T09:52:30.286130Z","shell.execute_reply":"2022-08-19T09:52:30.292895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read Data","metadata":{}},{"cell_type":"code","source":"# train data\ntrain_df = pd.read_csv(local_label_path)\ntrain_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:57:00.399759Z","iopub.execute_input":"2022-08-19T09:57:00.400138Z","iopub.status.idle":"2022-08-19T09:57:01.420928Z","shell.execute_reply.started":"2022-08-19T09:57:00.400103Z","shell.execute_reply":"2022-08-19T09:57:01.419588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter = Counter( train_df[\"landmark_id\"].values )\nplt.plot( sorted( [ v for (k,v) in counter.items()] ) )\nplt.yscale(\"log\")\nplt.show()\n\nplt.plot( sorted( [ v for (k,v) in counter.items()] )[-9000:] )\nplt.yscale(\"log\")\nplt.show()\n\nprint( min(sorted( [ v for (k,v) in counter.items()] )[-7000:]) )\n\ncounter_sorted = sorted(counter.items(), key=lambda x:x[1])[::-1]\nvalid_landmark_ids = [ k for i, (k, v) in enumerate(counter_sorted)  if i < 7000 ]\nprint( len(valid_landmark_ids) )","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:57:01.423278Z","iopub.execute_input":"2022-08-19T09:57:01.423721Z","iopub.status.idle":"2022-08-19T09:57:02.549250Z","shell.execute_reply.started":"2022-08-19T09:57:01.423671Z","shell.execute_reply":"2022-08-19T09:57:02.548227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter_sorted[0], counter_sorted[-1], counter_sorted[6999]","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:57:41.899321Z","iopub.execute_input":"2022-08-19T09:57:41.899709Z","iopub.status.idle":"2022-08-19T09:57:41.907834Z","shell.execute_reply.started":"2022-08-19T09:57:41.899674Z","shell.execute_reply":"2022-08-19T09:57:41.906589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sampling\ntrain_df = train_df[ train_df[\"landmark_id\"].isin(valid_landmark_ids) ]\ntrain_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:57:46.344607Z","iopub.execute_input":"2022-08-19T09:57:46.345294Z","iopub.status.idle":"2022-08-19T09:57:46.392387Z","shell.execute_reply.started":"2022-08-19T09:57:46.345257Z","shell.execute_reply":"2022-08-19T09:57:46.391576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_landmark_id = train_df[\"landmark_id\"].unique()\nprint(\"num_label =\", unique_landmark_id.shape[0])\n\nnew_label = { k:v + OFFSET_LABEL for (k,v) in zip( unique_landmark_id, range(unique_landmark_id.shape[0]) ) }\n\ntrain_df[\"label\"] = train_df[\"landmark_id\"].apply( lambda x: new_label[x] )\ntrain_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:57:46.710481Z","iopub.execute_input":"2022-08-19T09:57:46.711112Z","iopub.status.idle":"2022-08-19T09:57:47.157537Z","shell.execute_reply.started":"2022-08-19T09:57:46.711074Z","shell.execute_reply":"2022-08-19T09:57:47.156554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print( \"label (min, max) =\", train_df[\"label\"].min(), train_df[\"label\"].max() )","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:57:48.423868Z","iopub.execute_input":"2022-08-19T09:57:48.424378Z","iopub.status.idle":"2022-08-19T09:57:48.431979Z","shell.execute_reply.started":"2022-08-19T09:57:48.424330Z","shell.execute_reply":"2022-08-19T09:57:48.431107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter = Counter( train_df[\"label\"].values )\nplt.plot( sorted( [ v for (k,v) in counter.items()] ) )\nplt.yscale(\"log\")\nplt.show()\n\nprint( min(sorted( [ v for (k,v) in counter.items()] )) )","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:57:50.137653Z","iopub.execute_input":"2022-08-19T09:57:50.138013Z","iopub.status.idle":"2022-08-19T09:57:50.594934Z","shell.execute_reply.started":"2022-08-19T09:57:50.137980Z","shell.execute_reply":"2022-08-19T09:57:50.593704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# sampling images","metadata":{}},{"cell_type":"code","source":"dfs = []\nfor group_label, df_group in tqdm( train_df.groupby(\"label\") ):\n    n_sample = min( NUM_IMAGES_PER_CLASS, df_group.shape[0] )\n    dfs.append( df_group.iloc[:n_sample,:] )\ndfs = pd.concat( dfs )","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:57:54.848907Z","iopub.execute_input":"2022-08-19T09:57:54.849289Z","iopub.status.idle":"2022-08-19T09:57:57.094010Z","shell.execute_reply.started":"2022-08-19T09:57:54.849257Z","shell.execute_reply":"2022-08-19T09:57:57.093029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter = Counter( dfs[\"label\"].values )\nplt.plot( sorted( [ v for (k,v) in counter.items()] )[::-1] )\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:57:57.099260Z","iopub.execute_input":"2022-08-19T09:57:57.099973Z","iopub.status.idle":"2022-08-19T09:57:57.342692Z","shell.execute_reply.started":"2022-08-19T09:57:57.099931Z","shell.execute_reply":"2022-08-19T09:57:57.341455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"100.0 * dfs.shape[0] / train_df.shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:58:02.942589Z","iopub.execute_input":"2022-08-19T09:58:02.943169Z","iopub.status.idle":"2022-08-19T09:58:02.951768Z","shell.execute_reply.started":"2022-08-19T09:58:02.943130Z","shell.execute_reply":"2022-08-19T09:58:02.950658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = dfs","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:58:03.377936Z","iopub.execute_input":"2022-08-19T09:58:03.378458Z","iopub.status.idle":"2022-08-19T09:58:03.383140Z","shell.execute_reply.started":"2022-08-19T09:58:03.378414Z","shell.execute_reply":"2022-08-19T09:58:03.381859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kf = KFold(n_splits=N_GROUPS, shuffle=True)\nlist_fold = np.array([0] * train_df.shape[0])\nfor fold_no, (train_index, test_index) in enumerate(kf.split( train_df[\"id\"].values, train_df[\"label\"].values )):\n    for index in test_index:\n        list_fold[index] = fold_no\n    #train_df.iloc[test_index, \"fold\"] = fold_no\ntrain_df[\"fold\"] = list_fold\ntrain_df.to_csv(\"sampled_train.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:58:05.557332Z","iopub.execute_input":"2022-08-19T09:58:05.558114Z","iopub.status.idle":"2022-08-19T09:58:06.403494Z","shell.execute_reply.started":"2022-08-19T09:58:05.558045Z","shell.execute_reply":"2022-08-19T09:58:06.402380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:58:06.580402Z","iopub.execute_input":"2022-08-19T09:58:06.580753Z","iopub.status.idle":"2022-08-19T09:58:06.594125Z","shell.execute_reply.started":"2022-08-19T09:58:06.580721Z","shell.execute_reply":"2022-08-19T09:58:06.593033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load image","metadata":{}},{"cell_type":"code","source":"def load_image(path, img_size=IMAGE_SIZE):\n    img = np.array( Image.open(path) )\n    h, w = img.shape[:2]\n    ratio = float(IMAGE_SIZE) / max(h, w)\n    h2, w2 = int( ratio * h ), int( ratio * w )\n    img = cv2.resize(img, dsize=(w2, h2))\n    #print(h,w, h2, w2)\n    if len(img.shape) == 2:\n        img = np.concatenate( [ img[:,:,np.newaxis] ]*3, axis=2)\n        print(\"!!!\", img.shape)\n    img = tf.io.encode_jpeg(img, quality=70, optimize_size=True).numpy()\n    return img","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:58:08.718720Z","iopub.execute_input":"2022-08-19T09:58:08.719498Z","iopub.status.idle":"2022-08-19T09:58:08.729710Z","shell.execute_reply.started":"2022-08-19T09:58:08.719457Z","shell.execute_reply":"2022-08-19T09:58:08.728436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Convert to TFRecords","metadata":{}},{"cell_type":"code","source":"def _bytes_feature(value):\n    \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n    if isinstance(value, type(tf.constant(0))):\n        value = value.numpy() \n    return tf.train.Feature(bytes_list=tf.train.BytesList(value=[value]))\n\ndef _float_feature(value):\n    \"\"\"Returns a float_list from a float / double.\"\"\"\n    return tf.train.Feature(float_list=tf.train.FloatList(value=[value]))\n\ndef _int64_feature(value):\n    \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n    return tf.train.Feature(int64_list=tf.train.Int64List(value=[value]))","metadata":{"execution":{"iopub.status.busy":"2022-08-17T08:06:38.745848Z","iopub.execute_input":"2022-08-17T08:06:38.746223Z","iopub.status.idle":"2022-08-17T08:06:38.753632Z","shell.execute_reply.started":"2022-08-17T08:06:38.746192Z","shell.execute_reply":"2022-08-17T08:06:38.752448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def serialize_example(image, label):\n    feature = {\n        'image/encoded': _bytes_feature(image),\n        'image/class/label': _int64_feature(label),\n    }\n    example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n    return example_proto.SerializeToString()","metadata":{"execution":{"iopub.status.busy":"2022-08-17T08:08:19.096129Z","iopub.execute_input":"2022-08-17T08:08:19.096511Z","iopub.status.idle":"2022-08-17T08:08:19.101909Z","shell.execute_reply.started":"2022-08-17T08:08:19.096474Z","shell.execute_reply":"2022-08-17T08:08:19.101079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Dataset","metadata":{}},{"cell_type":"code","source":"if CREATE_DATASET:\n    import json\n\n    ### Create Kaggle Dataset if not exists \n\n    !rm -r /tmp/{DATASET_NAME}\n\n    os.makedirs(f'/tmp/{DATASET_NAME}', exist_ok=True)\n\n    with open('/kaggle/input/kaggle-api-creds/kaggle.json') as f:\n        kaggle_creds = json.load(f)\n\n    os.environ['KAGGLE_USERNAME'] = USER_NAME\n    os.environ['KAGGLE_KEY'] = kaggle_creds['key']\n\n    !kaggle datasets init -p /tmp/{DATASET_NAME}\n\n    with open(f'/tmp/{DATASET_NAME}/dataset-metadata.json') as f:\n        dataset_meta = json.load(f)\n    dataset_meta['id'] = f'{USER_NAME}/{DATASET_NAME}'\n    dataset_meta['title'] = DATASET_NAME\n    with open(f'/tmp/{DATASET_NAME}/dataset-metadata.json', \"w\") as outfile:\n        json.dump(dataset_meta, outfile)\n    print(dataset_meta)\n\n    !cp /tmp/{DATASET_NAME}/dataset-metadata.json /tmp/{DATASET_NAME}/meta.json\n    !ls /tmp/{DATASET_NAME}\n\n    !kaggle datasets create -u -p /tmp/{DATASET_NAME} \n","metadata":{"execution":{"iopub.status.busy":"2022-08-17T08:06:46.344718Z","iopub.execute_input":"2022-08-17T08:06:46.345138Z","iopub.status.idle":"2022-08-17T08:06:53.594499Z","shell.execute_reply.started":"2022-08-17T08:06:46.345103Z","shell.execute_reply":"2022-08-17T08:06:53.593512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_tf_records(df, group_no):\n    tfr_filename = f'/tmp/{DATASET_NAME}/guie-glr2021-train-{group_no:02d}-{df.shape[0]}.tfrec'\n    with tf.io.TFRecordWriter(tfr_filename) as writer:\n        for (x,y) in zip(tqdm(df[\"id\"].values), df[\"label\"].values):\n            file_path = f\"../input/landmark-recognition-2021/train/{x[0]}/{x[1]}/{x[2]}/{x}.jpg\"\n            img = load_image(file_path)\n            example = serialize_example(img, y)\n            writer.write(example)\n    !du -s -m {tfr_filename}","metadata":{"execution":{"iopub.status.busy":"2022-08-17T08:06:58.851872Z","iopub.execute_input":"2022-08-17T08:06:58.85224Z","iopub.status.idle":"2022-08-17T08:06:58.860505Z","shell.execute_reply.started":"2022-08-17T08:06:58.852204Z","shell.execute_reply":"2022-08-17T08:06:58.859403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if CREATE_DATASET:\n    !mkdir -p /tmp/{DATASET_NAME}\n    outpath_train = f\"/tmp/{DATASET_NAME}/\"\n\n    train_df = train_df.reset_index(drop=True)\n    max_group = train_df[\"fold\"].max()\n#    kf = KFold(n_splits=N_GROUPS, shuffle=True)\n#    for group_no, (train_index, test_index) in enumerate(kf.split( train_df[\"id\"].values, train_df[\"label\"].values )):\n    for group_no, df in train_df.groupby(\"fold\"):\n        !df -h\n        print(f\">>>> group {group_no} / {max_group} \" + \"-\"*60 )\n        create_tf_records(df, int(group_no) )","metadata":{"execution":{"iopub.status.busy":"2022-08-17T08:09:33.186417Z","iopub.execute_input":"2022-08-17T08:09:33.186836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if CREATE_DATASET:\n    version_name = datetime.now().strftime(\"%Y%m%d-%H%M%S\")\n    print(version_name)\n    !kaggle datasets version -m {version_name} -p /tmp/{DATASET_NAME} -r zip -q","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}