{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Load Meta Data","metadata":{}},{"cell_type":"code","source":"# LOAD LIBRARIES\nimport numpy as np, pandas as pd, os\nimport matplotlib.pyplot as plt, cv2\nimport tensorflow as tf, re, math\nfrom tqdm.notebook import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-29T13:35:32.095426Z","iopub.execute_input":"2022-09-29T13:35:32.095894Z","iopub.status.idle":"2022-09-29T13:35:38.187595Z","shell.execute_reply.started":"2022-09-29T13:35:32.095845Z","shell.execute_reply":"2022-09-29T13:35:38.186163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/landmark-recognition-2021/train.csv')\ndf","metadata":{"execution":{"iopub.status.busy":"2022-09-29T13:35:38.189695Z","iopub.execute_input":"2022-09-29T13:35:38.190116Z","iopub.status.idle":"2022-09-29T13:35:39.742432Z","shell.execute_reply.started":"2022-09-29T13:35:38.190083Z","shell.execute_reply":"2022-09-29T13:35:39.741618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['cnt'] = df.groupby('landmark_id')['landmark_id'].transform('count')\ndf = df.sort_values(by='cnt', ascending=False)\ndf","metadata":{"execution":{"iopub.status.busy":"2022-09-29T13:35:39.743501Z","iopub.execute_input":"2022-09-29T13:35:39.743889Z","iopub.status.idle":"2022-09-29T13:35:40.048471Z","shell.execute_reply.started":"2022-09-29T13:35:39.743861Z","shell.execute_reply":"2022-09-29T13:35:40.047394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N = 50\n\ndf = df[df.cnt >= N]\ndf","metadata":{"execution":{"iopub.status.busy":"2022-09-29T13:35:40.050394Z","iopub.execute_input":"2022-09-29T13:35:40.050902Z","iopub.status.idle":"2022-09-29T13:35:40.095564Z","shell.execute_reply.started":"2022-09-29T13:35:40.050850Z","shell.execute_reply":"2022-09-29T13:35:40.094440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp_df = df.groupby(\"landmark_id\").apply(lambda x:x.sample(n=N)).reset_index(drop=True)\ntmp_df","metadata":{"execution":{"iopub.status.busy":"2022-09-29T13:35:40.097974Z","iopub.execute_input":"2022-09-29T13:35:40.098271Z","iopub.status.idle":"2022-09-29T13:35:45.241977Z","shell.execute_reply.started":"2022-09-29T13:35:40.098243Z","shell.execute_reply":"2022-09-29T13:35:45.241039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp_df['img_path'] = '../input/landmark-recognition-2021/train/' +\\\n                     tmp_df.id.str[0] + '/' +\\\n                     tmp_df.id.str[1] + '/' +\\\n                     tmp_df.id.str[2] + '/' +\\\n                     tmp_df.id + '.jpg'\ntmp_df = tmp_df[['img_path', 'landmark_id']].rename({\"img_path\":\"img_path\", \"landmark_id\" : 'label'}, axis=1)\ntmp_df","metadata":{"execution":{"iopub.status.busy":"2022-09-29T13:35:45.243099Z","iopub.execute_input":"2022-09-29T13:35:45.243370Z","iopub.status.idle":"2022-09-29T13:35:46.448653Z","shell.execute_reply.started":"2022-09-29T13:35:45.243343Z","shell.execute_reply":"2022-09-29T13:35:46.447630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp_df['label'] = pd.factorize(tmp_df.label)[0] # make label starting with 0\ntmp_df","metadata":{"execution":{"iopub.status.busy":"2022-09-29T13:35:46.450085Z","iopub.execute_input":"2022-09-29T13:35:46.450394Z","iopub.status.idle":"2022-09-29T13:35:46.468350Z","shell.execute_reply.started":"2022-09-29T13:35:46.450362Z","shell.execute_reply":"2022-09-29T13:35:46.467222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Sanity Check","metadata":{}},{"cell_type":"code","source":"for row in tmp_df.itertuples():\n    print(row.Index, row.img_path, row.label)\n    break","metadata":{"execution":{"iopub.status.busy":"2022-09-29T13:35:46.469689Z","iopub.execute_input":"2022-09-29T13:35:46.470263Z","iopub.status.idle":"2022-09-29T13:35:46.478319Z","shell.execute_reply.started":"2022-09-29T13:35:46.470211Z","shell.execute_reply":"2022-09-29T13:35:46.477426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\n\nimg = Image.open(tmp_df.loc[0].img_path)\nimg","metadata":{"execution":{"iopub.status.busy":"2022-09-29T13:35:46.479606Z","iopub.execute_input":"2022-09-29T13:35:46.479961Z","iopub.status.idle":"2022-09-29T13:35:46.739115Z","shell.execute_reply.started":"2022-09-29T13:35:46.479907Z","shell.execute_reply":"2022-09-29T13:35:46.738227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _bytes_feature(value):\n    \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n    if isinstance(value, type(tf.constant(0))):\n        value = value.numpy() # BytesList won't unpack a string from an EagerTensor.\n    return tf.train.Feature(bytes_list=tf.train.BytesList(value=[value]))\n\ndef _float_feature(value):\n    \"\"\"Returns a float_list from a float / double.\"\"\"\n    return tf.train.Feature(float_list=tf.train.FloatList(value=[value]))\n\ndef _int64_feature(value):\n    \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n    return tf.train.Feature(int64_list=tf.train.Int64List(value=[value]))","metadata":{"execution":{"iopub.status.busy":"2022-09-29T13:35:46.740070Z","iopub.execute_input":"2022-09-29T13:35:46.740319Z","iopub.status.idle":"2022-09-29T13:35:46.748111Z","shell.execute_reply.started":"2022-09-29T13:35:46.740293Z","shell.execute_reply":"2022-09-29T13:35:46.747405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def serialize_example(img, label): # image, label\n    feature = {\n        'image/encoded': _bytes_feature(img),\n        'image/class/label': _int64_feature(label)\n    }\n    example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n    return example_proto.SerializeToString()","metadata":{"execution":{"iopub.status.busy":"2022-09-29T13:35:46.749163Z","iopub.execute_input":"2022-09-29T13:35:46.749555Z","iopub.status.idle":"2022-09-29T13:35:46.760239Z","shell.execute_reply.started":"2022-09-29T13:35:46.749525Z","shell.execute_reply":"2022-09-29T13:35:46.759546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"min_label = tmp_df.label.min()\nmax_label = tmp_df.label.max()\n\ntmp_df = tmp_df.sample(frac=1, random_state=42) # shuffle but no reproducibility :(\n\niterations = 50\nbatch_size = int(len(tmp_df) / iterations)\noutput_pat = 'guie-glr2021-train-'\n\nfor n in range(iterations):\n    with tf.io.TFRecordWriter(output_pat + f\"{str(n).zfill(2)}-{max_label}.tfrec\") as writer:\n        tdf = tmp_df.iloc[batch_size*n:batch_size*(n+1), :]\n        for row in tqdm(tdf.itertuples()):\n            idx = row.Index\n            img_path, label = row.img_path, row.label\n            example = serialize_example(img=open(img_path, 'rb').read(), label=label)\n            writer.write(example)","metadata":{"execution":{"iopub.status.busy":"2022-09-29T13:35:46.761202Z","iopub.execute_input":"2022-09-29T13:35:46.761629Z","iopub.status.idle":"2022-09-29T14:16:24.341550Z","shell.execute_reply.started":"2022-09-29T13:35:46.761598Z","shell.execute_reply":"2022-09-29T14:16:24.339221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}