{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Load Libraries","metadata":{}},{"cell_type":"code","source":"import os, json, random, cv2\nimport numpy as np, pandas as pd\nimport matplotlib.pyplot as plt\nimport tensorflow as tf, re, math\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-10-11T20:12:16.960725Z","iopub.execute_input":"2022-10-11T20:12:16.961253Z","iopub.status.idle":"2022-10-11T20:12:22.808696Z","shell.execute_reply.started":"2022-10-11T20:12:16.961205Z","shell.execute_reply":"2022-10-11T20:12:22.807672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATASET_NAME = 'landmark-recognition-2021-tfrecords-subset'\nos.makedirs(f'/tmp/{DATASET_NAME}', exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2022-10-11T20:13:12.244140Z","iopub.execute_input":"2022-10-11T20:13:12.244515Z","iopub.status.idle":"2022-10-11T20:13:12.249502Z","shell.execute_reply.started":"2022-10-11T20:13:12.244481Z","shell.execute_reply":"2022-10-11T20:13:12.248591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Configurations","metadata":{}},{"cell_type":"code","source":"RESIZE = True\nN_LABELS = 81313\nN_LABELS_SUBSET = 100\nIMAGE_SIZE = 224\nN_GROUPS = 12\nN_FOLDS = 5\nN_TFRs = N_GROUPS*N_FOLDS\nSUBSET = True  # Keep SUBSET=True while debugging (Faster Execution)\nBATCH_SIZE = 32\nFOLDS = [0]\nGROUPS = [11]\nassert max(FOLDS)<N_FOLDS, \"ELEMENTS OF FOLDS can't be greater than N_FOLDS\"\nassert max(GROUPS)<N_GROUPS, \"ELEMENTS OF FOLDS can't be greater than N_FOLDS\"","metadata":{"execution":{"iopub.status.busy":"2022-10-11T20:13:12.766669Z","iopub.execute_input":"2022-10-11T20:13:12.767044Z","iopub.status.idle":"2022-10-11T20:13:12.773775Z","shell.execute_reply.started":"2022-10-11T20:13:12.766980Z","shell.execute_reply":"2022-10-11T20:13:12.772720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preparing Folds and Groups\n### You may change heuristics for Groups as per your requirements","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('../input/landmark-recognition-2021/train.csv')\nif SUBSET:\n    landmarks = random.sample(list(train_df.landmark_id.unique()),100)\n    train_df = train_df[train_df.landmark_id.isin(landmarks)].reset_index(drop=True)\n    N_LABELS = N_LABELS_SUBSET\ntrain_df['original_landmark_id'] = train_df.landmark_id\nprint(train_df.shape)\ntrain_df['order'] = np.arange(train_df.shape[0])\ntrain_df['order'] = train_df.groupby('landmark_id').order.rank()-1\nlandmark_counts = train_df.landmark_id.value_counts()\ntrain_df['landmark_counts'] = landmark_counts.loc[train_df.landmark_id.values].values\ntrain_df['fold'] = (train_df['order']%N_FOLDS).astype(int)\nall_groups = [(1/N_GROUPS)*x for x in range(N_GROUPS)]\nfor i,partition_val in enumerate(train_df.landmark_counts.quantile(all_groups).values):\n                     train_df.loc[train_df.landmark_counts>=partition_val,'group'] = i \n        \nlandmark_map = train_df.sort_values(by='landmark_counts').landmark_id.drop_duplicates().reset_index(drop=True)\nlandmark_dict = {landmark_map.loc[x]:N_LABELS-x-1 for x in range(N_LABELS)}\ntrain_df['landmark_id'] = train_df.original_landmark_id.apply(lambda x: landmark_dict[x])\ntrain_df = train_df.sample(frac=1).reset_index(drop=True)\ntrain_df.to_csv(f'/tmp/{DATASET_NAME}/train_meta_data.csv',index=False)\ntrain_df.sample(5)","metadata":{"execution":{"iopub.status.busy":"2022-10-11T20:13:14.071090Z","iopub.execute_input":"2022-10-11T20:13:14.071473Z","iopub.status.idle":"2022-10-11T20:13:15.280788Z","shell.execute_reply.started":"2022-10-11T20:13:14.071421Z","shell.execute_reply":"2022-10-11T20:13:15.279929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Checking Null values\ntrain_df.isna().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2022-10-11T20:13:15.629860Z","iopub.execute_input":"2022-10-11T20:13:15.630239Z","iopub.status.idle":"2022-10-11T20:13:15.640371Z","shell.execute_reply.started":"2022-10-11T20:13:15.630208Z","shell.execute_reply":"2022-10-11T20:13:15.639087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Group Partitions","metadata":{}},{"cell_type":"code","source":"train_df.groupby('group').landmark_counts.agg(['min','max'])","metadata":{"execution":{"iopub.status.busy":"2022-10-11T20:13:19.154669Z","iopub.execute_input":"2022-10-11T20:13:19.155017Z","iopub.status.idle":"2022-10-11T20:13:19.172592Z","shell.execute_reply.started":"2022-10-11T20:13:19.154986Z","shell.execute_reply":"2022-10-11T20:13:19.171818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Some Statistics","metadata":{}},{"cell_type":"code","source":"#Landmark Counts\ntrain_df.landmark_id.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-10-11T20:13:21.091195Z","iopub.execute_input":"2022-10-11T20:13:21.091577Z","iopub.status.idle":"2022-10-11T20:13:21.100379Z","shell.execute_reply.started":"2022-10-11T20:13:21.091542Z","shell.execute_reply":"2022-10-11T20:13:21.099245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#No of images GroupBy landmark counts\ntrain_df.landmark_counts.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-10-11T20:13:21.378393Z","iopub.execute_input":"2022-10-11T20:13:21.378740Z","iopub.status.idle":"2022-10-11T20:13:21.386748Z","shell.execute_reply.started":"2022-10-11T20:13:21.378710Z","shell.execute_reply":"2022-10-11T20:13:21.385726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#No of Images in Each Folds\ntrain_df.fold.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-10-11T20:13:22.102932Z","iopub.execute_input":"2022-10-11T20:13:22.103285Z","iopub.status.idle":"2022-10-11T20:13:22.112324Z","shell.execute_reply.started":"2022-10-11T20:13:22.103253Z","shell.execute_reply":"2022-10-11T20:13:22.111115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#No of Images in Each Group\ntrain_df.group.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-10-11T20:13:22.706761Z","iopub.execute_input":"2022-10-11T20:13:22.707157Z","iopub.status.idle":"2022-10-11T20:13:22.715270Z","shell.execute_reply.started":"2022-10-11T20:13:22.707126Z","shell.execute_reply":"2022-10-11T20:13:22.714438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# No of Landmark in each Fold\ntrain_df.groupby('fold').landmark_id.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-10-11T20:13:23.279060Z","iopub.execute_input":"2022-10-11T20:13:23.279414Z","iopub.status.idle":"2022-10-11T20:13:23.288847Z","shell.execute_reply.started":"2022-10-11T20:13:23.279386Z","shell.execute_reply":"2022-10-11T20:13:23.287524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## No of images in each (fold,group)\n### Each row corresponds to single tf-records ","metadata":{}},{"cell_type":"code","source":"pd.pivot_table(train_df.groupby(['fold','group']).id.count().reset_index(),index='fold',columns='group')","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2022-10-11T20:13:26.779826Z","iopub.execute_input":"2022-10-11T20:13:26.780239Z","iopub.status.idle":"2022-10-11T20:13:26.822684Z","shell.execute_reply.started":"2022-10-11T20:13:26.780205Z","shell.execute_reply":"2022-10-11T20:13:26.821663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating TF-Records","metadata":{}},{"cell_type":"code","source":"def _bytes_feature(value):\n  \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n  if isinstance(value, type(tf.constant(0))):\n    value = value.numpy() # BytesList won't unpack a string from an EagerTensor.\n  return tf.train.Feature(bytes_list=tf.train.BytesList(value=[value]))\n\ndef _float_feature(value):\n  \"\"\"Returns a float_list from a float / double.\"\"\"\n  return tf.train.Feature(float_list=tf.train.FloatList(value=[value]))\n\ndef _int64_feature(value):\n  \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n  return tf.train.Feature(int64_list=tf.train.Int64List(value=[value]))\n\ndef serialize_example(image,image_name,label):\n    feature = {\n        'image': _bytes_feature(image),\n        'image_name': _bytes_feature(image_name),\n        'target': _int64_feature(label),\n      }\n    example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n    return example_proto.SerializeToString()","metadata":{"execution":{"iopub.status.busy":"2022-10-11T20:13:33.595014Z","iopub.execute_input":"2022-10-11T20:13:33.595388Z","iopub.status.idle":"2022-10-11T20:13:33.605549Z","shell.execute_reply.started":"2022-10-11T20:13:33.595354Z","shell.execute_reply":"2022-10-11T20:13:33.604390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_tf_records(fold  = 0, group = 0):\n    df = train_df[(train_df.fold==fold) & (train_df.group==group)]\n    tfr_filename = f'/tmp/{DATASET_NAME}/landmark-2021-train-{fold}-{group}-{df.shape[0]}.tfrec'\n    with tf.io.TFRecordWriter(tfr_filename) as writer:\n        for i,row in df.iterrows():\n            image_id = row.id\n            target = row.landmark_id\n            image_path = \"../input/landmark-recognition-2020/train/{}/{}/{}/{}.jpg\".format(image_id[0],image_id[1],image_id[2],image_id) \n            image_encoded = tf.io.read_file(image_path)\n            image_name = str.encode(image_id)\n            example = serialize_example(image_encoded,image_name,target)\n            writer.write(example)","metadata":{"execution":{"iopub.status.busy":"2022-10-11T20:13:34.122802Z","iopub.execute_input":"2022-10-11T20:13:34.123195Z","iopub.status.idle":"2022-10-11T20:13:34.131235Z","shell.execute_reply.started":"2022-10-11T20:13:34.123158Z","shell.execute_reply":"2022-10-11T20:13:34.130225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import joblib\nfor fold in range(N_FOLDS):\n    _ = joblib.Parallel(n_jobs=8)(\n        joblib.delayed(create_tf_records)(fold,group) for group in tqdm(range(N_GROUPS))\n    )","metadata":{"execution":{"iopub.status.busy":"2022-10-11T20:13:35.334777Z","iopub.execute_input":"2022-10-11T20:13:35.335177Z","iopub.status.idle":"2022-10-11T20:13:52.824324Z","shell.execute_reply.started":"2022-10-11T20:13:35.335141Z","shell.execute_reply":"2022-10-11T20:13:52.823309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Great, Now you have to run this notebook for different folds & prepare all the tf records!! Change to SUBSET = False and FOLDS = [i] where i is the fold number.\n\n## The good part is that I have already done this & you can find all the tf records & metadata here:\n\nFold 0: https://www.kaggle.com/ks2019/landmark-recognition-2021-tfrecords-fold0\nFold 1: https://www.kaggle.com/ks2019/landmark-recognition-2021-tfrecords-fold1\nFold 2: https://www.kaggle.com/ks2019/landmark-recognition-2021-tfrecords-fold2\nFold 3: https://www.kaggle.com/ks2019/landmark-recognition-2021-tfrecords-fold3\nFold 4: https://www.kaggle.com/ks2019/landmark-recognition-2021-tfrecords-fold4\nMetadata: https://www.kaggle.com/ks2019/landmark-recognition-2021-tfrecords-fold1?select=train_meta_data.csv","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}