{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, json, random, cv2\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport pathlib\nimport tensorflow as tf\nimport re\nimport math\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2021-08-24T04:43:48.157816Z","iopub.execute_input":"2021-08-24T04:43:48.158338Z","iopub.status.idle":"2021-08-24T04:43:48.164775Z","shell.execute_reply.started":"2021-08-24T04:43:48.158289Z","shell.execute_reply":"2021-08-24T04:43:48.163574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BASE_CFG:\n    comp_path = \"/kaggle/input/landmark-retrieval-2021\"\n    n_labels = 81313\n    NUMBER_OF_CLASSES = 81313\n    BATCH_SIZE = 256\n    EPOCHS = 10\n    LEARNING_RATE=0.0001\n    OBJ_HEIGHT = 256\n    OBJ_WIDTH = 256\n    IMAGE_SIZE = 256\n    CHANNELS = 0\n    NET = 0\n    dtype = 'float32'\n    VAL_CLASS_NUM = 1000\n    FOLD = 3","metadata":{"execution":{"iopub.status.busy":"2021-08-24T04:43:48.725077Z","iopub.execute_input":"2021-08-24T04:43:48.725468Z","iopub.status.idle":"2021-08-24T04:43:48.7317Z","shell.execute_reply.started":"2021-08-24T04:43:48.725436Z","shell.execute_reply":"2021-08-24T04:43:48.730597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CFG = BASE_CFG()\ncomp_path = pathlib.Path(CFG.comp_path)\nprint(comp_path)","metadata":{"execution":{"iopub.status.busy":"2021-08-24T04:43:49.145298Z","iopub.execute_input":"2021-08-24T04:43:49.145676Z","iopub.status.idle":"2021-08-24T04:43:49.150767Z","shell.execute_reply.started":"2021-08-24T04:43:49.145642Z","shell.execute_reply":"2021-08-24T04:43:49.149971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DATASET MAKING","metadata":{}},{"cell_type":"code","source":"DATASET_NAME = f'landmark-retrieval-2021-stratify-fold{CFG.FOLD}'","metadata":{"execution":{"iopub.status.busy":"2021-08-24T04:43:50.063496Z","iopub.execute_input":"2021-08-24T04:43:50.064317Z","iopub.status.idle":"2021-08-24T04:43:50.069418Z","shell.execute_reply.started":"2021-08-24T04:43:50.064273Z","shell.execute_reply":"2021-08-24T04:43:50.068147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -r /tmp/{DATASET_NAME}\nos.makedirs(f'/tmp/{DATASET_NAME}', exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2021-08-24T04:43:50.681101Z","iopub.execute_input":"2021-08-24T04:43:50.681564Z","iopub.status.idle":"2021-08-24T04:43:51.418068Z","shell.execute_reply.started":"2021-08-24T04:43:50.681522Z","shell.execute_reply":"2021-08-24T04:43:51.416779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('../input/statking-kaggle-api/kaggle.json') as f:\n    kaggle_creds = json.load(f)\nos.environ['KAGGLE_USERNAME'] = kaggle_creds['username']\nos.environ['KAGGLE_KEY'] = kaggle_creds['key']","metadata":{"execution":{"iopub.status.busy":"2021-08-24T04:43:51.419914Z","iopub.execute_input":"2021-08-24T04:43:51.420411Z","iopub.status.idle":"2021-08-24T04:43:51.430757Z","shell.execute_reply.started":"2021-08-24T04:43:51.420375Z","shell.execute_reply":"2021-08-24T04:43:51.42989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!kaggle datasets init -p /tmp/{DATASET_NAME}\nwith open(f'/tmp/{DATASET_NAME}/dataset-metadata.json') as f:\n    dataset_meta = json.load(f)\ndataset_meta['id'] = f'deepkim/{DATASET_NAME}'\ndataset_meta['title'] = DATASET_NAME\nwith open(f'/tmp/{DATASET_NAME}/dataset-metadata.json', \"w\") as outfile:\n    json.dump(dataset_meta, outfile)\nprint(dataset_meta)\n\n!cp /tmp/{DATASET_NAME}/dataset-metadata.json /tmp/{DATASET_NAME}/meta.json\n!ls /tmp/{DATASET_NAME}\n\n!kaggle datasets create -u -p /tmp/{DATASET_NAME}","metadata":{"execution":{"iopub.status.busy":"2021-08-24T04:43:51.719682Z","iopub.execute_input":"2021-08-24T04:43:51.720196Z","iopub.status.idle":"2021-08-24T04:44:05.193287Z","shell.execute_reply.started":"2021-08-24T04:43:51.720162Z","shell.execute_reply":"2021-08-24T04:44:05.191871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#os.listdir(\"/tmp/landmark-retrieval-2021-tfrecords-size256\")","metadata":{"execution":{"iopub.status.busy":"2021-08-24T04:44:05.195319Z","iopub.execute_input":"2021-08-24T04:44:05.195663Z","iopub.status.idle":"2021-08-24T04:44:05.200137Z","shell.execute_reply.started":"2021-08-24T04:44:05.195626Z","shell.execute_reply":"2021-08-24T04:44:05.199252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /tmp/{DATASET_NAME}\n","metadata":{"execution":{"iopub.status.busy":"2021-08-24T04:44:05.202403Z","iopub.execute_input":"2021-08-24T04:44:05.202758Z","iopub.status.idle":"2021-08-24T04:44:05.940761Z","shell.execute_reply.started":"2021-08-24T04:44:05.202727Z","shell.execute_reply":"2021-08-24T04:44:05.939517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(comp_path / \"train.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-08-24T04:44:05.942981Z","iopub.execute_input":"2021-08-24T04:44:05.943454Z","iopub.status.idle":"2021-08-24T04:44:07.646179Z","shell.execute_reply.started":"2021-08-24T04:44:05.943401Z","shell.execute_reply":"2021-08-24T04:44:07.645128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"landmark_id_count_dict = train['landmark_id'].value_counts().to_dict()","metadata":{"execution":{"iopub.status.busy":"2021-08-24T04:44:07.647431Z","iopub.execute_input":"2021-08-24T04:44:07.647775Z","iopub.status.idle":"2021-08-24T04:44:07.732395Z","shell.execute_reply.started":"2021-08-24T04:44:07.647743Z","shell.execute_reply":"2021-08-24T04:44:07.731245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#class_pair_dict = {}\nclass_pair_dict_keys = train.groupby('landmark_id').count().reset_index().index.tolist()\nclass_pair_dict_values = train.groupby('landmark_id').count().reset_index()['landmark_id'].tolist()","metadata":{"execution":{"iopub.status.busy":"2021-08-24T03:40:49.310533Z","iopub.execute_input":"2021-08-24T03:40:49.310844Z","iopub.status.idle":"2021-08-24T03:40:49.798611Z","shell.execute_reply.started":"2021-08-24T03:40:49.310816Z","shell.execute_reply":"2021-08-24T03:40:49.797568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_pair_dict = {key:value for key, value in zip(class_pair_dict_keys, class_pair_dict_values)}\nreverse_class_pair_dict = {value:key for key, value in zip(class_pair_dict_keys, class_pair_dict_values)}","metadata":{"execution":{"iopub.status.busy":"2021-08-24T03:40:49.800151Z","iopub.execute_input":"2021-08-24T03:40:49.80044Z","iopub.status.idle":"2021-08-24T03:40:49.827994Z","shell.execute_reply.started":"2021-08-24T03:40:49.800412Z","shell.execute_reply":"2021-08-24T03:40:49.827003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['fixed_landmark_id'] = train['landmark_id'].map(reverse_class_pair_dict)","metadata":{"execution":{"iopub.status.busy":"2021-08-24T03:40:50.779422Z","iopub.execute_input":"2021-08-24T03:40:50.779794Z","iopub.status.idle":"2021-08-24T03:40:50.872277Z","shell.execute_reply.started":"2021-08-24T03:40:50.779763Z","shell.execute_reply":"2021-08-24T03:40:50.871251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['landmark_id_count'] = train['landmark_id'].map(landmark_id_count_dict)","metadata":{"execution":{"iopub.status.busy":"2021-08-24T03:40:51.005472Z","iopub.execute_input":"2021-08-24T03:40:51.005858Z","iopub.status.idle":"2021-08-24T03:40:51.098496Z","shell.execute_reply.started":"2021-08-24T03:40:51.005826Z","shell.execute_reply":"2021-08-24T03:40:51.097391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.id.nunique()","metadata":{"execution":{"iopub.status.busy":"2021-08-24T03:40:51.719529Z","iopub.execute_input":"2021-08-24T03:40:51.719904Z","iopub.status.idle":"2021-08-24T03:40:52.384735Z","shell.execute_reply.started":"2021-08-24T03:40:51.719873Z","shell.execute_reply":"2021-08-24T03:40:52.38369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2021-08-24T03:41:25.809366Z","iopub.execute_input":"2021-08-24T03:41:25.809765Z","iopub.status.idle":"2021-08-24T03:41:25.8338Z","shell.execute_reply.started":"2021-08-24T03:41:25.80973Z","shell.execute_reply":"2021-08-24T03:41:25.832725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\ntrain['fold'] = -1\nskf = StratifiedKFold(n_splits=40)\nfor fold,(tr_idx, val_idx) in enumerate(skf.split(train, y = train['fixed_landmark_id'])):\n    train.loc[val_idx,'fold'] = fold","metadata":{"execution":{"iopub.status.busy":"2021-08-24T04:03:04.898977Z","iopub.execute_input":"2021-08-24T04:03:04.899973Z","iopub.status.idle":"2021-08-24T04:04:29.813965Z","shell.execute_reply.started":"2021-08-24T04:03:04.899922Z","shell.execute_reply":"2021-08-24T04:04:29.812933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tqdm.pandas()","metadata":{"execution":{"iopub.status.busy":"2021-08-24T04:04:29.815709Z","iopub.execute_input":"2021-08-24T04:04:29.816012Z","iopub.status.idle":"2021-08-24T04:04:29.822116Z","shell.execute_reply.started":"2021-08-24T04:04:29.815982Z","shell.execute_reply":"2021-08-24T04:04:29.821182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#image_path = \"../input/landmark-recognition-2020/train/{}/{}/{}/{}.jpg\".format(image_id[0],image_id[1],image_id[2],image_id) \ntrain['file_path'] = train['id'].map(lambda x: \"train/{}/{}/{}/{}.jpg\".format(x[0],x[1],x[2],x))","metadata":{"execution":{"iopub.status.busy":"2021-08-24T04:04:50.293723Z","iopub.execute_input":"2021-08-24T04:04:50.294086Z","iopub.status.idle":"2021-08-24T04:04:51.988541Z","shell.execute_reply.started":"2021-08-24T04:04:50.294056Z","shell.execute_reply":"2021-08-24T04:04:51.987561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['kaggle_file_path'] = train['id'].map(lambda x: \"../input/landmark-retrieval-2021/train/{}/{}/{}/{}.jpg\".format(x[0],x[1],x[2],x))","metadata":{"execution":{"iopub.status.busy":"2021-08-24T04:04:58.516587Z","iopub.execute_input":"2021-08-24T04:04:58.516962Z","iopub.status.idle":"2021-08-24T04:05:00.181896Z","shell.execute_reply.started":"2021-08-24T04:04:58.516932Z","shell.execute_reply":"2021-08-24T04:05:00.180882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-24T04:05:01.933512Z","iopub.execute_input":"2021-08-24T04:05:01.93391Z","iopub.status.idle":"2021-08-24T04:05:01.950486Z","shell.execute_reply.started":"2021-08-24T04:05:01.933877Z","shell.execute_reply":"2021-08-24T04:05:01.949192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for fold in range(40):\n    print(train.loc[train['fold']==fold]['fixed_landmark_id'].value_counts().sort_values())","metadata":{"execution":{"iopub.status.busy":"2021-08-24T04:05:55.257783Z","iopub.execute_input":"2021-08-24T04:05:55.258143Z","iopub.status.idle":"2021-08-24T04:05:56.383385Z","shell.execute_reply.started":"2021-08-24T04:05:55.258113Z","shell.execute_reply":"2021-08-24T04:05:56.382379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-08-24T04:06:45.500429Z","iopub.execute_input":"2021-08-24T04:06:45.500817Z","iopub.status.idle":"2021-08-24T04:06:45.772582Z","shell.execute_reply.started":"2021-08-24T04:06:45.500786Z","shell.execute_reply":"2021-08-24T04:06:45.771443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2021-08-24T04:06:55.534261Z","iopub.execute_input":"2021-08-24T04:06:55.534656Z","iopub.status.idle":"2021-08-24T04:06:55.55292Z","shell.execute_reply.started":"2021-08-24T04:06:55.534608Z","shell.execute_reply":"2021-08-24T04:06:55.551965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\ntrain.to_csv(\"startified_train.csv\", index=False)\n","metadata":{"execution":{"iopub.status.busy":"2021-08-24T04:07:11.350826Z","iopub.execute_input":"2021-08-24T04:07:11.351203Z","iopub.status.idle":"2021-08-24T04:07:23.745723Z","shell.execute_reply.started":"2021-08-24T04:07:11.351172Z","shell.execute_reply":"2021-08-24T04:07:23.744787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _bytes_feature(value):\n  \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n  if isinstance(value, type(tf.constant(0))):\n    value = value.numpy() # BytesList won't unpack a string from an EagerTensor.\n  return tf.train.Feature(bytes_list=tf.train.BytesList(value=[value]))\n\ndef _float_feature(value):\n  \"\"\"Returns a float_list from a float / double.\"\"\"\n  return tf.train.Feature(float_list=tf.train.FloatList(value=[value]))\n\ndef _int64_feature(value):\n  \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n  return tf.train.Feature(int64_list=tf.train.Int64List(value=[value]))\n\ndef serialize_example(image,image_name,label):\n    feature = {\n        'image': _bytes_feature(image),\n        'image_id': _bytes_feature(image_name),\n        'target': _int64_feature(label),\n      }\n    example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n    return example_proto.SerializeToString()","metadata":{"execution":{"iopub.status.busy":"2021-08-24T04:07:23.747068Z","iopub.execute_input":"2021-08-24T04:07:23.747363Z","iopub.status.idle":"2021-08-24T04:07:23.756854Z","shell.execute_reply.started":"2021-08-24T04:07:23.747326Z","shell.execute_reply":"2021-08-24T04:07:23.755935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_real_train_tf_records(fold  = 0):\n    df = train.loc[train['fold']==fold].reset_index(drop=True)\n    tfr_filename = f'/tmp/{DATASET_NAME}/landmark-2021-train-{fold}-{df.shape[0]}.tfrec'\n    with tf.io.TFRecordWriter(tfr_filename) as writer:\n        for i,row in df.iterrows():\n            image_id = row.id\n            target = row.fixed_landmark_id\n            image_path = \"../input/landmark-retrieval-2021/train/{}/{}/{}/{}.jpg\".format(image_id[0],image_id[1],image_id[2],image_id) \n            image_encoded = tf.io.read_file(image_path)\n            image_name = str.encode(image_id)\n            example = serialize_example(image_encoded,image_name,target)\n            writer.write(example)","metadata":{"execution":{"iopub.status.busy":"2021-08-20T10:53:12.444762Z","iopub.execute_input":"2021-08-20T10:53:12.445339Z","iopub.status.idle":"2021-08-20T10:53:12.456471Z","shell.execute_reply.started":"2021-08-20T10:53:12.445293Z","shell.execute_reply":"2021-08-20T10:53:12.455457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if CFG.FOLD == 0:\n    foldrange=range(0,10)\nelif CFG.FOLD==1:\n    foldrange = range(10,20)\nelif CFG.FOLD==2:\n    foldrange = range(20,30)\nelif CFG.FOLD==3:\n    foldrange = range(30,40)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import joblib\n_ = joblib.Parallel(n_jobs=8)(\n        joblib.delayed(create_real_train_tf_records)(fold) for fold in tqdm(foldrange))","metadata":{"execution":{"iopub.status.busy":"2021-08-20T10:53:12.458772Z","iopub.execute_input":"2021-08-20T10:53:12.459313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datetime import datetime\nversion_name = datetime.now().strftime(\"%Y%m%d-%H%M%S\")\nprint(version_name)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp /kaggle/working/train.csv /tmp/{DATASET_NAME}/train.csv\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /tmp/{DATASET_NAME}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!kaggle datasets version -m {version_name} -p /tmp/{DATASET_NAME} -r zip -q","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}