{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import sys\n!cp ../input/rapids/rapids.21.06 /opt/conda/envs/rapids.tar.gz\n!cd /opt/conda/envs/ && tar -xzvf rapids.tar.gz > /dev/null\nsys.path = [\"/opt/conda/envs/rapids/lib/python3.7/site-packages\"] + sys.path\nsys.path = [\"/opt/conda/envs/rapids/lib/python3.7\"] + sys.path\nsys.path = [\"/opt/conda/envs/rapids/lib\"] + sys.path \n!cp /opt/conda/envs/rapids/lib/libxgboost.so /opt/conda/lib/\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-09-13T13:12:51.212086Z","iopub.execute_input":"2021-09-13T13:12:51.212404Z","iopub.status.idle":"2021-09-13T13:15:30.9619Z","shell.execute_reply.started":"2021-09-13T13:12:51.21237Z","shell.execute_reply":"2021-09-13T13:15:30.960604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import operator\nimport gc\nimport pathlib\nimport shutil\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras import backend as K\nfrom scipy import spatial\nimport tensorflow.keras.layers as L\nimport cudf, cuml, cupy\nfrom sklearn.preprocessing import normalize\nimport cv2\nfrom numba import cuda\n\n!pip install ../input/kerasapplications/Keras_Applications-1.0.8-py3-none-any.whl\n!pip install ../input/efficientnetrepo110/efficientnet-1.1.0-py3-none-any.whl\nimport efficientnet.tfkeras as efn\nimport math\nfrom tqdm.notebook import tqdm\nNUM_CLASSES = NUMBER_OF_CLASSES = 81313\nIMAGE_SIZE = [512, 512]\nLR = 0.0001\nEMB_SIZE = 512\nAUTO = tf.data.experimental.AUTOTUNE\nBATCH_SIZE = 8\nefficientnet_size = 7","metadata":{"execution":{"iopub.status.busy":"2021-09-13T13:15:30.964148Z","iopub.execute_input":"2021-09-13T13:15:30.96483Z","iopub.status.idle":"2021-09-13T13:16:33.475982Z","shell.execute_reply.started":"2021-09-13T13:15:30.964783Z","shell.execute_reply":"2021-09-13T13:16:33.475023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ArcMarginProduct(tf.keras.layers.Layer):\n    '''\n    Implements large margin arc distance.\n\n    Reference:\n        https://arxiv.org/pdf/1801.07698.pdf\n        https://github.com/lyakaap/Landmark2019-1st-and-3rd-Place-Solution/\n            blob/master/src/modeling/metric_learning.py\n    '''\n    def __init__(self, n_classes, s=30, m=0.50, easy_margin=False,\n                 ls_eps=0.0, **kwargs):\n\n        super(ArcMarginProduct, self).__init__(**kwargs)\n\n        self.n_classes = n_classes\n        self.s = s\n        self.m = m\n        self.ls_eps = ls_eps\n        self.easy_margin = easy_margin\n        self.cos_m = tf.math.cos(m)\n        self.sin_m = tf.math.sin(m)\n        self.th = tf.math.cos(math.pi - m)\n        self.mm = tf.math.sin(math.pi - m) * m\n\n    def get_config(self):\n\n        config = super().get_config().copy()\n        config.update({\n            'n_classes': self.n_classes,\n            's': self.s,\n            'm': self.m,\n            'ls_eps': self.ls_eps,\n            'easy_margin': self.easy_margin,\n        })\n        return config\n\n    def build(self, input_shape):\n        super(ArcMarginProduct, self).build(input_shape[0])\n\n        self.W = self.add_weight(\n            name='W',\n            shape=(int(input_shape[0][-1]), self.n_classes),\n            initializer='glorot_uniform',\n            dtype='float32',\n            trainable=True,\n            regularizer=None)\n\n    def call(self, inputs):\n        X, y = inputs\n        y = tf.cast(y, dtype=tf.int32)\n        cosine = tf.matmul(\n            tf.math.l2_normalize(X, axis=1),\n            tf.math.l2_normalize(self.W, axis=0)\n        )\n        sine = tf.math.sqrt(1.0 - tf.math.pow(cosine, 2))\n        phi = cosine * self.cos_m - sine * self.sin_m\n        if self.easy_margin:\n            phi = tf.where(cosine > 0, phi, cosine)\n        else:\n            phi = tf.where(cosine > self.th, phi, cosine - self.mm)\n        one_hot = tf.cast(\n            tf.one_hot(y, depth=self.n_classes),\n            dtype=cosine.dtype\n        )\n        if self.ls_eps > 0:\n            one_hot = (1 - self.ls_eps) * one_hot + self.ls_eps / self.n_classes\n\n        output = (one_hot * phi) + ((1.0 - one_hot) * cosine)\n        output *= self.s\n        return output","metadata":{"execution":{"iopub.status.busy":"2021-09-13T13:16:33.478177Z","iopub.execute_input":"2021-09-13T13:16:33.478627Z","iopub.status.idle":"2021-09-13T13:16:33.72036Z","shell.execute_reply.started":"2021-09-13T13:16:33.478583Z","shell.execute_reply":"2021-09-13T13:16:33.719108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class GeM(tf.keras.layers.Layer):\n    def __init__(self, pool_size, init_norm=3.0, normalize=False, **kwargs):\n        self.pool_size = pool_size\n        self.init_norm = init_norm\n        self.normalize = normalize\n\n        super(GeM, self).__init__(**kwargs)\n\n    def get_config(self):\n        config = super().get_config().copy()\n        config.update({\n            'pool_size': self.pool_size,\n            'init_norm': self.init_norm,\n            'normalize': self.normalize,\n        })\n        return config\n\n    def build(self, input_shape):\n        feature_size = input_shape[-1]\n        self.p = self.add_weight(name='norms', shape=(feature_size,),\n                                 initializer=tf.keras.initializers.constant(self.init_norm),\n                                 trainable=True)\n        super(GeM, self).build(input_shape)\n\n    def call(self, inputs):\n        x = inputs\n        x = tf.math.maximum(x, 1e-6)\n        x = tf.pow(x, self.p)\n\n        x = tf.nn.avg_pool(x, self.pool_size, self.pool_size, 'VALID')\n        x = tf.pow(x, 1.0 / self.p)\n\n        if self.normalize:\n            x = tf.nn.l2_normalize(x, 1)\n        return x\n\n    def compute_output_shape(self, input_shape):\n        return tuple([None, input_shape[-1]])","metadata":{"execution":{"iopub.status.busy":"2021-09-13T13:16:33.722602Z","iopub.execute_input":"2021-09-13T13:16:33.723189Z","iopub.status.idle":"2021-09-13T13:16:33.734802Z","shell.execute_reply.started":"2021-09-13T13:16:33.723151Z","shell.execute_reply":"2021-09-13T13:16:33.73369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(size=256, efficientnet_size=0, weights=\"imagenet\", count=0):\n    inp = tf.keras.layers.Input(shape=(size, size, 3), name=\"inp1\")\n    label = tf.keras.layers.Input(shape=(), name=\"inp2\")\n    x = getattr(efn, f\"EfficientNetB{efficientnet_size}\")(\n        weights=weights, include_top=False, input_shape=(size, size, 3))(inp)\n    x = GeM(16)(x)\n    x = tf.keras.layers.Flatten()(x)\n    x = tf.keras.layers.Dense(512, name=\"dense_before_arcface\", kernel_initializer=\"he_normal\")(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = ArcMarginProduct(\n        n_classes=NUM_CLASSES,\n        s=30,\n        m=0.5,\n        name=\"head/arc_margin\",\n        dtype=\"float32\"\n    )([x, label])\n    output = tf.keras.layers.Softmax(dtype=\"float32\")(x)\n    model = tf.keras.Model(inputs=[inp, label], outputs=[output])\n    lr_decayed_fn = tf.keras.experimental.CosineDecay(1e-3, count)\n    opt = tf.optimizers.Adam(learning_rate=1e-4)\n    model.compile(\n        optimizer=opt,\n        loss=[tf.keras.losses.SparseCategoricalCrossentropy()],\n        metrics=[tf.keras.metrics.SparseCategoricalAccuracy()]\n    )\n    return model","metadata":{"execution":{"iopub.status.busy":"2021-09-13T13:17:37.186674Z","iopub.execute_input":"2021-09-13T13:17:37.18701Z","iopub.status.idle":"2021-09-13T13:17:37.197912Z","shell.execute_reply.started":"2021-09-13T13:17:37.186979Z","shell.execute_reply":"2021-09-13T13:17:37.197097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model_for_inference(weights_path: str, efficientnet_size=efficientnet_size):\n    with strategy.scope():\n        base_model = build_model(\n            size=IMAGE_SIZE[0],\n            efficientnet_size=efficientnet_size,\n            weights=None,\n            count=0)\n        base_model.load_weights(weights_path)\n        model = tf.keras.Model(inputs=base_model.get_layer(\"inp1\").input,\n                               outputs=base_model.get_layer(\"dense_before_arcface\").output)\n        return model","metadata":{"execution":{"iopub.status.busy":"2021-09-13T13:17:37.389859Z","iopub.execute_input":"2021-09-13T13:17:37.390265Z","iopub.status.idle":"2021-09-13T13:17:37.396566Z","shell.execute_reply.started":"2021-09-13T13:17:37.390228Z","shell.execute_reply":"2021-09-13T13:17:37.395742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model_B7():\n\n    margin = ArcMarginProduct(\n        n_classes = NUM_CLASSES, \n        s = 64, \n        m = 0.15, \n        name='head/arc_margin', \n        dtype='float32'\n        )\n\n    inp = tf.keras.layers.Input(shape = (512, 512, 3), name = 'inp1')\n    label = tf.keras.layers.Input(shape = (), name = 'inp2')\n    x4 = efn.EfficientNetB7(weights = None, include_top = False)(inp)\n    x = tf.keras.layers.GlobalAveragePooling2D()(x4)\n    x = tf.keras.layers.Dropout(0.3)(x)\n    x = tf.keras.layers.Dense(512)(x)\n    x = margin([x, label])\n\n    output = tf.keras.layers.Softmax(dtype='float32')(x)\n\n    model = tf.keras.models.Model(inputs = [inp, label], outputs = [output])\n\n    opt = tf.keras.optimizers.Adam(learning_rate = LR)\n\n    model.compile(\n        optimizer = opt,\n        loss = [tf.keras.losses.SparseCategoricalCrossentropy()],\n        metrics = [tf.keras.metrics.SparseCategoricalAccuracy()]\n        ) \n\n    return model","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def auto_select_accelerator():\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"Running on TPU:\", tpu.master())\n    except ValueError:\n        strategy = tf.distribute.get_strategy()\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n    return strategy","metadata":{"execution":{"iopub.status.busy":"2021-09-13T13:17:37.656959Z","iopub.execute_input":"2021-09-13T13:17:37.657261Z","iopub.status.idle":"2021-09-13T13:17:37.664848Z","shell.execute_reply.started":"2021-09-13T13:17:37.657233Z","shell.execute_reply":"2021-09-13T13:17:37.6639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_image_embeddings(filepaths, model, return_ids=True):\n    image_paths = [x for x in pathlib.Path(filepaths).rglob('*.jpg')]\n    df = pd.DataFrame(image_paths)\n    df.columns = ['path']\n    df['path'] = df['path'].map(str)\n    df['ids'] = df['path'].map(lambda x:str(x).split('/')[-1].split('.')[0])\n    \n    embeds = []\n    chunk = 5000\n    iterator = np.arange(np.ceil(len(image_paths) / chunk))\n    for j in iterator:\n        a = int(j * chunk)\n        b = int((j + 1) * chunk)\n        image_dataset = get_dataset(df['path'].iloc[a:b])\n        image_embeddings = model.predict(image_dataset)\n        embeds.append(image_embeddings)\n    #del model\n    image_embeddings = np.concatenate(embeds)\n    image_embeddings = normalize(image_embeddings, axis=1)\n    print(f'Our image embeddings shape is {image_embeddings.shape}')\n    del embeds\n    gc.collect()\n    if return_ids:\n        return list(df['ids']), image_embeddings\n    else:\n        return image_embeddings","metadata":{"execution":{"iopub.status.busy":"2021-09-13T13:17:37.908413Z","iopub.execute_input":"2021-09-13T13:17:37.908839Z","iopub.status.idle":"2021-09-13T13:17:37.917367Z","shell.execute_reply.started":"2021-09-13T13:17:37.908805Z","shell.execute_reply":"2021-09-13T13:17:37.91645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"strategy = auto_select_accelerator()\nREPLICAS = strategy.num_replicas_in_sync\nAUTO = tf.data.experimental.AUTOTUNE","metadata":{"execution":{"iopub.status.busy":"2021-09-13T13:17:38.148462Z","iopub.execute_input":"2021-09-13T13:17:38.14876Z","iopub.status.idle":"2021-09-13T13:17:38.153666Z","shell.execute_reply.started":"2021-09-13T13:17:38.148734Z","shell.execute_reply":"2021-09-13T13:17:38.152679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUM_EMBEDDING_DIMENSIONS = 512\nDATASET_DIR = '../input/landmark-recognition-2021/train.csv'\nTEST_IMAGE_DIR = '../input/landmark-recognition-2021/test'\nTRAIN_IMAGE_DIR = '../input/landmark-recognition-2021/train'\nNON_LANDMARK = '../input/google-landmark-2021-validation/valid'\ntf.keras.backend.clear_session()","metadata":{"execution":{"iopub.status.busy":"2021-09-13T13:17:38.62907Z","iopub.execute_input":"2021-09-13T13:17:38.62938Z","iopub.status.idle":"2021-09-13T13:17:38.636204Z","shell.execute_reply.started":"2021-09-13T13:17:38.62935Z","shell.execute_reply":"2021-09-13T13:17:38.635239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to decode our images\ndef decode_image(image_data, image_size=IMAGE_SIZE):\n    image = tf.image.decode_jpeg(image_data, channels = 3)\n    image = tf.image.resize(image, image_size)\n    image = tf.cast(image, tf.float32) /255\n    return image\n\n# Function to read our test image and return image\ndef read_image(image):\n    image = tf.io.read_file(image)\n    image = decode_image(image)\n    return image\n\n# Function to get our dataset that read images\ndef get_dataset(image):\n    dataset = tf.data.Dataset.from_tensor_slices(image)\n    dataset = dataset.map(read_image, num_parallel_calls = AUTO)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO)\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2021-09-13T13:17:39.652005Z","iopub.execute_input":"2021-09-13T13:17:39.65234Z","iopub.status.idle":"2021-09-13T13:17:39.658357Z","shell.execute_reply.started":"2021-09-13T13:17:39.652308Z","shell.execute_reply":"2021-09-13T13:17:39.657558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def topk_(matrix, K, axis=1):\n    if axis == 0:\n        row_index = np.arange(matrix.shape[1 - axis])\n        topk_index = np.argpartition(-matrix, K, axis=axis)[0:K, :]\n        topk_data = matrix[topk_index, row_index]\n        topk_index_sort = np.argsort(-topk_data,axis=axis)\n        topk_data_sort = topk_data[topk_index_sort,row_index]\n        topk_index_sort = topk_index[0:K,:][topk_index_sort,row_index]\n    else:\n        column_index = cupy.arange(matrix.shape[1 - axis])[:, None]\n        topk_index = cupy.argpartition(-matrix, K, axis=axis)[:, 0:K]\n        topk_data = matrix[column_index, topk_index]\n        topk_index_sort = cupy.argsort(-topk_data, axis=axis)\n        topk_data_sort = topk_data[column_index, topk_index_sort]\n        topk_index_sort = topk_index[:,0:K][column_index,topk_index_sort]\n    return topk_data_sort, topk_index_sort","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUM_TO_RERANK = 4\nNUM_PUBLIC_TEST_IMAGES = 10345 # Used to detect if in session or re-run.\n\ndef get_similarities(train_csv, test_directory, train_directory):\n    # Get target dictionary\n    df = pd.read_csv(train_csv)\n    df = df[['id', 'landmark_id']]\n    df.set_index('id', inplace = True)\n    df = df.to_dict()['landmark_id']\n    \n    model = create_model_for_inference(f\"../input/glret21-efficientnetb7-training-s2/fold0.h5\", efficientnet_size=7)\n    \n    test_ids, test_embeddings = get_image_embeddings(test_directory, model, return_ids=True)\n    train_ids, train_embeddings = get_image_embeddings(train_directory, model, return_ids=True)\n    #nonland_embeddings = get_image_embeddings(NON_LANDMARK, model, return_ids=False)\n    tf.keras.backend.clear_session()\n    \n    model = create_model_for_inference(f\"../input/glret21-efficientnetb6-training-f1-s2/fold1.h5\" , efficientnet_size=6)\n    test_embeddings2 = get_image_embeddings(test_directory, model, return_ids=False)\n    train_embeddings2 = get_image_embeddings(train_directory, model, return_ids=False)\n    #nonland_embeddings2 = get_image_embeddings(NON_LANDMARK, model, return_ids=False)\n    tf.keras.backend.clear_session()\n    \n    model = create_model_for_inference(f\"../input/glret21-efficientnetb6-training-f1-s2-ep60/fold1.h5\" , efficientnet_size=6)\n    test_embeddings3 = get_image_embeddings(test_directory, model, return_ids=False)\n    train_embeddings3 = get_image_embeddings(train_directory, model, return_ids=False)\n    #nonland_embeddings3 = get_image_embeddings(NON_LANDMARK, model, return_ids=False)\n    tf.keras.backend.clear_session()\n    \n    model = create_model_for_inference(f\"../input/glret21-efficientnetb7-f2s2/fold2.h5\" , efficientnet_size=7)\n    test_embeddings4 = get_image_embeddings(test_directory, model, return_ids=False)\n    train_embeddings4 = get_image_embeddings(train_directory, model, return_ids=False)\n    nonland_embeddings4 = get_image_embeddings(NON_LANDMARK, model, return_ids=False)\n    tf.keras.backend.clear_session()\n    \n    model = create_model_for_inference(f\"../input/glret21-efficientnetb7-training-f3-s1-p2/fold3.h5\" , efficientnet_size=7)\n    test_embeddings5 = get_image_embeddings(test_directory, model, return_ids=False)\n    train_embeddings5 = get_image_embeddings(train_directory, model, return_ids=False)\n    #nonland_embeddings5 = get_image_embeddings(NON_LANDMARK, model, return_ids=False)\n    \n    model = get_model_B7()\n    model.load_weights('../input/effb7-512-ep12/effb7model512-12.h5')\n    model = tf.keras.models.Model(inputs = model.input[0], outputs = model.layers[-4].output)\n    test_embeddings6 = get_image_embeddings(test_directory, model, return_ids=False)\n    train_embeddings6 = get_image_embeddings(train_directory, model, return_ids=False)\n    \n    test_embeddings = np.concatenate([test_embeddings, test_embeddings2, test_embeddings3, test_embeddings4, test_embeddings5, test_embeddings6], axis=1)\n    train_embeddings = np.concatenate([train_embeddings, train_embeddings2, train_embeddings3, train_embeddings4, train_embeddings5, train_embeddings6], axis=1)\n    #nonland_embeddings = np.concatenate([nonland_embeddings, nonland_embeddings2, nonland_embeddings3, nonland_embeddings4, nonland_embeddings5], axis=1)\n    \n    del test_embeddings2, model, test_embeddings3, test_embeddings4,test_embeddings5,test_embeddings6\n    del train_embeddings2, train_embeddings3,train_embeddings5,train_embeddings6\n    #del nonland_embeddings2, nonland_embeddings3, nonland_embeddings4, nonland_embeddings5\n    \n    gc.collect()\n    train_ids_labels_and_scores = [None] * test_embeddings.shape[0]\n    cuda.select_device(0)\n    cuda.close()\n    cuda.select_device(0)\n    test_embeddings = cupy.asarray(test_embeddings)\n    train_embeddings = cupy.asarray(train_embeddings)\n    train_embeddings4 = cupy.asarray(train_embeddings4)\n    nonland_embeddings4 = cupy.asarray(nonland_embeddings4)\n    train_penalties_list = []\n    \n    for i in range(0, train_embeddings4.shape[0], 128):\n        x = cupy.matmul(train_embeddings4[i:i + 128], nonland_embeddings4.T)\n        x = topk_(x, K=10)[0].mean(axis=1)\n        train_penalties_list.append(x)\n    train_penalties = cupy.concatenate(train_penalties_list)\n    \n    CHUNK = 1024*2\n    test_index = 0\n    CTS = len(test_embeddings)//CHUNK\n    if len(test_embeddings)%CHUNK!=0: CTS += 1\n    for j in range( CTS ):\n        a = j*CHUNK\n        b = (j+1)*CHUNK\n        b = min(b, len(test_embeddings))\n        print('chunk',a,'to',b)\n\n        # COSINE SIMILARITY DISTANCE\n        cts = cupy.matmul(train_embeddings, test_embeddings[a:b].T).T\n        cts -= train_penalties[None, :]*0.9\n        cts = 1.0-cts\n\n        partition = cupy.argpartition(cts, NUM_TO_RERANK)[:, :NUM_TO_RERANK]\n        cts = cupy.asnumpy(cts)\n        partition = cupy.asnumpy(partition)\n        for k in range(b-a):\n            nearest = sorted([(train_ids[p], cts[k][p]) for p in partition[k]], key = lambda x: x[1])\n            train_ids_labels_and_scores[test_index] = [(df[train_id], 1.0 - cosine_distance) for train_id, cosine_distance in nearest]\n            test_index += 1\n        \n    del test_embeddings\n    del train_embeddings\n    #del tf_similarity\n    gc.collect()\n    return test_ids, train_ids_labels_and_scores","metadata":{"execution":{"iopub.status.busy":"2021-09-13T13:17:40.289968Z","iopub.execute_input":"2021-09-13T13:17:40.290283Z","iopub.status.idle":"2021-09-13T13:17:40.305738Z","shell.execute_reply.started":"2021-09-13T13:17:40.290251Z","shell.execute_reply":"2021-09-13T13:17:40.304762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUM_TO_RERANK = 3\nNUM_PUBLIC_TEST_IMAGES = 10345 # Used to detect if in session or re-run.\n\n# This function aggregate top simlarities and make predictions\ndef generate_predictions(test_ids, train_ids_labels_and_scores):\n    targets = []\n    scores = []\n    \n    # Iterate through each test id\n    for test_index, test_id in tqdm(enumerate(test_ids), total=len(test_ids)):\n        aggregate_scores = {}\n        # Iterate through the similar images with their corresponing score for the given test image\n        for target, score in train_ids_labels_and_scores[test_index]:\n            if target not in aggregate_scores:\n                aggregate_scores[target] = 0\n            aggregate_scores[target] += score\n        # Get the best score\n        target, score = max(aggregate_scores.items(), key = operator.itemgetter(1))\n        targets.append(target)\n        scores.append(score)\n        \n    final = pd.DataFrame({'id': test_ids, 'target': targets, 'scores': scores})\n    final['landmarks'] = final['target'].astype(str) + ' ' + final['scores'].astype(str)\n    final[['id', 'landmarks']].to_csv('submission.csv', index = False)\n    return final\n\ndef inference_and_save_submission_csv(train_csv, test_directory, train_directory):\n    image_paths = [x for x in pathlib.Path(test_directory).rglob('*.jpg')]\n    test_len = len(image_paths)\n    if test_len == NUM_PUBLIC_TEST_IMAGES:\n        # Dummy submission\n        shutil.copyfile('../input/landmark-recognition-2021/sample_submission.csv', 'submission.csv')\n        return 'Job Done'\n    else:\n        test_ids, train_ids_labels_and_scores = get_similarities(train_csv, test_directory, train_directory)\n        final = generate_predictions(test_ids, train_ids_labels_and_scores)\n        return final\n    \nfinal = inference_and_save_submission_csv(DATASET_DIR, TEST_IMAGE_DIR, TRAIN_IMAGE_DIR)","metadata":{"execution":{"iopub.status.busy":"2021-09-13T13:17:40.900096Z","iopub.execute_input":"2021-09-13T13:17:40.900406Z","iopub.status.idle":"2021-09-13T13:22:14.113105Z","shell.execute_reply.started":"2021-09-13T13:17:40.900375Z","shell.execute_reply":"2021-09-13T13:22:14.112336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}