{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Baseline submission\n\nA notebook to generate a valid submission. Implements three local feature/matcher methods: LoFTR, DISK, and KeyNetAffNetHardNet.\n\nRemember to enable a GPU accelerator and disable internet access, then press \"submit\" on the right pane.","metadata":{}},{"cell_type":"code","source":"!cp -r /kaggle/input/hierarchical-localization/ /kaggle/working/\n%cd /kaggle/working/hierarchical-localization/Hierarchical-Localization\n\nimport os\n!python -m pip install -e .","metadata":{"execution":{"iopub.status.busy":"2023-06-01T04:57:18.304836Z","iopub.execute_input":"2023-06-01T04:57:18.305154Z","iopub.status.idle":"2023-06-01T04:58:08.025336Z","shell.execute_reply.started":"2023-06-01T04:57:18.305103Z","shell.execute_reply":"2023-06-01T04:58:08.024086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir -p /root/.cache/torch/hub/netvlad/\n!mkdir -p /root/.cache/torch/hub/checkpoints","metadata":{"execution":{"iopub.status.busy":"2023-06-01T04:58:08.027693Z","iopub.execute_input":"2023-06-01T04:58:08.028013Z","iopub.status.idle":"2023-06-01T04:58:09.899621Z","shell.execute_reply.started":"2023-06-01T04:58:08.027979Z","shell.execute_reply":"2023-06-01T04:58:09.898239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp /kaggle/input/loftr-weights/loftr_outdoor.ckpt /root/.cache/torch/hub/checkpoints/loftr_outdoor.ckpt\n!cp /kaggle/working/hierarchical-localization/Hierarchical-Localization/third_party/netvlad/VGG16-NetVLAD-Pitts30K.mat /root/.cache/torch/hub/netvlad/VGG16-NetVLAD-Pitts30K.mat","metadata":{"execution":{"iopub.status.busy":"2023-06-01T04:58:09.901753Z","iopub.execute_input":"2023-06-01T04:58:09.902388Z","iopub.status.idle":"2023-06-01T04:58:13.540021Z","shell.execute_reply.started":"2023-06-01T04:58:09.902340Z","shell.execute_reply":"2023-06-01T04:58:13.538646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tqdm, tqdm.notebook\ntqdm.tqdm = tqdm.notebook.tqdm  # notebook-friendly progress bars\nfrom pathlib import Path\n\nfrom hloc import extract_features, match_features, reconstruction, visualization, pairs_from_exhaustive, pairs_from_retrieval, match_dense\nfrom hloc.visualization import plot_images, read_image\nfrom hloc.utils import viz_3d","metadata":{"execution":{"iopub.status.busy":"2023-06-01T04:58:13.544225Z","iopub.execute_input":"2023-06-01T04:58:13.545096Z","iopub.status.idle":"2023-06-01T04:58:16.778579Z","shell.execute_reply.started":"2023-06-01T04:58:13.545059Z","shell.execute_reply":"2023-06-01T04:58:16.777509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-06-01T04:58:16.780061Z","iopub.execute_input":"2023-06-01T04:58:16.780766Z","iopub.status.idle":"2023-06-01T04:58:16.785149Z","shell.execute_reply.started":"2023-06-01T04:58:16.780733Z","shell.execute_reply":"2023-06-01T04:58:16.784071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get data from csv.\nsrc = '/kaggle/input/image-matching-challenge-2023'\n\ndata_dict = {}\nwith open(f'{src}/sample_submission.csv', 'r') as f:\n    for i, l in enumerate(f):\n        # Skip header.\n        if l and i > 0:\n            image, dataset, scene, _, _ = l.strip().split(',')\n            if dataset not in data_dict:\n                data_dict[dataset] = {}\n            if scene not in data_dict[dataset]:\n                data_dict[dataset][scene] = []\n            data_dict[dataset][scene].append(image)","metadata":{"execution":{"iopub.status.busy":"2023-06-01T05:03:54.532706Z","iopub.execute_input":"2023-06-01T05:03:54.533464Z","iopub.status.idle":"2023-06-01T05:03:54.541299Z","shell.execute_reply.started":"2023-06-01T05:03:54.533421Z","shell.execute_reply":"2023-06-01T05:03:54.540202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in data_dict:\n    for scene in data_dict[dataset]:\n        print(f'{dataset} / {scene} -> {len(data_dict[dataset][scene])} images')","metadata":{"execution":{"iopub.status.busy":"2023-06-01T05:03:55.344089Z","iopub.execute_input":"2023-06-01T05:03:55.345024Z","iopub.status.idle":"2023-06-01T05:03:55.351977Z","shell.execute_reply.started":"2023-06-01T05:03:55.344971Z","shell.execute_reply":"2023-06-01T05:03:55.350780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def arr_to_str(a):\n    return ';'.join([str(x) for x in a.reshape(-1)])","metadata":{"execution":{"iopub.status.busy":"2023-06-01T05:03:56.210335Z","iopub.execute_input":"2023-06-01T05:03:56.210713Z","iopub.status.idle":"2023-06-01T05:03:56.216314Z","shell.execute_reply.started":"2023-06-01T05:03:56.210678Z","shell.execute_reply":"2023-06-01T05:03:56.215011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to create a submission file.\ndef create_submission(out_results, data_dict):\n    with open(f'submission.csv', 'w') as f:\n        f.write('image_path,dataset,scene,rotation_matrix,translation_vector\\n')\n        for dataset in data_dict:\n            if dataset in out_results:\n                res = out_results[dataset]\n            else:\n                res = {}\n            for scene in data_dict[dataset]:\n                if scene in res:\n                    scene_res = res[scene]\n                else:\n                    scene_res = {\"R\":{}, \"t\":{}}\n                for image in data_dict[dataset][scene]:\n                    if image in scene_res:\n                        print (image)\n                        R = scene_res[image]['R'].reshape(-1)\n                        T = scene_res[image]['t'].reshape(-1)\n                    else:\n                        R = np.eye(3).reshape(-1)\n                        T = np.zeros((3))\n                    f.write(f'{image},{dataset},{scene},{arr_to_str(R)},{arr_to_str(T)}\\n')","metadata":{"execution":{"iopub.status.busy":"2023-06-01T05:03:56.368026Z","iopub.execute_input":"2023-06-01T05:03:56.368657Z","iopub.status.idle":"2023-06-01T05:03:56.381076Z","shell.execute_reply.started":"2023-06-01T05:03:56.368618Z","shell.execute_reply":"2023-06-01T05:03:56.379975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #By Jesper Andersson https://www.kaggle.com/code/jesperandersson/visualize-superglue-sfm-reconstruction\nimport gc\nfrom copy import deepcopy\n\nEXHAUSTIVE_IF_LESS = 60\n\ngc.collect()\nout_results = {}\n\ndatasets = []\nfor dataset in data_dict:\n    datasets.append(dataset)\n\nfor dataset in datasets:\n    try:\n        print(dataset)\n        if dataset not in out_results:\n            out_results[dataset] = {}\n        for scene in data_dict[dataset]:\n            out_results[dataset][scene] = {}\n            images = Path(f'{src}/test/{dataset}/{scene}/images')\n\n            outputs = Path(f'outputs/sfm/{dataset}-{scene}')\n            sfm_pairs = outputs / 'pairs-netvlad.txt' #pairs-netvlad\n            sfm_dir = outputs / 'sfm_ens'\n\n            retrieval_conf = extract_features.confs['netvlad']\n            feature_conf = extract_features.confs['superpoint_inloc']\n            matcher_conf = match_features.confs['superglue']\n            matcher_conf_dense = match_dense.confs['loftr_superpoint']\n            matcher_conf['model']['max_keypoints'] = 8192 \n            \n\n            retrieval_path = extract_features.main(retrieval_conf, images, outputs)\n\n            if len(os.listdir(images)) > EXHAUSTIVE_IF_LESS:\n                pairs_from_retrieval.main(retrieval_path, sfm_pairs, num_matched=5)\n            else:\n                pairs_from_exhaustive.main(sfm_pairs, features=retrieval_path)\n            feature_path = extract_features.main(feature_conf, images, outputs)\n            feature_path_loftr, match_path_loftr = match_dense.main(matcher_conf_dense, sfm_pairs, images, outputs, max_kps=4096)\n            match_path = match_features.main(matcher_conf, sfm_pairs, feature_conf['output'], outputs)\n\n            mapper_options = {'min_model_size': 3}\n            model = reconstruction.main(sfm_dir, images, sfm_pairs, feature_path, match_path, verbose=False, mapper_options=mapper_options)\n            model_dense = reconstruction.main(sfm_dir, images, sfm_pairs, feature_path_loftr, match_path_loftr, verbose=False, mapper_options=mapper_options)\n\n            for (k_, im_loftr) in model_dense.images.items():\n                key1 = f'{dataset}/{scene}/images/{im_loftr.name}'\n                out_results[dataset][scene][key1] = {}\n                out_results[dataset][scene][key1][\"R\"] = im_loftr.rotmat() \n                out_results[dataset][scene][key1][\"t\"] = np.array(im_loftr.tvec) \n\n            #visualization.visualize_sfm_2d(model, images, color_by='track_length', n=5)\n            for (k_, im) in model.images.items():\n                \n                key1 = f'{dataset}/{scene}/images/{im.name}'\n                if key1 in out_results.keys():\n                    out_results[dataset][scene][key1][\"R\"] += im.rotmat() \n                    out_results[dataset][scene][key1][\"R\"] /= 2 \n                    \n                    out_results[dataset][scene][key1][\"t\"] += np.array(im.tvec) \n                    out_results[dataset][scene][key1][\"t\"] /= 2 \n\n                    \n                else:\n                    out_results[dataset][scene][key1] = {}\n                    out_results[dataset][scene][key1][\"R\"] = im.rotmat()\n                    out_results[dataset][scene][key1][\"t\"] = np.array(im.tvec) \n    except:\n        pass","metadata":{"execution":{"iopub.status.busy":"2023-06-01T05:04:11.328662Z","iopub.execute_input":"2023-06-01T05:04:11.329159Z","iopub.status.idle":"2023-06-01T05:04:11.552488Z","shell.execute_reply.started":"2023-06-01T05:04:11.329083Z","shell.execute_reply":"2023-06-01T05:04:11.550392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd /kaggle/working","metadata":{"execution":{"iopub.status.busy":"2023-06-01T05:04:14.177783Z","iopub.execute_input":"2023-06-01T05:04:14.178481Z","iopub.status.idle":"2023-06-01T05:04:14.185636Z","shell.execute_reply.started":"2023-06-01T05:04:14.178440Z","shell.execute_reply":"2023-06-01T05:04:14.184409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_submission(out_results, data_dict)","metadata":{"execution":{"iopub.status.busy":"2023-06-01T05:04:14.616069Z","iopub.execute_input":"2023-06-01T05:04:14.616828Z","iopub.status.idle":"2023-06-01T05:04:14.622896Z","shell.execute_reply.started":"2023-06-01T05:04:14.616787Z","shell.execute_reply":"2023-06-01T05:04:14.621642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Evaluation metric.\n\n# @dataclass\n# class Camera:\n#     rotmat: np.array\n#     tvec: np.array\n\n# def quaternion_from_matrix(matrix):\n#     M = np.array(matrix, dtype=np.float64, copy=False)[:4, :4]\n#     m00 = M[0, 0]\n#     m01 = M[0, 1]\n#     m02 = M[0, 2]\n#     m10 = M[1, 0]\n#     m11 = M[1, 1]\n#     m12 = M[1, 2]\n#     m20 = M[2, 0]\n#     m21 = M[2, 1]\n#     m22 = M[2, 2]\n\n#     # Symmetric matrix K.\n#     K = np.array([[m00 - m11 - m22, 0.0, 0.0, 0.0],\n#                   [m01 + m10, m11 - m00 - m22, 0.0, 0.0],\n#                   [m02 + m20, m12 + m21, m22 - m00 - m11, 0.0],\n#                   [m21 - m12, m02 - m20, m10 - m01, m00 + m11 + m22]])\n#     K /= 3.0\n\n#     # Quaternion is eigenvector of K that corresponds to largest eigenvalue.\n#     w, V = np.linalg.eigh(K)\n#     q = V[[3, 0, 1, 2], np.argmax(w)]\n\n#     if q[0] < 0.0:\n#         np.negative(q, q)\n#     return q\n\n# def evaluate_R_t(R_gt, t_gt, R, t, eps=1e-15):\n#     t = t.flatten()\n#     t_gt = t_gt.flatten()\n\n#     q_gt = quaternion_from_matrix(R_gt)\n#     q = quaternion_from_matrix(R)\n#     q = q / (np.linalg.norm(q) + eps)\n#     q_gt = q_gt / (np.linalg.norm(q_gt) + eps)\n#     loss_q = np.maximum(eps, (1.0 - np.sum(q * q_gt)**2))\n#     err_q = np.arccos(1 - 2 * loss_q)\n\n#     GT_SCALE = np.linalg.norm(t_gt)\n#     t = GT_SCALE * (t / (np.linalg.norm(t) + eps))\n#     err_t = min(np.linalg.norm(t_gt - t), np.linalg.norm(t_gt + t))\n    \n#     return np.degrees(err_q), err_t\n\n# def compute_dR_dT(R1, T1, R2, T2):\n#     '''Given absolute (R, T) pairs for two cameras, compute the relative pose difference, from the first.'''\n    \n#     dR = np.dot(R2, R1.T)\n#     dT = T2 - np.dot(dR, T1)\n#     return dR, dT\n\n# def compute_mAA(err_q, err_t, ths_q, ths_t):\n#     '''Compute the mean average accuracy over a set of thresholds. Additionally returns the metric only over rotation and translation.'''\n\n#     acc, acc_q, acc_t = [], [], []\n#     for th_q, th_t in zip(ths_q, ths_t):\n#         cur_acc_q = (err_q <= th_q)\n#         cur_acc_t = (err_t <= th_t)\n#         cur_acc = cur_acc_q & cur_acc_t\n        \n#         acc.append(cur_acc.astype(np.float32).mean())\n#         acc_q.append(cur_acc_q.astype(np.float32).mean())\n#         acc_t.append(cur_acc_t.astype(np.float32).mean())\n#     return np.array(acc), np.array(acc_q), np.array(acc_t)\n\n# def dict_from_csv(csv_path, has_header):\n#     csv_dict = {}\n#     with open(csv_path, 'r') as f:\n#         for i, l in enumerate(f):\n#             if has_header and i == 0:\n#                 continue\n#             if l:\n#                 image, dataset, scene, R_str, T_str = l.strip().split(',')\n#                 R = np.fromstring(R_str.strip(), sep=';').reshape(3, 3)\n#                 T = np.fromstring(T_str.strip(), sep=';')\n#                 if dataset not in csv_dict:\n#                     csv_dict[dataset] = {}\n#                 if scene not in csv_dict[dataset]:\n#                     csv_dict[dataset][scene] = {}\n#                 csv_dict[dataset][scene][image] = Camera(rotmat=R, tvec=T)\n#     return csv_dict\n\n# def eval_submission(submission_csv_path, ground_truth_csv_path, rotation_thresholds_degrees_dict, translation_thresholds_meters_dict, verbose=False):\n#     '''Compute final metric given submission and ground truth files. Thresholds are specified per dataset.'''\n\n#     submission_dict = dict_from_csv(submission_csv_path, has_header=True)\n#     gt_dict = dict_from_csv(ground_truth_csv_path, has_header=True)\n\n#     # Check that all necessary keys exist in the submission file\n# #     for dataset in gt_dict:\n# #         assert dataset in submission_dict, f'Unknown dataset: {dataset}'\n# #         for scene in gt_dict[dataset]:\n# #             assert scene in submission_dict[dataset], f'Unknown scene: {dataset}->{scene}'\n# #             for image in gt_dict[dataset][scene]:\n# #                 assert image in submission_dict[dataset][scene], f'Unknown image: {dataset}->{scene}->{image}'\n\n#     # Iterate over all the scenes\n#     if verbose:\n#         t = time()\n#         print('*** METRICS ***')\n\n#     metrics_per_dataset = []\n#     for dataset in gt_dict:\n#         if dataset not in submission_dict:\n#             continue\n#         metrics_per_scene = []\n#         for scene in gt_dict[dataset]:\n#             err_q_all = []\n#             err_t_all = []\n#             images = [camera for camera in gt_dict[dataset][scene]]\n#             # Process all pairs in a scene\n#             for i in range(len(images)):\n#                 for j in range(i + 1, len(images)):\n#                     gt_i = gt_dict[dataset][scene][images[i]]\n#                     gt_j = gt_dict[dataset][scene][images[j]]\n#                     dR_gt, dT_gt = compute_dR_dT(gt_i.rotmat, gt_i.tvec, gt_j.rotmat, gt_j.tvec)\n\n#                     pred_i = submission_dict[dataset][scene][images[i]]\n#                     pred_j = submission_dict[dataset][scene][images[j]]\n#                     dR_pred, dT_pred = compute_dR_dT(pred_i.rotmat, pred_i.tvec, pred_j.rotmat, pred_j.tvec)\n\n#                     err_q, err_t = evaluate_R_t(dR_gt, dT_gt, dR_pred, dT_pred)\n#                     err_q_all.append(err_q)\n#                     err_t_all.append(err_t)\n\n#             mAA, mAA_q, mAA_t = compute_mAA(err_q=err_q_all,\n#                                             err_t=err_t_all,\n#                                             ths_q=rotation_thresholds_degrees_dict[(dataset, scene)],\n#                                             ths_t=translation_thresholds_meters_dict[(dataset, scene)])\n#             if verbose:\n#                 print(f'{dataset} / {scene} ({len(images)} images, {len(err_q_all)} pairs) -> mAA={np.mean(mAA):.06f}, mAA_q={np.mean(mAA_q):.06f}, mAA_t={np.mean(mAA_t):.06f}')\n#             metrics_per_scene.append(np.mean(mAA))\n\n#         metrics_per_dataset.append(np.mean(metrics_per_scene))\n#         if verbose:\n#             print(f'{dataset} -> mAA={np.mean(metrics_per_scene):.06f}')\n#             print()\n\n#     if verbose:\n#         print(f'Final metric -> mAA={np.mean(metrics_per_dataset):.06f} (t: {time() - t} sec.)')\n#         print()\n\n#     return np.mean(metrics_per_dataset)","metadata":{"execution":{"iopub.status.busy":"2023-06-01T05:04:15.421119Z","iopub.execute_input":"2023-06-01T05:04:15.421681Z","iopub.status.idle":"2023-06-01T05:04:15.432604Z","shell.execute_reply.started":"2023-06-01T05:04:15.421642Z","shell.execute_reply":"2023-06-01T05:04:15.431516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Set rotation thresholds per scene.\n\n# rotation_thresholds_degrees_dict = {\n#     **{('haiper', scene): np.linspace(1, 10, 10) for scene in ['bike', 'chairs', 'fountain']},\n#     **{('heritage', scene): np.linspace(1, 10, 10) for scene in ['cyprus', 'dioscuri']},\n#     **{('heritage', 'wall'): np.linspace(0.2, 10, 10)},\n#     **{('urban', 'kyiv-puppet-theater'): np.linspace(1, 10, 10)},\n# }\n\n# translation_thresholds_meters_dict = {\n#     **{('haiper', scene): np.geomspace(0.05, 0.5, 10) for scene in ['bike', 'chairs', 'fountain']},\n#     **{('heritage', scene): np.geomspace(0.1, 2, 10) for scene in ['cyprus', 'dioscuri']},\n#     **{('heritage', 'wall'): np.geomspace(0.05, 1, 10)},\n#     **{('urban', 'kyiv-puppet-theater'): np.geomspace(0.5, 5, 10)},\n# }","metadata":{"execution":{"iopub.status.busy":"2023-06-01T05:04:15.592808Z","iopub.execute_input":"2023-06-01T05:04:15.593566Z","iopub.status.idle":"2023-06-01T05:04:15.598898Z","shell.execute_reply.started":"2023-06-01T05:04:15.593527Z","shell.execute_reply":"2023-06-01T05:04:15.597498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Generate and evaluate a random submission.\n\n# src = '/kaggle/input/image-matching-challenge-2023'\n\n# # Note that the fields were reordered between dataset versions so you need to be careful.\n# # Here we regenerate the ground truth file using the submission format, which lists the image path first.\n# # with open(f'{src}/train/train_labels.csv', 'r') as fr, open('submission.csv', 'w') as fw:\n# #     for i, l in enumerate(fr):\n# #         if i == 0:\n# #             fw.write('image_path,dataset,scene,rotation_matrix,translation_vector\\n')\n# #         else:\n# #             dataset, scene, image, _, _ = l.strip().split(',')\n# #             R = np.random.rand(9)\n# #             T = np.random.rand(3)\n# #             fw.write(f'{image},{dataset},{scene},{arr_to_str(R)},{arr_to_str(T)}\\n')\n\n# with open(f'{src}/train/train_labels.csv', 'r') as fr, open('ground_truth.csv', 'w') as fw:\n#     for i, l in enumerate(fr):\n#         if i == 0:\n#             fw.write('image_path,dataset,scene,rotation_matrix,translation_vector\\n')\n#         else:\n#             dataset, scene, image, R, T = l.strip().split(',')\n#             fw.write(f'{image},{dataset},{scene},{R},{T}\\n')\n\n# eval_submission(submission_csv_path='submission.csv',\n#                 ground_truth_csv_path='ground_truth.csv',\n#                 rotation_thresholds_degrees_dict=rotation_thresholds_degrees_dict,\n#                 translation_thresholds_meters_dict=translation_thresholds_meters_dict,\n#                 verbose=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-01T05:04:15.751731Z","iopub.execute_input":"2023-06-01T05:04:15.752449Z","iopub.status.idle":"2023-06-01T05:04:15.757792Z","shell.execute_reply.started":"2023-06-01T05:04:15.752409Z","shell.execute_reply":"2023-06-01T05:04:15.756677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -r /kaggle/working/hierarchical-localization","metadata":{"execution":{"iopub.status.busy":"2023-06-01T05:04:15.918363Z","iopub.execute_input":"2023-06-01T05:04:15.919357Z","iopub.status.idle":"2023-06-01T05:04:17.043356Z","shell.execute_reply.started":"2023-06-01T05:04:15.919316Z","shell.execute_reply":"2023-06-01T05:04:17.041853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}