{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":71885,"databundleVersionId":8069805,"sourceType":"competition"},{"sourceId":7884485,"sourceType":"datasetVersion","datasetId":4628051},{"sourceId":7884725,"sourceType":"datasetVersion","datasetId":4628331},{"sourceId":4534,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":3326},{"sourceId":17191,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":14317},{"sourceId":17555,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":14611}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**The goal of this competition** is to construct precise 3D maps using sets of images in diverse scenarios and environments","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"# Exploratory Data Analysis","metadata":{}},{"cell_type":"markdown","source":"🏛️ **Phototourism and historical preservation:** different viewpoints, sensor types, time of day/year, and occlusions. Ancient historical sites add a unique set of challenges\n\n☀️ **Night vs day and temporal changes:** combination of day and night photographs, including poor lighting, or photographs taken months or years apart, in different weather\n\n**✈️ Aerial and mixed aerial-ground:** images from drones, featuring arbitrary in-plane rotations, matched against similar images and also images taken from the ground\n\n**♻️ Repeated structures:** symmetrical objects require details to disambiguate perspective\n\n**🌲 Natural environments:** highly non-regular structures such as trees and foliage\n\n**🪞 Transparencies and reflections:** objects like glassware are lacking in texture and create reflections and specularities which pose a different set of problems","metadata":{}},{"cell_type":"markdown","source":"# Install & Import dependencies","metadata":{}},{"cell_type":"code","source":"!pip install -q mediapy\n%cd /kaggle/working/\n!rm -rf /kaggle/working/Hierarchical-Localization\n!git clone --quiet --recursive https://github.com/cvg/Hierarchical-Localization/\n%cd /kaggle/working/Hierarchical-Localization\n!pip install -e .\n\nfrom hloc import extract_features, match_features, reconstruction, visualization, pairs_from_exhaustive\nfrom hloc.visualization import plot_images, read_image\nfrom hloc.utils import viz_3d\n\n%cd /kaggle/working/\nfrom pathlib import Path\n\nimport cv2\nimport mediapy\nimport pandas as pd\nimport plotly.express as px\nimport pycolmap\n","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:13:13.099901Z","iopub.execute_input":"2024-03-26T22:13:13.100259Z","iopub.status.idle":"2024-03-26T22:14:16.965889Z","shell.execute_reply.started":"2024-03-26T22:13:13.100229Z","shell.execute_reply":"2024-03-26T22:14:16.965060Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd \n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/imc2024-packages-lightglue-rerun-kornia'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n!pip install --no-index /kaggle/input/imc2024-packages-lightglue-rerun-kornia/* --no-deps\n!mkdir -p /root/.cache/torch/hub/checkpoints\n!cp /kaggle/input/aliked/pytorch/aliked-n16/1/* /root/.cache/torch/hub/checkpoints/\n!cp /kaggle/input/lightglue/pytorch/aliked/1/* /root/.cache/torch/hub/checkpoints/\n!cp /kaggle/input/lightglue/pytorch/aliked/1/aliked_lightglue.pth /root/.cache/torch/hub/checkpoints/aliked_lightglue_v0-1_arxiv-pth\n","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:14:16.967780Z","iopub.execute_input":"2024-03-26T22:14:16.968260Z","iopub.status.idle":"2024-03-26T22:14:24.655589Z","shell.execute_reply.started":"2024-03-26T22:14:16.968232Z","shell.execute_reply":"2024-03-26T22:14:24.653846Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nimport os\nfrom tqdm import tqdm\nfrom time import time, sleep\nfrom fastprogress import progress_bar\nimport gc\nimport numpy as np\nimport h5py\nfrom IPython.display import clear_output\nfrom collections import defaultdict\nfrom copy import deepcopy\n\n# CV/MLe\nimport cv2\nimport torch\nimport torch.nn.functional as F\nimport kornia as K\nimport kornia.feature as KF\nfrom PIL import Image\nfrom transformers import AutoImageProcessor, AutoModel\n\n# 3D reconstruction\nimport pycolmap\n\n# Data importing into colmap\nimport sys\nsys.path.append('/kaggle/input/colmap-db-import')\nfrom database import *\nfrom h5_to_db import *\n","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:14:24.657707Z","iopub.execute_input":"2024-03-26T22:14:24.658096Z","iopub.status.idle":"2024-03-26T22:14:40.006427Z","shell.execute_reply.started":"2024-03-26T22:14:24.658058Z","shell.execute_reply":"2024-03-26T22:14:40.005477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" # Dataset Overview","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- `[train/test]/*/*/images`: A batch of images all taken near the same location. Some of training datasets may also contain a folder named images_full with additional images. The published test folder comprises a subset of the church scene from train and is provided solely for example purposes. The training data usually has a sequential capture ordering and significant image-to-image content overlap while the test set has limited image-to-image overlap and the image ordering is randomized.\n\n- `train/*/*/smf`: A 3-D reconstruction for this batch of images, which can be opened with colmap, the 3-D structure-from-motion library bundled with this competition.\n\n- `train/*/*/LICENSE.txt`: The license for this dataset.\n\n- `train/train_labels.csv`: A list of images in these datasets, with ground truths.","metadata":{}},{"cell_type":"markdown","source":"# Label details\n- `dataset`: The unique identifier for the dataset.\n- `scene`: The unique identifier for the scene.\n- `image_path`: The image filename, including the path.\n- `rotation_matrix`: The first target column. A 3x3 matrix, flattened into a vector in row-major convention, with values separated by `;`.\n- `translation_vector`: The second target column. A 3-D dimensional vector, with values separated by ;.","metadata":{}},{"cell_type":"code","source":"train_labels = pd.read_csv(\"/kaggle/input/image-matching-challenge-2024/train/train_labels.csv\")\ntrain_labels","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:14:40.009994Z","iopub.execute_input":"2024-03-26T22:14:40.011244Z","iopub.status.idle":"2024-03-26T22:14:40.059740Z","shell.execute_reply.started":"2024-03-26T22:14:40.011208Z","shell.execute_reply":"2024-03-26T22:14:40.058629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Relationship between datasets and scenes","metadata":{}},{"cell_type":"code","source":"train_labels.groupby(\"dataset\")[\"scene\"].nunique()","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:14:40.060847Z","iopub.execute_input":"2024-03-26T22:14:40.061167Z","iopub.status.idle":"2024-03-26T22:14:40.198658Z","shell.execute_reply.started":"2024-03-26T22:14:40.061140Z","shell.execute_reply":"2024-03-26T22:14:40.197800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Distribution of the datasets","metadata":{}},{"cell_type":"code","source":"dataset_counts = train_labels[\"dataset\"].value_counts()\n\nfig = px.pie(values=dataset_counts.values, names=dataset_counts.index)\nfig.update_traces(textposition='inside', textfont_size=14)\nfig.update_layout(\n    title={\n        'text': \"Pie distribution of dataset images\",\n        'y':0.95,\n        'x':0.5,\n        'xanchor': 'center',\n        'yanchor': 'top'\n    },\n    legend_title_text='Dataset names:'\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:14:40.199965Z","iopub.execute_input":"2024-03-26T22:14:40.200625Z","iopub.status.idle":"2024-03-26T22:14:41.787873Z","shell.execute_reply.started":"2024-03-26T22:14:40.200590Z","shell.execute_reply":"2024-03-26T22:14:41.786843Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  Categories in the train dataset","metadata":{}},{"cell_type":"code","source":"train_categories = pd.read_csv(\"/kaggle/input/image-matching-challenge-2024/train/categories.csv\")\n\n# From comma separated list of categories for each dataset\n# To one dataset & category per row\ntrain_categories[\"category\"] = train_categories[\"categories\"].str.split(\";\")\ntrain_categories = train_categories.explode(\"category\")","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:14:41.789160Z","iopub.execute_input":"2024-03-26T22:14:41.789466Z","iopub.status.idle":"2024-03-26T22:14:41.813377Z","shell.execute_reply.started":"2024-03-26T22:14:41.789439Z","shell.execute_reply":"2024-03-26T22:14:41.812494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Categories distribution","metadata":{}},{"cell_type":"code","source":"category_counts = train_categories[\"category\"].value_counts()\n\nfig = px.pie(values=dataset_counts.values, names=category_counts.index)\nfig.update_traces(textposition='inside', textfont_size=14)\nfig.update_layout(\n    title={\n        'text': \"Pie distribution of categories of datasets\",\n        'y':0.95,\n        'x':0.5,\n        'xanchor': 'center',\n        'yanchor': 'top'\n    },\n    legend_title_text='Category names:'\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:14:41.814686Z","iopub.execute_input":"2024-03-26T22:14:41.815044Z","iopub.status.idle":"2024-03-26T22:14:41.884292Z","shell.execute_reply.started":"2024-03-26T22:14:41.815003Z","shell.execute_reply":"2024-03-26T22:14:41.883403Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Relationship between a scene  and a category","metadata":{}},{"cell_type":"code","source":"fig = px.sunburst(train_categories, path=['scene', 'category'])\nfig.update_layout(\n    title={\n        'text': \"Scene and category relation\",\n        'y':0.95,\n        'x':0.5,\n        'xanchor': 'center',\n        'yanchor': 'top'\n    }\n)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:14:41.885297Z","iopub.execute_input":"2024-03-26T22:14:41.885573Z","iopub.status.idle":"2024-03-26T22:14:41.988195Z","shell.execute_reply.started":"2024-03-26T22:14:41.885548Z","shell.execute_reply":"2024-03-26T22:14:41.987421Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploring each dataset","metadata":{}},{"cell_type":"code","source":"def explore(split: str, dataset: str, plot_image_limit: int = 12) -> None:\n    path = Path(\"/kaggle/input/image-matching-challenge-2024\") / split / dataset\n    images_path = path / \"images\"\n    smf_path = path / \"smf\"    \n\n    images = [cv2.cvtColor(cv2.imread(str(p)), cv2.COLOR_BGR2RGB) for p in list(images_path.glob(\"*\"))[:plot_image_limit]]\n    mediapy.show_images(images, height=300, columns=3)\n    \n    if split != \"test\":\n        rec_gt = pycolmap.Reconstruction(smf_path)\n\n        fig = viz_3d.init_figure()\n        viz_3d.plot_reconstruction(fig, rec_gt, cameras=False, color='rgba(227,168,30,0.5)', name=\"Ground Truth\", cs=5)\n        fig.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:14:41.991306Z","iopub.execute_input":"2024-03-26T22:14:41.991588Z","iopub.status.idle":"2024-03-26T22:14:41.998182Z","shell.execute_reply.started":"2024-03-26T22:14:41.991564Z","shell.execute_reply":"2024-03-26T22:14:41.997244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Church**","metadata":{}},{"cell_type":"code","source":"explore(split=\"train\", dataset=\"church\")","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:14:41.999253Z","iopub.execute_input":"2024-03-26T22:14:41.999574Z","iopub.status.idle":"2024-03-26T22:14:53.486170Z","shell.execute_reply.started":"2024-03-26T22:14:41.999543Z","shell.execute_reply":"2024-03-26T22:14:53.485162Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explore(split=\"test\", dataset=\"church\")","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:14:53.487601Z","iopub.execute_input":"2024-03-26T22:14:53.487931Z","iopub.status.idle":"2024-03-26T22:14:54.437188Z","shell.execute_reply.started":"2024-03-26T22:14:53.487901Z","shell.execute_reply":"2024-03-26T22:14:54.436050Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Dioscuri**","metadata":{}},{"cell_type":"code","source":"explore(split=\"train\", dataset=\"dioscuri\")","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:14:54.438516Z","iopub.execute_input":"2024-03-26T22:14:54.438877Z","iopub.status.idle":"2024-03-26T22:14:58.956258Z","shell.execute_reply.started":"2024-03-26T22:14:54.438838Z","shell.execute_reply":"2024-03-26T22:14:58.955203Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explore(split=\"test\", dataset=\"dioscuri\")","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:14:58.957691Z","iopub.execute_input":"2024-03-26T22:14:58.958021Z","iopub.status.idle":"2024-03-26T22:14:58.963798Z","shell.execute_reply.started":"2024-03-26T22:14:58.957981Z","shell.execute_reply":"2024-03-26T22:14:58.962944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Lizard**","metadata":{}},{"cell_type":"code","source":"explore(split=\"train\", dataset=\"lizard\")","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:14:58.965023Z","iopub.execute_input":"2024-03-26T22:14:58.965820Z","iopub.status.idle":"2024-03-26T22:15:30.128405Z","shell.execute_reply.started":"2024-03-26T22:14:58.965788Z","shell.execute_reply":"2024-03-26T22:15:30.126646Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explore(split=\"test\", dataset=\"lizard\")","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:15:30.130832Z","iopub.execute_input":"2024-03-26T22:15:30.131747Z","iopub.status.idle":"2024-03-26T22:15:31.056181Z","shell.execute_reply.started":"2024-03-26T22:15:30.131700Z","shell.execute_reply":"2024-03-26T22:15:31.054971Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Temple**","metadata":{}},{"cell_type":"code","source":"explore(split=\"train\", dataset=\"multi-temporal-temple-baalshamin\")","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:15:31.057427Z","iopub.execute_input":"2024-03-26T22:15:31.057761Z","iopub.status.idle":"2024-03-26T22:15:38.917164Z","shell.execute_reply.started":"2024-03-26T22:15:31.057731Z","shell.execute_reply":"2024-03-26T22:15:38.915552Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Pond**","metadata":{}},{"cell_type":"code","source":"explore(split=\"train\", dataset=\"pond\")","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:15:38.918617Z","iopub.execute_input":"2024-03-26T22:15:38.918990Z","iopub.status.idle":"2024-03-26T22:16:20.424333Z","shell.execute_reply.started":"2024-03-26T22:15:38.918952Z","shell.execute_reply":"2024-03-26T22:16:20.423145Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Glass Cup**","metadata":{}},{"cell_type":"code","source":"explore(split=\"train\", dataset=\"transp_obj_glass_cup\")","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:16:20.426061Z","iopub.execute_input":"2024-03-26T22:16:20.426383Z","iopub.status.idle":"2024-03-26T22:16:30.810982Z","shell.execute_reply.started":"2024-03-26T22:16:20.426356Z","shell.execute_reply":"2024-03-26T22:16:30.810091Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Glass Cylinder**","metadata":{}},{"cell_type":"code","source":"explore(split=\"train\", dataset=\"transp_obj_glass_cylinder\")","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:16:30.812362Z","iopub.execute_input":"2024-03-26T22:16:30.812667Z","iopub.status.idle":"2024-03-26T22:16:49.025055Z","shell.execute_reply.started":"2024-03-26T22:16:30.812640Z","shell.execute_reply":"2024-03-26T22:16:49.023736Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Lets do some progression","metadata":{}},{"cell_type":"code","source":"\ndef arr_to_str(a):\n    return ';'.join([str(x) for x in a.reshape(-1)])\n\n\ndef load_torch_image(fname, device=torch.device('cpu')):\n    img = K.io.load_image(fname, K.io.ImageLoadType.RGB32, device=device)[None, ...]\n    return img\n\ndevice = K.utils.get_cuda_device_if_available(0)\nprint (device)","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:16:49.026983Z","iopub.execute_input":"2024-03-26T22:16:49.027379Z","iopub.status.idle":"2024-03-26T22:16:49.087090Z","shell.execute_reply.started":"2024-03-26T22:16:49.027345Z","shell.execute_reply":"2024-03-26T22:16:49.086249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_global_desc(fnames, device = torch.device('cpu')):\n    processor = AutoImageProcessor.from_pretrained('/kaggle/input/dinov2/pytorch/base/1')\n    model = AutoModel.from_pretrained('/kaggle/input/dinov2/pytorch/base/1')\n    model = model.eval()\n    model = model.to(device)\n    global_descs_dinov2 = []\n    for i, img_fname_full in tqdm(enumerate(fnames),total= len(fnames)):\n        key = os.path.splitext(os.path.basename(img_fname_full))[0]\n        timg = load_torch_image(img_fname_full)\n        with torch.inference_mode():\n            inputs = processor(images=timg, return_tensors=\"pt\", do_rescale=False).to(device)\n            outputs = model(**inputs)\n            dino_mac = F.normalize(outputs.last_hidden_state[:,1:].max(dim=1)[0], dim=1, p=2)\n        global_descs_dinov2.append(dino_mac.detach().cpu())\n    global_descs_dinov2 = torch.cat(global_descs_dinov2, dim=0)\n    return global_descs_dinov2\n\n\ndef get_img_pairs_exhaustive(img_fnames):\n    index_pairs = []\n    for i in range(len(img_fnames)):\n        for j in range(i+1, len(img_fnames)):\n            index_pairs.append((i,j))\n    return index_pairs\n\n\ndef get_image_pairs_shortlist(fnames,\n                              sim_th = 0.6, # should be strict\n                              min_pairs = 20,\n                              exhaustive_if_less = 20,\n                              device=torch.device('cpu')):\n    num_imgs = len(fnames)\n    if num_imgs <= exhaustive_if_less:\n        return get_img_pairs_exhaustive(fnames)\n    descs = get_global_desc(fnames, device=device)\n    dm = torch.cdist(descs, descs, p=2).detach().cpu().numpy()\n    # removing half\n    mask = dm <= sim_th\n    total = 0\n    matching_list = []\n    ar = np.arange(num_imgs)\n    already_there_set = []\n    for st_idx in range(num_imgs-1):\n        mask_idx = mask[st_idx]\n        to_match = ar[mask_idx]\n        if len(to_match) < min_pairs:\n            to_match = np.argsort(dm[st_idx])[:min_pairs]  \n        for idx in to_match:\n            if st_idx == idx:\n                continue\n            if dm[st_idx, idx] < 1000:\n                matching_list.append(tuple(sorted((st_idx, idx.item()))))\n                total+=1\n    matching_list = sorted(list(set(matching_list)))\n    return matching_list","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:16:49.088289Z","iopub.execute_input":"2024-03-26T22:16:49.089145Z","iopub.status.idle":"2024-03-26T22:16:49.125400Z","shell.execute_reply.started":"2024-03-26T22:16:49.089118Z","shell.execute_reply":"2024-03-26T22:16:49.124434Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nfrom lightglue import match_pair\nfrom lightglue import ALIKED, LightGlue\nfrom lightglue.utils import load_image, rbd\n\n\ndef detect_aliked(img_fnames,\n                  feature_dir = '.featureout',\n                  num_features = 4096,\n                  resize_to = 1024,\n                  device=torch.device('cpu')):\n    dtype = torch.float32 # ALIKED has issues with float16\n    extractor = ALIKED(max_num_keypoints=num_features, detection_threshold=0.01, resize=resize_to).eval().to(device, dtype)\n    if not os.path.isdir(feature_dir):\n        os.makedirs(feature_dir)\n    with h5py.File(f'{feature_dir}/keypoints.h5', mode='w') as f_kp, \\\n         h5py.File(f'{feature_dir}/descriptors.h5', mode='w') as f_desc:\n        for img_path in tqdm(img_fnames):\n            img_fname = img_path.split('/')[-1]\n            key = img_fname\n            with torch.inference_mode():\n                image0 = load_torch_image(img_path, device=device).to(dtype)\n                feats0 = extractor.extract(image0)  # auto-resize the image, disable with resize=None\n                kpts = feats0['keypoints'].reshape(-1, 2).detach().cpu().numpy()\n                descs = feats0['descriptors'].reshape(len(kpts), -1).detach().cpu().numpy()\n                f_kp[key] = kpts\n                f_desc[key] = descs\n    return\ndef match_with_lightglue(img_fnames,\n                   index_pairs,\n                   feature_dir = '.featureout',\n                   device=torch.device('cpu'),\n                   min_matches=15,verbose=True):\n    lg_matcher = KF.LightGlueMatcher(\"aliked\", {\"width_confidence\": -1,\n                                                \"depth_confidence\": -1,\n                                                 \"mp\": True if 'cuda' in str(device) else False}).eval().to(device)\n    with h5py.File(f'{feature_dir}/keypoints.h5', mode='r') as f_kp, \\\n        h5py.File(f'{feature_dir}/descriptors.h5', mode='r') as f_desc, \\\n        h5py.File(f'{feature_dir}/matches.h5', mode='w') as f_match:\n        for pair_idx in tqdm(index_pairs):\n            idx1, idx2 = pair_idx\n            fname1, fname2 = img_fnames[idx1], img_fnames[idx2]\n            key1, key2 = fname1.split('/')[-1], fname2.split('/')[-1]\n            kp1 = torch.from_numpy(f_kp[key1][...]).to(device)\n            kp2 = torch.from_numpy(f_kp[key2][...]).to(device)\n            desc1 = torch.from_numpy(f_desc[key1][...]).to(device)\n            desc2 = torch.from_numpy(f_desc[key2][...]).to(device)\n            with torch.inference_mode():\n                dists, idxs = lg_matcher(desc1,\n                                         desc2,\n                                         KF.laf_from_center_scale_ori(kp1[None]),\n                                         KF.laf_from_center_scale_ori(kp2[None]))\n            if len(idxs)  == 0:\n                continue\n            n_matches = len(idxs)\n            if verbose:\n                print (f'{key1}-{key2}: {n_matches} matches')\n            group  = f_match.require_group(key1)\n            if n_matches >= min_matches:\n                 group.create_dataset(key2, data=idxs.detach().cpu().numpy().reshape(-1, 2))\n    return\n\ndef import_into_colmap(img_dir,\n                       feature_dir ='.featureout',\n                       database_path = 'colmap.db'):\n    db = COLMAPDatabase.connect(database_path)\n    db.create_tables()\n    single_camera = False\n    fname_to_id = add_keypoints(db, feature_dir, img_dir, '', 'simple-pinhole', single_camera)\n    add_matches(\n        db,\n        feature_dir,\n        fname_to_id,\n    )\n    db.commit()\n    return","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:16:49.126927Z","iopub.execute_input":"2024-03-26T22:16:49.127222Z","iopub.status.idle":"2024-03-26T22:16:49.449002Z","shell.execute_reply.started":"2024-03-26T22:16:49.127197Z","shell.execute_reply":"2024-03-26T22:16:49.448057Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"src = '/kaggle/input/image-matching-challenge-2024/'\n# Get data from csv.\n\ndata_dict = {}\nwith open(f'{src}/sample_submission.csv', 'r') as f:\n    for i, l in enumerate(f):\n        if i== 0:\n            print (l)\n        # Skip header.\n        if l and i > 0:\n            image_path, dataset, scene, _, _ = l.strip().split(',')\n            if dataset not in data_dict:\n                data_dict[dataset] = {}\n            if scene not in data_dict[dataset]:\n                data_dict[dataset][scene] = []\n            data_dict[dataset][scene].append(image_path)\nfor dataset in data_dict:\n    for scene in data_dict[dataset]:\n        print(f'{dataset} / {scene} -> {len(data_dict[dataset][scene])} images')\n\nout_results = {}\ntimings = {\"shortlisting\":[],\n           \"feature_detection\": [],\n           \"feature_matching\":[],\n           \"RANSAC\": [],\n           \"Reconstruction\": []}","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:16:49.450270Z","iopub.execute_input":"2024-03-26T22:16:49.450623Z","iopub.status.idle":"2024-03-26T22:16:49.462083Z","shell.execute_reply.started":"2024-03-26T22:16:49.450592Z","shell.execute_reply":"2024-03-26T22:16:49.461217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to create a submission file.\ndef create_submission(out_results, data_dict):\n    with open(f'submission.csv', 'w') as f:\n        f.write('image_path,dataset,scene,rotation_matrix,translation_vector\\n')\n        for dataset in data_dict:\n            if dataset in out_results:\n                res = out_results[dataset]\n            else:\n                res = {}\n            for scene in data_dict[dataset]:\n                if scene in res:\n                    scene_res = res[scene]\n                else:\n                    scene_res = {\"R\":{}, \"t\":{}}\n                for image in data_dict[dataset][scene]:\n                    if image in scene_res:\n                        print (image)\n                        R = scene_res[image]['R'].reshape(-1)\n                        T = scene_res[image]['t'].reshape(-1)\n                    else:\n                        R = np.eye(3).reshape(-1)\n                        T = np.zeros((3))\n                    f.write(f'{image},{dataset},{scene},{arr_to_str(R)},{arr_to_str(T)}\\n')","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:16:49.463221Z","iopub.execute_input":"2024-03-26T22:16:49.463552Z","iopub.status.idle":"2024-03-26T22:16:49.476604Z","shell.execute_reply.started":"2024-03-26T22:16:49.463521Z","shell.execute_reply":"2024-03-26T22:16:49.475803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\nfrom copy import deepcopy\ndatasets = []\nfor dataset in data_dict:\n    datasets.append(dataset)\nprint (f\"Extracting on device {device}\")\nfor dataset in data_dict:\n    print(dataset)\n    if dataset not in out_results:\n        out_results[dataset] = {}\n    for scene in data_dict[dataset]:\n        print(scene)\n        img_dir =  os.path.join(src,  '/'.join(data_dict[dataset][scene][0].split('/')[:-1]))\n        # Fail gently if the notebook has not been submitted and the test data is not populated.\n        # You may want to run this on the training data in that case?\n        # Wrap the meaty part in a try-except block.\n        try:\n            out_results[dataset][scene] = {}\n            img_fnames = [os.path.join(src, x) for x in data_dict[dataset][scene] ] \n            print (f\"Got {len(img_fnames)} images\")\n            feature_dir = f'featureout/{dataset}_{scene}'\n            os.makedirs(feature_dir, exist_ok=True)\n            t=time()\n            #if len(img_fnames) > 60:\n            #    continue\n            index_pairs = get_image_pairs_shortlist(img_fnames,\n                                  sim_th = 0.3, # should be strict\n                                  min_pairs = 20, # we select at least min_pairs PER IMAGE with biggest similarity\n                                  exhaustive_if_less = 20,\n                                  device=device)\n            t=time() -t \n            timings['shortlisting'].append(t)\n            print (f'{len(index_pairs)}, pairs to match, {t:.4f} sec')\n            gc.collect()\n            t=time()\n            detect_aliked(img_fnames, feature_dir, 4096, device=device)\n            gc.collect()\n            t=time() -t \n            timings['feature_detection'].append(t)\n            print(f'Features detected in  {t:.4f} sec')\n            t=time()\n            match_with_lightglue(img_fnames, index_pairs, feature_dir=feature_dir,device=device)\n            t=time() -t \n            timings['feature_matching'].append(t)\n            print(f'Features matched in  {t:.4f} sec')\n            database_path = f'{feature_dir}/colmap.db'\n            if os.path.isfile(database_path):\n                os.remove(database_path)\n            gc.collect()\n            sleep(1)\n            import_into_colmap(img_dir, feature_dir=feature_dir, database_path=database_path)\n            output_path = f'{feature_dir}/colmap_rec_aliked'\n            t=time()\n            pycolmap.match_exhaustive(database_path)\n            t=time() - t \n            timings['RANSAC'].append(t)\n            print(f'RANSAC in  {t:.4f} sec')\n            t=time()\n            # By default colmap does not generate a reconstruction if less than 10 images are registered. Lower it to 3.\n            mapper_options = pycolmap.IncrementalPipelineOptions()\n            mapper_options.min_model_size = 3\n            mapper_options.max_num_models = 2\n            os.makedirs(output_path, exist_ok=True)\n            maps = pycolmap.incremental_mapping(database_path=database_path, \n                                                image_path=img_dir,\n                                                output_path=output_path, options=mapper_options)\n            sleep(1)\n            print(maps)\n            clear_output(wait=False)\n            t=time() - t\n            timings['Reconstruction'].append(t)\n            print(f'Reconstruction done in  {t:.4f} sec')\n            imgs_registered  = 0\n            best_idx = None\n            print (\"Looking for the best reconstruction\")\n            if isinstance(maps, dict):\n                for idx1, rec in maps.items():\n                    print (idx1, rec.summary())\n                    try:\n                        if len(rec.images) > imgs_registered:\n                            imgs_registered = len(rec.images)\n                            best_idx = idx1\n                    except:\n                        continue\n            if best_idx is not None:\n                print (maps[best_idx].summary())\n                for k, im in maps[best_idx].images.items():\n                    key1 = f'test/{scene}/images/{im.name}'\n                    print(key1)\n                    out_results[dataset][scene][key1] = {}\n                    out_results[dataset][scene][key1][\"R\"] = deepcopy(im.cam_from_world.rotation.matrix())\n                    out_results[dataset][scene][key1][\"t\"] = deepcopy(np.array(im.cam_from_world.translation))\n            print(f'Registered: {dataset} / {scene} -> {len(out_results[dataset][scene])} images')\n            print(f'Total: {dataset} / {scene} -> {len(data_dict[dataset][scene])} images')\n            create_submission(out_results, data_dict)\n            gc.collect()\n        except Exception as e:\n            print (e)\n            pass","metadata":{"execution":{"iopub.status.busy":"2024-03-26T22:16:49.477777Z","iopub.execute_input":"2024-03-26T22:16:49.478078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Submission file","metadata":{}},{"cell_type":"code","source":"!cat submission.csv","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}