{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":71885,"databundleVersionId":8143495,"sourceType":"competition"},{"sourceId":7884485,"sourceType":"datasetVersion","datasetId":4628051},{"sourceId":7884725,"sourceType":"datasetVersion","datasetId":4628331},{"sourceId":7975259,"sourceType":"datasetVersion","datasetId":4693319},{"sourceId":4534,"sourceType":"modelInstanceVersion","modelInstanceId":3326},{"sourceId":17191,"sourceType":"modelInstanceVersion","modelInstanceId":14317},{"sourceId":17555,"sourceType":"modelInstanceVersion","modelInstanceId":14611},{"sourceId":21324,"sourceType":"modelInstanceVersion","modelInstanceId":17657},{"sourceId":21712,"sourceType":"modelInstanceVersion","modelInstanceId":17776},{"sourceId":30326,"sourceType":"modelInstanceVersion","modelInstanceId":25478}],"dockerImageVersionId":30665,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# EXAMPLE SUBMISSION USING DEEP-IMAGE-MATCHING (DIM) PACKAGE\n# Additional doc at https://github.com/3DOM-FBK/deep-image-matching and https://3dom-fbk.github.io/deep-image-matching/\n#\n# DIM is a library of deep-learning and hand-crafted local features and matchers, and extends them to high-resolution images, \n# making full use of the available image information. It is mainly based on the KORNIA (https://github.com/kornia/kornia)\n# and HLOC (https://github.com/cvg/Hierarchical-Localization) library, and has several options to reduce the number of image pairs \n# to be processed (DL global descriptors, low-resolution matching, ..). DIM can run SfM with pycolmap, OpenMVG, and MICMAC. \n# This notebook shows pycolmap usage. For licences, see authors' projects according to the DL features used.\n\n\n# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/imc2024-packages-lightglue-rerun-kornia'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n!pip install --no-index /kaggle/input/imc2024-packages-lightglue-rerun-kornia/* --no-deps\n!pip install --no-index /kaggle/input/dim-dependencies/* --no-deps\n\n# Import checkpoints to use the notebook without internet\n!mkdir -p /root/.cache/torch/hub/checkpoints\n!cp -r /kaggle/input/dim-weights/pytorch/local_features_and_matchers/3/* /root/.cache/torch/hub/checkpoints/\n!mv /root/.cache/torch/hub/checkpoints/gmberton_CosPlace_main/gmberton_CosPlace_main /root/.cache/torch/hub/\n!mv /root/.cache/torch/hub/checkpoints/netvlad/netvlad /root/.cache/torch/hub/\n!mv /root/.cache/torch/hub/checkpoints/yxgeee_OpenIBL_master/yxgeee_OpenIBL_master /root/.cache/torch/hub/\n!mv /root/.cache/torch/hub/checkpoints/trusted_list /root/.cache/torch/hub","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-13T07:41:09.857842Z","iopub.execute_input":"2024-04-13T07:41:09.858157Z","iopub.status.idle":"2024-04-13T07:41:55.191899Z","shell.execute_reply.started":"2024-04-13T07:41:09.858132Z","shell.execute_reply":"2024-04-13T07:41:55.190509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# General utilities\nimport matplotlib.pyplot as plt\n\nimport os\nfrom tqdm import tqdm\nfrom time import time, sleep\nfrom fastprogress import progress_bar\nimport gc\nimport numpy as np\nimport h5py\nfrom IPython.display import clear_output\nfrom collections import defaultdict\nfrom copy import deepcopy\n\n# CV/MLe\nimport cv2\nimport torch\nimport torch.nn.functional as F\nimport kornia as K\nimport kornia.feature as KF\nfrom PIL import Image\nfrom transformers import AutoImageProcessor, AutoModel\n\n# 3D reconstruction\nimport pycolmap\n\n# Data importing into colmap\nimport sys\nsys.path.append('/kaggle/input/colmap-db-import')\n\ndef arr_to_str(a):\n    return ';'.join([str(x) for x in a.reshape(-1)])\n\ndevice = K.utils.get_cuda_device_if_available(0)\nprint (device)","metadata":{"execution":{"iopub.status.busy":"2024-04-13T07:41:55.194285Z","iopub.execute_input":"2024-04-13T07:41:55.194988Z","iopub.status.idle":"2024-04-13T07:42:12.839737Z","shell.execute_reply.started":"2024-04-13T07:41:55.194954Z","shell.execute_reply":"2024-04-13T07:42:12.838725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"src = '/kaggle/input/image-matching-challenge-2024/'\n# Get data from csv.\n\ndata_dict = {}\nwith open(f'{src}/sample_submission.csv', 'r') as f:\n    for i, l in enumerate(f):\n        if i== 0:\n            print (l)\n        # Skip header.\n        if l and i > 0:\n            image_path, dataset, scene, _, _ = l.strip().split(',')\n            if dataset not in data_dict:\n                data_dict[dataset] = {}\n            if scene not in data_dict[dataset]:\n                data_dict[dataset][scene] = []\n            data_dict[dataset][scene].append(image_path)\nfor dataset in data_dict:\n    for scene in data_dict[dataset]:\n        print(f'{dataset} / {scene} -> {len(data_dict[dataset][scene])} images')\n\nout_results = {}\nprint(data_dict)","metadata":{"execution":{"iopub.status.busy":"2024-04-13T07:42:12.840883Z","iopub.execute_input":"2024-04-13T07:42:12.841426Z","iopub.status.idle":"2024-04-13T07:42:12.858461Z","shell.execute_reply.started":"2024-04-13T07:42:12.8414Z","shell.execute_reply":"2024-04-13T07:42:12.856941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to create a submission file.\ndef create_submission(out_results, data_dict):\n    with open(f'submission.csv', 'w') as f:\n        f.write('image_path,dataset,scene,rotation_matrix,translation_vector\\n')\n        for dataset in data_dict:\n            if dataset in out_results:\n                res = out_results[dataset]\n            else:\n                res = {}\n            for scene in data_dict[dataset]:\n                if scene in res:\n                    scene_res = res[scene]\n                else:\n                    scene_res = {\"R\":{}, \"t\":{}}\n                for image in data_dict[dataset][scene]:\n                    if image in scene_res:\n                        print (image)\n                        R = scene_res[image]['R'].reshape(-1)\n                        T = scene_res[image]['t'].reshape(-1)\n                    else:\n                        R = np.eye(3).reshape(-1)\n                        T = np.zeros((3))\n                    f.write(f'{image},{dataset},{scene},{arr_to_str(R)},{arr_to_str(T)}\\n')","metadata":{"execution":{"iopub.status.busy":"2024-04-13T07:42:12.860607Z","iopub.execute_input":"2024-04-13T07:42:12.860941Z","iopub.status.idle":"2024-04-13T07:42:12.874408Z","shell.execute_reply.started":"2024-04-13T07:42:12.860918Z","shell.execute_reply":"2024-04-13T07:42:12.873553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Install DIM\n# Instead of cloning from github, since internet is disabled when submitting the notebook\n# git clone https://github.com/3DOM-FBK/deep-image-matching.git /kaggle/working/deep-image-matching-dev\n# install from the DIM repo uploaded on Kaggle as model (See MODELS>deep-image-matching-dev V1)\n!cp -r /kaggle/input/deep-image-matching-dev/pytorch/deep-image-matching-dev/1 /kaggle/working/deep-image-matching-dev\n!cd /kaggle/working/deep-image-matching-dev && pip install --no-index . --no-deps --no-build-isolation","metadata":{"execution":{"iopub.status.busy":"2024-04-13T07:50:56.225598Z","iopub.execute_input":"2024-04-13T07:50:56.226002Z","iopub.status.idle":"2024-04-13T07:51:11.962025Z","shell.execute_reply.started":"2024-04-13T07:50:56.225969Z","shell.execute_reply":"2024-04-13T07:51:11.961038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# DIM package can run from CLI or used as library (see following block)\n#!python /kaggle/working/deep-image-matching-dev/main.py --help","metadata":{"execution":{"iopub.status.busy":"2024-04-13T07:42:14.892653Z","iopub.execute_input":"2024-04-13T07:42:14.893594Z","iopub.status.idle":"2024-04-13T07:42:14.898836Z","shell.execute_reply.started":"2024-04-13T07:42:14.89355Z","shell.execute_reply":"2024-04-13T07:42:14.897721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from deep_image_matching.config import Config\nfrom deep_image_matching.image_matching import ImageMatching\nfrom deep_image_matching.io.h5_to_db import export_to_colmap\n\nworking_dir = '/kaggle/working/results'\n\n# Local features/matchers:\n# superpoint+lightglue,disk+lightglue,aliked+lightglue,orb+kornia_matcher,sift+kornia_matcher,loftr,se2loftr,roma,keynetaffnethardnet+kornia_matcher,dedode+kornia_matcher\n# Easily extendable to other local features (see https://github.com/3DOM-FBK/deep-image-matching/tree/master/src/deep_image_matching/extractors)\n# For now Roma and LoFTR not available for multiview\npipeline = 'superpoint+lightglue'\n\n# Match image pairs at full resolution\n# preselection : corresponding tiles preselected on image pairs at low resolution with superpoint+lightglue\n# grid : x-th tile matched with x-th tile\ntiling = \"none\" # none,preselection,grid,exhaustive\n\n# Strategy to match image pairs:\n# matching_lowres (match all images ath low resolution to find candidate pairs),bruteforce,sequential,retrieval,custom_pairs\nstrategy = \"bruteforce\"\nglobal_feature = \"netvlad\" # netvlad,cosplace,openibl\n\n# To resize image quality if too large format\n# highest : for subpixel accuracy, keypoints are extracted on double sized images\nquality = \"high\" # lowest,low,medium,high,highest\n\nos.makedirs(working_dir, exist_ok=True)\ngc.collect()\nfrom copy import deepcopy\ndatasets = []\nfor dataset in data_dict:\n    datasets.append(dataset)\n    \nprint (f\"Extracting on device {device}\")\n\nfor dataset in data_dict:\n    if dataset not in out_results:\n        out_results[dataset] = {}\n    for scene in data_dict[dataset]:\n        print(scene)\n        os.makedirs(f\"{working_dir}/{scene}\", exist_ok=True)\n        img_dir =  os.path.join(src,  '/'.join(data_dict[dataset][scene][0].split('/')[:-1]))\n        out_results[dataset][scene] = {}\n\n        # Run DIM as library to extract and match features\n        cli_params = {\n            \"dir\": f\"{working_dir}/{scene}\",\n            \"images\": img_dir,\n            \"pipeline\": pipeline,\n            \"strategy\": strategy,\n            \"global_feature\": global_feature,\n            \"tiling\": tiling,\n            \"quality\": quality,\n            \"skip_reconstruction\": False,\n            \"force\": True,\n            \"camera_options\": \"../config/cameras.yaml\",\n            \"openmvg\": None,\n            \"verbose\": True,\n        }\n\n        # All the configuration params present in https://github.com/3DOM-FBK/deep-image-matching/blob/master/src/deep_image_matching/config.py can be modified here\n        # Also local features and matchers config\n        config = Config(cli_params)\n        config.general[\"min_inliers_per_pair\"] = 25\n        config.general[\"gv_threshold\"] = 4\n        config.extractor[\"max_keypoints\"] = 8000 # Max keypoint per image or per tile if used\n        config.general[\"min_inlier_ratio_per_pair\"] = 0.01\n        config.save()\n\n        imgs_dir = config.general[\"image_dir\"]\n        output_dir = config.general[\"output_dir\"]\n        matching_strategy = config.general[\"matching_strategy\"]\n        extractor = config.extractor[\"name\"]\n        matcher = config.matcher[\"name\"]\n\n        img_matching = ImageMatching(\n            imgs_dir=imgs_dir,\n            output_dir=output_dir,\n            matching_strategy=matching_strategy,\n            local_features=extractor,\n            matching_method=matcher,\n            pair_file=config.general[\"pair_file\"],\n            retrieval_option=config.general[\"retrieval\"],\n            overlap=config.general[\"overlap\"],\n            existing_colmap_model=config.general[\"db_path\"],\n            custom_config=config.as_dict(),\n        )\n\n        pair_path = img_matching.generate_pairs()\n        feature_path = img_matching.extract_features()\n        match_path = img_matching.match_pairs(feature_path)\n\n        # See usage https://3dom-fbk.github.io/deep-image-matching/camera_models/\n        camera_options = {\n           'general' : {\n            \"camera_model\" : \"simple-radial\", # [\"simple-pinhole\", \"pinhole\", \"simple-radial\", \"opencv\"]\n            \"single_camera\" : True,\n           },\n           'cam0' : {\n                \"camera_model\" : \"pinhole\",\n                \"images\" : \"DSC_6468.JPG,DSC_6468.JPG\",\n           },\n           'cam1' : {\n                \"camera_model\" : \"pinhole\",\n                \"images\" : \"\",\n           },\n        }\n\n        # You can also read matches directly from the h5 database, the advantage is that in a second moment the database.db can be imported to COLMAP\n        # to visualize matches\n        database_path = output_dir / \"database.db\"\n        export_to_colmap(\n            img_dir=imgs_dir,\n            feature_path=feature_path,\n            match_path=match_path,\n            database_path=database_path,\n            camera_options=camera_options,\n        )\n        \n        if not config.general[\"skip_reconstruction\"]:\n            # import reconstruction module\n            from deep_image_matching import reconstruction\n            \n            # Define database path and camera mode\n            database = output_dir / \"database.db\"\n            camera_mode = pycolmap.CameraMode.AUTO\n            cameras = None\n            \n            reconst_opts = {\n                    \"ba_refine_focal_length\": True,\n                    \"ba_refine_principal_point\": True,\n                    \"ba_refine_extra_params\": True,\n                    \"min_model_size\" : 2,\n                    \"max_num_models\" : 2,\n                }\n            \n            best_model = reconstruction.main(\n                database=database,\n                image_dir=imgs_dir,\n                feature_path=feature_path,\n                match_path=match_path,\n                pair_path=pair_path,\n                sfm_dir=output_dir,\n                camera_mode=camera_mode,\n                cameras=cameras,\n                skip_geometric_verification=True,\n                reconst_opts=reconst_opts,\n                verbose=config.general[\"verbose\"],\n            )\n\n            clear_output(wait=False)\n            print('best model', best_model)\n            \n            print (best_model.summary())\n            for k, im in best_model.images.items():\n                key1 = f'test/{scene}/images/{im.name}'\n                print(key1)\n                out_results[dataset][scene][key1] = {}\n                out_results[dataset][scene][key1][\"R\"] = deepcopy(im.cam_from_world.rotation.matrix())\n                out_results[dataset][scene][key1][\"t\"] = deepcopy(np.array(im.cam_from_world.translation))\n            \n            print(f'Registered: {dataset} / {scene} -> {len(out_results[dataset][scene])} images')\n            print(f'Total: {dataset} / {scene} -> {len(data_dict[dataset][scene])} images')\n            create_submission(out_results, data_dict)\n            gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-13T07:52:36.65194Z","iopub.execute_input":"2024-04-13T07:52:36.652336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%rm -r /kaggle/working/deep-image-matching-dev\n%rm -r results","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create_submission(out_results, data_dict)","metadata":{"execution":{"iopub.status.busy":"2024-04-13T07:42:15.654595Z","iopub.status.idle":"2024-04-13T07:42:15.654951Z","shell.execute_reply.started":"2024-04-13T07:42:15.654783Z","shell.execute_reply":"2024-04-13T07:42:15.654798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cat submission.csv","metadata":{"execution":{"iopub.status.busy":"2024-04-13T07:42:15.656015Z","iopub.status.idle":"2024-04-13T07:42:15.656395Z","shell.execute_reply.started":"2024-04-13T07:42:15.656225Z","shell.execute_reply":"2024-04-13T07:42:15.656239Z"},"trusted":true},"execution_count":null,"outputs":[]}]}