{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":91498,"databundleVersionId":11655853,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":4534,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":3326,"modelId":986},{"sourceId":17191,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":14317,"modelId":21716},{"sourceId":17555,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":14611,"modelId":22086},{"sourceId":17578,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":14634,"modelId":22086},{"sourceId":329184,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":276193,"modelId":297081}],"dockerImageVersionId":30919,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Problem Statement\n\nIn many computer vision applications, especially in image matching and reconstruction challenges, the goal is to reliably find correspondences between images. These correspondences are important for tasks such as 3D reconstruction, object recognition, and image retrieval. The pipeline described here is built for a Kaggle image-matching challenge and consists of several key stages:\n\n1. Global Descriptor Extraction\n\nObtain a high-level summary (global descriptor) of each image.\n\n2. Image Pair Shortlisting\n\nSelect candidate image pairs based on similarity to reduce computation in later stages.\n\n3. Local Keypoint Detection (ALIKED)\n\nExtract local features (keypoints and their descriptors) from images.\n\n4. Feature Matching (LightGlue with Fallback)\n\nMatch corresponding keypoints between image pairs, combining advanced matching with a classical fallback approach.\n\nEach stage is designed to improve the quality of the matches and ultimately lead to more robust image clustering and reconstruction.","metadata":{}},{"cell_type":"code","source":"# !git clone https://github.com/cvg/LightGlue.git \n# !python -m pip install -e LightGlue/","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !cp -r /kaggle/input/lightglue /kaggle/working/LightGlue\n# !python -m pip install -e /kaggle/working/LightGlue/","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!cp -r /kaggle/input/lightglue_model_v2 /kaggle/working/\n\n# Add the copied repository to sys.path so Python can find it.\nimport sys\nsys.path.insert(0, \"/kaggle/input/lightglue_model_v2/pytorch/v2/1\")\nfrom lightglue import ALIKED, LightGlue","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-09T19:29:57.473855Z","iopub.execute_input":"2025-04-09T19:29:57.474276Z","iopub.status.idle":"2025-04-09T19:30:06.207537Z","shell.execute_reply.started":"2025-04-09T19:29:57.474236Z","shell.execute_reply":"2025-04-09T19:30:06.206670Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install dependencies and copy model weights to run the notebook without internet access when submitting to the competition.\n\n# !pip install --no-index /kaggle/input/imc2024-packages-lightglue-rerun-kornia/* --no-deps\n!mkdir -p /root/.cache/torch/hub/checkpoints\n!cp /kaggle/input/aliked/pytorch/aliked-n16/1/aliked-n16.pth /root/.cache/torch/hub/checkpoints/\n!cp /kaggle/input/lightglue/pytorch/aliked/1/aliked_lightglue.pth /root/.cache/torch/hub/checkpoints/\n!cp /kaggle/input/lightglue/pytorch/aliked/1/aliked_lightglue.pth /root/.cache/torch/hub/checkpoints/aliked_lightglue_v0-1_arxiv-pth","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-09T19:30:08.453543Z","iopub.execute_input":"2025-04-09T19:30:08.454029Z","iopub.status.idle":"2025-04-09T19:30:09.976709Z","shell.execute_reply.started":"2025-04-09T19:30:08.454002Z","shell.execute_reply":"2025-04-09T19:30:09.975507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport gc\nimport sys\nimport cv2\nimport h5py\nimport torch\nimport dataclasses\nimport kornia as K\nimport numpy as np\nimport pandas as pd\nimport networkx as nx\nimport kornia.feature as KF\nimport torch.nn.functional as F\n\nfrom PIL import Image\nfrom tqdm import tqdm\nfrom copy import deepcopy\nfrom time import time, sleep\nfrom collections import defaultdict\n# from lightglue import ALIKED, LightGlue\nfrom IPython.display import clear_output\nfrom transformers import AutoImageProcessor, AutoModel\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T04:57:02.628617Z","iopub.execute_input":"2025-04-10T04:57:02.629000Z","iopub.status.idle":"2025-04-10T04:57:18.489786Z","shell.execute_reply.started":"2025-04-10T04:57:02.628972Z","shell.execute_reply":"2025-04-10T04:57:18.489136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/image-matching-challenge-2025/train/\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T04:57:18.490879Z","iopub.execute_input":"2025-04-10T04:57:18.491559Z","iopub.status.idle":"2025-04-10T04:57:18.495563Z","shell.execute_reply.started":"2025-04-10T04:57:18.491531Z","shell.execute_reply":"2025-04-10T04:57:18.494784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_thresholds = pd.read_csv('/kaggle/input/image-matching-challenge-2025/train_thresholds.csv')\n# data_dir = '/kaggle/input/image-matching-challenge-2025'\n# is_train=False\n# if is_train:\n#     train_labels_path = os.path.join(data_dir, 'train_labels.csv')\n# else:\n#     train_labels_path = os.path.join(data_dir, 'sample_submission.csv')\n# train_labels = pd.read_csv(train_labels_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T04:58:20.187575Z","iopub.execute_input":"2025-04-10T04:58:20.187928Z","iopub.status.idle":"2025-04-10T04:58:20.226768Z","shell.execute_reply.started":"2025-04-10T04:58:20.187893Z","shell.execute_reply":"2025-04-10T04:58:20.226104Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Parse rotation matrix and translation vector\n# def parse_matrix(matrix_str):\n#     return np.array([float(x) for x in matrix_str.split(';')]).reshape(3, 3)\n\n# def parse_vector(vector_str):\n#     return np.array([float(x) for x in vector_str.split(';')])\n\n# train_labels[\"rotation_matrix\"] = train_labels[\"rotation_matrix\"].apply(parse_matrix)\n# train_labels[\"translation_vector\"] = train_labels[\"translation_vector\"].apply(parse_vector)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T04:58:23.568207Z","iopub.execute_input":"2025-04-10T04:58:23.568553Z","iopub.status.idle":"2025-04-10T04:58:23.590379Z","shell.execute_reply.started":"2025-04-10T04:58:23.568525Z","shell.execute_reply":"2025-04-10T04:58:23.589247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Collect info from the dataset\n\n@dataclasses.dataclass\nclass Prediction:\n    image_id: str | None  # A unique identifier for the row -- unused otherwise. Used only on the hidden test set.\n    dataset: str\n    filename: str\n    cluster_index: int | None = None\n    rotation: np.ndarray | None = None\n    translation: np.ndarray | None = None\n\n# Set is_train=True to run the notebook on the training data.\n# Set is_train=False if submitting an entry to the competition (test data is hidden, and different from what you see on the \"test\" folder).\nis_train = False\ndata_dir = '/kaggle/input/image-matching-challenge-2025'\nworkdir = '/kaggle/working/result/'\nos.makedirs(workdir, exist_ok=True)\n\nif is_train:\n    sample_submission_csv = os.path.join(data_dir, 'train_labels.csv')\nelse:\n    sample_submission_csv = os.path.join(data_dir, 'sample_submission.csv')\n\nsamples = {}\ncompetition_data = pd.read_csv(sample_submission_csv)\nfor _, row in competition_data.iterrows():\n    # Note: For the test data, the \"scene\" column has no meaning, and the rotation_matrix and translation_vector columns are random.\n    if row.dataset not in samples:\n        samples[row.dataset] = []\n    samples[row.dataset].append(\n        Prediction(\n            image_id=None if is_train else row.image_id,\n            dataset=row.dataset,\n            filename=row.image\n        )\n    )\n\nfor dataset in samples:\n    print(f'Dataset \"{dataset}\" -> num_images={len(samples[dataset])}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T05:31:57.181527Z","iopub.execute_input":"2025-04-10T05:31:57.181848Z","iopub.status.idle":"2025-04-10T05:31:57.346178Z","shell.execute_reply.started":"2025-04-10T05:31:57.181825Z","shell.execute_reply":"2025-04-10T05:31:57.345321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_torch_image(fname, device=torch.device('cpu')):\n    img = cv2.imread(fname)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = torch.from_numpy(img).permute(2, 0, 1).float() / 255.0\n    return img.unsqueeze(0).to(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T05:32:03.776250Z","iopub.execute_input":"2025-04-10T05:32:03.776562Z","iopub.status.idle":"2025-04-10T05:32:03.781120Z","shell.execute_reply.started":"2025-04-10T05:32:03.776539Z","shell.execute_reply":"2025-04-10T05:32:03.780257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = K.utils.get_cuda_device_if_available(0)\nprint(f'{device=}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T05:32:08.087138Z","iopub.execute_input":"2025-04-10T05:32:08.087481Z","iopub.status.idle":"2025-04-10T05:32:08.163293Z","shell.execute_reply.started":"2025-04-10T05:32:08.087457Z","shell.execute_reply":"2025-04-10T05:32:08.162493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"processor = AutoImageProcessor.from_pretrained('/kaggle/input/dinov2/pytorch/base/1')\nmodel = AutoModel.from_pretrained('/kaggle/input/dinov2/pytorch/base/1')\nmodel = model.eval()\nmodel = model.to(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T05:32:10.103565Z","iopub.execute_input":"2025-04-10T05:32:10.104032Z","iopub.status.idle":"2025-04-10T05:32:15.689865Z","shell.execute_reply.started":"2025-04-10T05:32:10.103989Z","shell.execute_reply":"2025-04-10T05:32:15.688797Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Global Descriptor Extraction\n- Purpose:\n\nGlobal descriptors are used to capture the overall appearance and content of an image. They provide a compact feature vector that summarizes the image and can be compared using distance metrics (like Euclidean distance or cosine similarity).\n\n- How It Works:\n\n- Model Choice:\nWe use a pretrained DINOv2 model, a self-supervised vision transformer, to extract these descriptors.\n\n- Pooling Strategies:\nSince the output of the transformer consists of multiple tokens (each representing a part of the image), we need to aggregate these tokens into a single descriptor. Several strategies are provided:\n\n1. Max Pooling: Selects the highest value in each feature dimension.\n\n2. Average Pooling: Takes the mean value across tokens.\n\n3. GeM (Generalized-Mean) Pooling: A tunable pooling mechanism that generalizes max and average pooling.","metadata":{}},{"cell_type":"code","source":"# ===========================\n#  Extract Global Descriptors\n# ===========================\ndef extract_dino_descriptor(fnames):\n    global_descs = []\n    for i, img_fname_full in tqdm(enumerate(fnames), total=len(fnames)):\n        timg = load_torch_image(img_fname_full, device=device)\n        with torch.inference_mode():\n            inputs = processor(images=timg, return_tensors=\"pt\", do_rescale=False).to(device)\n            outputs = model(**inputs)\n            # Here we take the [CLS] token embedding; adjust if needed\n            cls_feature = F.normalize(outputs.last_hidden_state[:, 1:].max(dim=1)[0], dim=1, p=2) #L2 Normalization\n        global_descs.append(cls_feature.detach().cpu())# CLS with max pooling\n    global_descs = torch.cat(global_descs, dim=0)\n    return global_descs\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T05:33:00.058416Z","iopub.execute_input":"2025-04-10T05:33:00.058725Z","iopub.status.idle":"2025-04-10T05:33:00.064110Z","shell.execute_reply.started":"2025-04-10T05:33:00.058702Z","shell.execute_reply":"2025-04-10T05:33:00.063126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sim_th=0.6\nmin_pairs=20\ntopN=20","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T05:33:02.457482Z","iopub.execute_input":"2025-04-10T05:33:02.457813Z","iopub.status.idle":"2025-04-10T05:33:02.462122Z","shell.execute_reply.started":"2025-04-10T05:33:02.457788Z","shell.execute_reply":"2025-04-10T05:33:02.461007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"timings = {\n    \"shortlisting\":[],\n    \"feature_detection\": [],\n    \"feature_matching\":[],\n    \"RANSAC\": [],\n    \"Reconstruction\": [],\n}\nt = time()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T05:33:03.749770Z","iopub.execute_input":"2025-04-10T05:33:03.750141Z","iopub.status.idle":"2025-04-10T05:33:03.754576Z","shell.execute_reply.started":"2025-04-10T05:33:03.750108Z","shell.execute_reply":"2025-04-10T05:33:03.753570Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize LightGlue matcher\n# matcher = LightGlue()\n\n# Feature extraction and storage\ndef detect_aliked(img_fnames, feature_dir='.featureout', num_features=4096, resize_to=1024, device=torch.device('cpu')):\n    dtype = torch.float32  # ALIKED has issues with float16\n    extractor = ALIKED(max_num_keypoints=num_features, detection_threshold=0.01, resize=resize_to).eval().to(device, dtype)\n    \n    if not os.path.isdir(feature_dir):\n        os.makedirs(feature_dir)\n    \n    with h5py.File(f'{feature_dir}/keypoints.h5', mode='w') as f_kp, \\\n         h5py.File(f'{feature_dir}/descriptors.h5', mode='w') as f_desc:\n        for img_path in tqdm(img_fnames):\n            img_fname = img_path.split('/')[-1]\n            key = img_fname\n            with torch.inference_mode():\n                image0 = load_torch_image(img_path, device=device).to(dtype)\n                feats0 = extractor.extract(image0)  # auto-resize the image, disable with resize=None\n                kpts = feats0['keypoints'].reshape(-1, 2).detach().cpu().numpy()\n                descs = feats0['descriptors'].reshape(len(kpts), -1).detach().cpu().numpy()\n                f_kp[key] = kpts\n                f_desc[key] = descs\n    return","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T05:33:07.582243Z","iopub.execute_input":"2025-04-10T05:33:07.582536Z","iopub.status.idle":"2025-04-10T05:33:07.589183Z","shell.execute_reply.started":"2025-04-10T05:33:07.582514Z","shell.execute_reply":"2025-04-10T05:33:07.588068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Image pair matching using stored features\ndef match_with_lightglue(img_fnames, index_pairs, feature_dir='.featureout', device=torch.device('cpu'), min_matches=15, verbose=True):\n    lg_matcher = KF.LightGlueMatcher(\"aliked\", {\n        \"width_confidence\": -1,\n        \"depth_confidence\": -1,\n        \"mp\": True if 'cuda' in str(device) else False\n    }).eval().to(device)\n    \n    with h5py.File(f'{feature_dir}/keypoints.h5', mode='r') as f_kp, \\\n        h5py.File(f'{feature_dir}/descriptors.h5', mode='r') as f_desc, \\\n        h5py.File(f'{feature_dir}/matches.h5', mode='w') as f_match:\n        \n        for pair_idx in tqdm(index_pairs):\n            fname1, fname2, _ = pair_idx  # Ignore the third element (the score)\n            key1, key2 = fname1.split('/')[-1], fname2.split('/')[-1]\n            kp1 = torch.from_numpy(f_kp[key1][...]).to(device)\n            kp2 = torch.from_numpy(f_kp[key2][...]).to(device)\n            desc1 = torch.from_numpy(f_desc[key1][...]).to(device)\n            desc2 = torch.from_numpy(f_desc[key2][...]).to(device)\n            \n            with torch.inference_mode():\n                dists, idxs = lg_matcher(\n                    desc1, desc2,\n                    KF.laf_from_center_scale_ori(kp1[None]),\n                    KF.laf_from_center_scale_ori(kp2[None])\n                )\n            \n            if len(idxs) == 0:\n                continue\n            \n            n_matches = len(idxs)\n            if verbose:\n                print(f'{key1}-{key2}: {n_matches} matches')\n            \n            group = f_match.require_group(key1)\n            if n_matches >= min_matches:\n                group.create_dataset(key2, data=idxs.detach().cpu().numpy().reshape(-1, 2))\n    return","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T05:33:10.400583Z","iopub.execute_input":"2025-04-10T05:33:10.400867Z","iopub.status.idle":"2025-04-10T05:33:10.408264Z","shell.execute_reply.started":"2025-04-10T05:33:10.400846Z","shell.execute_reply":"2025-04-10T05:33:10.407420Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Example intrinsic matrix K (adjust with your real values)\nfx = fy = 1200  # focal lengths\ncx = cy = 640   # principal point (image center)\nintrinsics = np.array([\n    [fx,  0, cx],\n    [ 0, fy, cy],\n    [ 0,  0,  1]\n], dtype=np.float32)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T05:33:12.799232Z","iopub.execute_input":"2025-04-10T05:33:12.799558Z","iopub.status.idle":"2025-04-10T05:33:12.803710Z","shell.execute_reply.started":"2025-04-10T05:33:12.799532Z","shell.execute_reply":"2025-04-10T05:33:12.802780Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"workdir = '/kaggle/working/result/'\nos.makedirs(workdir, exist_ok=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T05:33:15.018228Z","iopub.execute_input":"2025-04-10T05:33:15.018525Z","iopub.status.idle":"2025-04-10T05:33:15.022802Z","shell.execute_reply.started":"2025-04-10T05:33:15.018503Z","shell.execute_reply":"2025-04-10T05:33:15.021986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reconstruct_scene(image_paths, intrinsics, feature_dir='.featureout', match_file='matches.h5', output_dir='output'):\n    os.makedirs(output_dir, exist_ok=True)\n    \n    # Load keypoints\n    with h5py.File(os.path.join(feature_dir, 'keypoints.h5'), 'r') as f_kp, \\\n         h5py.File(os.path.join(feature_dir, match_file), 'r') as f_match:\n        \n        # Choose first pair\n        img1, img2 = image_paths[0], image_paths[1]\n        name1, name2 = os.path.basename(img1), os.path.basename(img2)\n\n        kpts1 = f_kp[name1][()]\n        kpts2 = f_kp[name2][()]\n        \n        matches = f_match[name1][name2][()]\n        valid = matches >= 0\n        idx1 = np.where(valid)[0]\n        idx2 = matches[valid]\n        \n        pts1 = kpts1[idx1]\n        pts2 = kpts2[idx2]\n\n        # Estimate Essential matrix\n        E, mask = cv2.findEssentialMat(pts1, pts2, intrinsics, method=cv2.RANSAC, prob=0.999, threshold=1.0)\n        print(f\"[INFO] Essential matrix shape: {E.shape}\")\n\n        # Recover relative pose\n        _, R, t, mask_pose = cv2.recoverPose(E, pts1, pts2, intrinsics)\n        print(\"[INFO] Recovered pose\")\n\n        # Projection matrices\n        P1 = intrinsics @ np.hstack((np.eye(3), np.zeros((3, 1))))  # Camera 1\n        P2 = intrinsics @ np.hstack((R, t))  # Camera 2\n\n        # Triangulate\n        pts1_h = cv2.convertPointsToHomogeneous(pts1[mask_pose.ravel() == 1])[:, 0, :]\n        pts2_h = cv2.convertPointsToHomogeneous(pts2[mask_pose.ravel() == 1])[:, 0, :]\n\n        pts4d_hom = cv2.triangulatePoints(P1, P2, pts1[mask_pose.ravel() == 1].T, pts2[mask_pose.ravel() == 1].T)\n        pts3d = (pts4d_hom[:3] / pts4d_hom[3]).T  # Convert to 3D\n\n        print(f\"[INFO] Triangulated {pts3d.shape[0]} 3D points\")\n\n        # Save to PLY\n        save_ply(os.path.join(output_dir, 'reconstruction.ply'), pts3d)\n\ndef save_ply(filename, points, colors=None):\n    with open(filename, 'w') as f:\n        f.write('ply\\n')\n        f.write('format ascii 1.0\\n')\n        f.write(f'element vertex {len(points)}\\n')\n        f.write('property float x\\n')\n        f.write('property float y\\n')\n        f.write('property float z\\n')\n        if colors is not None:\n            f.write('property uchar red\\n')\n            f.write('property uchar green\\n')\n            f.write('property uchar blue\\n')\n        f.write('end_header\\n')\n        for i, pt in enumerate(points):\n            line = f\"{pt[0]} {pt[1]} {pt[2]}\"\n            if colors is not None:\n                c = colors[i]\n                line += f\" {int(c[0])} {int(c[1])} {int(c[2])}\"\n            f.write(line + '\\n')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T05:33:19.270661Z","iopub.execute_input":"2025-04-10T05:33:19.271010Z","iopub.status.idle":"2025-04-10T05:33:19.281268Z","shell.execute_reply.started":"2025-04-10T05:33:19.270973Z","shell.execute_reply":"2025-04-10T05:33:19.280370Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reconstruct_scene_sfm(image_paths, intrinsics, feature_dir='.featureout', match_file='matches.h5', output_dir='output'):\n    os.makedirs(output_dir, exist_ok=True)\n    \n    # Load keypoints and matches\n    with h5py.File(os.path.join(feature_dir, 'keypoints.h5'), 'r') as f_kp, \\\n         h5py.File(os.path.join(feature_dir, match_file), 'r') as f_match:\n        \n        # Load keypoints from the first image pair\n        img1, img2 = image_paths[0], image_paths[1]\n        name1, name2 = os.path.basename(img1), os.path.basename(img2)\n             \n\n        kpts1 = f_kp[name1][()]\n        kpts2 = f_kp[name2][()]\n        matches = f_match[name1][name2][()]\n        \n\n        valid = matches >= 0\n        idx1 = np.where(valid)[0]\n        idx2 = matches[valid]\n        \n        pts1 = kpts1[idx1]\n        pts2 = kpts2[idx2]\n             \n        # Estimate the Fundamental matrix (for camera pose)\n        F, mask = cv2.findFundamentalMat(pts1, pts2, method=cv2.FM_RANSAC)\n        print(f\"[INFO] Fundamental matrix shape: {F.shape}\")\n\n        # Recover relative pose (rotation and translation)\n        _, R, t, mask_pose = cv2.recoverPose(F, pts1, pts2, intrinsics)\n        print(\"[INFO] Recovered relative pose\")\n             \n        if mask_pose is None:\n            raise ValueError(\"[ERROR] pose recovery failed, mask is None\")\n        \n\n        # Create Projection matrices\n        P1 = np.hstack((np.eye(3), np.zeros((3, 1))))  # Camera 1\n        P2 = np.hstack((R, t))  # Camera 2\n\n        # Triangulate 3D points from matching points in the two images\n        pts1_h = cv2.convertPointsToHomogeneous(pts1[mask_pose.ravel() == 1])[:, 0, :]\n        pts2_h = cv2.convertPointsToHomogeneous(pts2[mask_pose.ravel() == 1])[:, 0, :]\n\n        pts4d_hom = cv2.triangulatePoints(P1, P2, pts1[mask_pose.ravel() == 1].T, pts2[mask_pose.ravel() == 1].T)\n        pts3d = (pts4d_hom[:3] / pts4d_hom[3]).T  # Convert to 3D\n\n        print(f\"[INFO] Triangulated {pts3d.shape[0]} 3D points\")\n\n        # Save to PLY\n        save_ply(os.path.join(output_dir, 'reconstruction_sfm.ply'), pts3d)\n\ndef save_ply(filename, points, colors=None):\n    with open(filename, 'w') as f:\n        f.write('ply\\n')\n        f.write('format ascii 1.0\\n')\n        f.write(f'element vertex {len(points)}\\n')\n        f.write('property float x\\n')\n        f.write('property float y\\n')\n        f.write('property float z\\n')\n        if colors is not None:\n            f.write('property uchar red\\n')\n            f.write('property uchar green\\n')\n            f.write('property uchar blue\\n')\n        f.write('end_header\\n')\n        for i, pt in enumerate(points):\n            line = f\"{pt[0]} {pt[1]} {pt[2]}\"\n            if colors is not None:\n                c = colors[i]\n                line += f\" {int(c[0])} {int(c[1])} {int(c[2])}\"\n            f.write(line + '\\n')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T05:33:26.425508Z","iopub.execute_input":"2025-04-10T05:33:26.425811Z","iopub.status.idle":"2025-04-10T05:33:26.436125Z","shell.execute_reply.started":"2025-04-10T05:33:26.425788Z","shell.execute_reply":"2025-04-10T05:33:26.435248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ------------------------------------------------------------------------------\n# STEP 1: Relative Pose Estimation from Deep Matches\n# ------------------------------------------------------------------------------\ndef compute_relative_pose_from_deep_matches(kp1, kp2, matches, K):\n    \"\"\"\n    Given keypoints from two images and deep matching index pairs, estimate the\n    relative pose (rotation and translation) using classical geometry.\n    \n    Parameters:\n      kp1: np.array of shape (N1, 2) – keypoints from image 1\n      kp2: np.array of shape (N2, 2) – keypoints from image 2\n      matches: np.array of shape (M, 2) – each row contains indices [i, j] where\n               kp1[i] matches kp2[j]\n      K: 3x3 intrinsic camera matrix\n      \n    Returns:\n      R: 3x3 relative rotation matrix (from image1 to image2)\n      t: 3x1 relative translation vector\n      mask: inlier mask from cv2.recoverPose (optional for debugging)\n    \"\"\"\n    # Select matching keypoint coordinates\n    pts1 = kp1[matches[:, 0]]\n    pts2 = kp2[matches[:, 1]]\n    \n    # Compute the Essential matrix robustly using RANSAC.\n    # The function returns the Essential matrix and an inlier mask.\n    E, mask = cv2.findEssentialMat(pts1, pts2, K, method=cv2.RANSAC, prob=0.999, threshold=1.0)\n    \n    # Recover the pose (rotation and translation) from the Essential matrix.\n    # cv2.recoverPose uses the inlier keypoints to return the best estimate.\n    _, R, t, mask_pose = cv2.recoverPose(E, pts1, pts2, K)\n    return R, t, mask","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T05:33:30.199496Z","iopub.execute_input":"2025-04-10T05:33:30.199802Z","iopub.status.idle":"2025-04-10T05:33:30.205313Z","shell.execute_reply.started":"2025-04-10T05:33:30.199778Z","shell.execute_reply":"2025-04-10T05:33:30.204352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ------------------------------------------------------------------------------\n# STEP 2: Build a Relative Pose Graph from Stored Matches\n# ------------------------------------------------------------------------------\ndef build_relative_pose_graph(feature_dir, K):\n    \"\"\"\n    Reads in the saved keypoints and matches from HDF5 files, and computes a graph\n    where each node is an image (identified by its filename) and each edge contains\n    the relative pose between the two images.\n    \n    Parameters:\n      feature_dir: Directory containing 'keypoints.h5' and 'matches.h5'\n      K: Intrinsic camera matrix\n    \n    Returns:\n      G: NetworkX undirected graph with each edge storing keys 'R' and 't'\n    \"\"\"\n    G = nx.Graph()\n    \n    with h5py.File(os.path.join(feature_dir, 'keypoints.h5'), 'r') as f_kp, \\\n         h5py.File(os.path.join(feature_dir, 'matches.h5'), 'r') as f_match:\n        \n        # The matches file is organized as f_match[img1][img2]\n        for key1 in f_match.keys():\n            for key2 in f_match[key1].keys():\n                # Load the matches (each row has two indices)\n                matches = f_match[key1][key2][...]\n                # Load keypoints for the two images (assume shape: (num_points, 2))\n                kp1 = np.array(f_kp[key1][...])\n                kp2 = np.array(f_kp[key2][...])\n                \n                try:\n                    # Compute the relative pose using deep matching and geometry.\n                    R, t, mask = compute_relative_pose_from_deep_matches(kp1, kp2, matches, K)\n                    # Add this relative transformation as an edge attribute in the graph.\n                    G.add_edge(key1, key2, R=R, t=t)\n                except Exception as e:\n                    print(f\"Failed to compute relative pose for pair ({key1}, {key2}): {e}\")\n    \n    return G","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T05:33:32.738634Z","iopub.execute_input":"2025-04-10T05:33:32.738938Z","iopub.status.idle":"2025-04-10T05:33:32.745079Z","shell.execute_reply.started":"2025-04-10T05:33:32.738903Z","shell.execute_reply":"2025-04-10T05:33:32.744244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ------------------------------------------------------------------------------\n# STEP 3: Propagate a Global Pose via a Spanning Tree\n# ------------------------------------------------------------------------------\ndef compute_global_poses(G):\n    \"\"\"\n    Using the relative pose graph, compute a simplified set of global poses. The root\n    image (first in the graph) is fixed as identity. Then, using a BFS traversal,\n    we propagate the relative poses to obtain an initial guess for each image’s global\n    orientation and position.\n    \n    Parameters:\n      G: NetworkX graph with edge attributes 'R' and 't'\n    \n    Returns:\n      global_poses: dict mapping image filename (node) to tuple (R, t) as global pose.\n    \"\"\"\n    global_poses = {}\n    # Choose an arbitrary root node and set its pose as identity.\n    root = list(G.nodes())[0]\n    global_poses[root] = (np.eye(3), np.zeros((3, 1)))\n    \n    # BFS traversal to propagate the poses.\n    visited = set([root])\n    queue = [root]\n    \n    while queue:\n        current = queue.pop(0)\n        current_R, current_t = global_poses[current]\n        \n        # Iterate over neighbors (images that have a relative transformation with current)\n        for neighbor in G.neighbors(current):\n            if neighbor not in visited:\n                edge_data = G.get_edge_data(current, neighbor)\n                R_edge = edge_data['R']\n                t_edge = edge_data['t']\n                \n                # Propagate the pose.\n                # If the relative pose from current to neighbor is (R_edge, t_edge),\n                # then the global pose of the neighbor is:\n                #   R_neighbor = R_edge @ current_R\n                #   t_neighbor = R_edge @ current_t + t_edge\n                neighbor_R = R_edge @ current_R\n                neighbor_t = R_edge @ current_t + t_edge\n                global_poses[neighbor] = (neighbor_R, neighbor_t)\n                \n                visited.add(neighbor)\n                queue.append(neighbor)\n    \n    return global_poses","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T05:33:35.482351Z","iopub.execute_input":"2025-04-10T05:33:35.482784Z","iopub.status.idle":"2025-04-10T05:33:35.492838Z","shell.execute_reply.started":"2025-04-10T05:33:35.482752Z","shell.execute_reply":"2025-04-10T05:33:35.491783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# A list to store prediction output for every processed image.\npredictions = []\nid_dict = {}\n\n# Loop over each dataset/scene group from your train_labels.\n# grouped = train_labels.groupby(['dataset', 'scene'])\nfor dataset, predictions in samples.items():\n    print(f\"Processing: dataset = {dataset}, scene = {predictions}\")\n    \n    # Define a feature directory per dataset.\n    feature_dir = os.path.join(workdir, 'featureout', dataset)\n    os.makedirs(feature_dir, exist_ok=True)\n    \n    # Form full image file paths (adjust DATA_DIR according to your project)\n    image_paths = group[\"image\"].apply(lambda img: os.path.join(DATA_DIR, dataset, img)).tolist()\n    \n    # Step 1: Global Descriptor Extraction using DINOv2.\n    descriptors = extract_dino_descriptor(image_paths)\n    \n    # Compute cosine similarity between descriptors.\n    sim_matrix = descriptors @ descriptors.T  # shape: (num_images x num_images)\n    num_images = len(image_paths)\n    \n    # Mask self-similarity entries.\n    sim_matrix.fill_diagonal_(-1.0)\n    \n    # Shortlist image pairs based on a similarity threshold.\n    shortlisted_pairs = set()\n    for i in range(num_images):\n        sim_scores = sim_matrix[i]\n        top_similar = (sim_scores >= sim_th).nonzero(as_tuple=True)[0].tolist()\n        \n        # If the number of similar images is below a set minimum, take top-k.\n        if len(top_similar) < min_pairs:\n            num_valid = sim_scores.numel()\n            k = min(min_pairs, num_valid)\n            top_similar = torch.topk(sim_scores, k).indices.tolist()\n            \n        for j in top_similar:\n            if i == j:\n                continue\n            pair = tuple(sorted((i, j)))\n            shortlisted_pairs.add(pair)\n    \n    # Convert shortlisted pairs to a list containing tuples of (img_path1, img_path2, similarity)\n    shortlisted_images_result = [\n        (image_paths[i], image_paths[j], sim_matrix[i, j].item())\n        for i, j in shortlisted_pairs\n    ]\n    \n    # Optionally, limit the number of pairs globally if topN is set.\n    if topN is not None:\n        shortlisted_images_result = sorted(shortlisted_images_result, key=lambda x: -x[2])[:topN]\n    \n    # Step 2: Local Feature Extraction using ALIKED.\n    t = time()\n    detect_aliked(image_paths, feature_dir, num_features=4096, resize_to=1024, device=device)\n    gc.collect()\n    timings['feature_detection'].append(time() - t)\n    print(f'Features detected in {time() - t:.4f} sec')\n    \n    # Step 3: Feature Matching using LightGlue.\n    t = time()\n    match_with_lightglue(image_paths, shortlisted_images_result, feature_dir=feature_dir, device=device, verbose=False)\n    gc.collect()\n    timings['feature_matching'].append(time() - t)\n    print(f'Features matched in {time() - t:.4f} sec')\n    \n    # Step 4: Build the deep-enhanced relative pose graph and compute global poses.\n    pose_graph = build_relative_pose_graph(feature_dir, intrinsics)\n    global_poses = compute_global_poses(pose_graph)\n    \n    # Step 5: Store the predictions from this group.\n    # Here each item in global_poses has the filename as key, and (R, t) as value.\n    for image_file, (R, t_vec) in global_poses.items():\n        image_id = id_dict.get(image_file, None)\n        print(image_id)\n        # We store the dataset, scene, image filename, and corresponding global pose.\n        prediction = {\n            'dataset': dataset,\n            'scene': scene,\n            'image_id': image_id,\n            'image': image_file,  # image_file should be a filename, e.g., \"img001.jpg\"\n            'rotation_matrix': R,  # 3x3 rotation matrix\n            'translation_vector': t_vec.flatten()  # flatten in case it's a column vector\n        }\n        predictions.append(prediction)\n    \n    # reconstruct_scene_sfm(\n    #     image_paths=image_paths,\n    #     intrinsics=intrinsics,\n    #     feature_dir=feature_dir,\n    #     match_file='matches.h5',\n    #     output_dir=os.path.join('output', f'{dataset}_{scene}')\n    # )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T05:34:11.405042Z","iopub.execute_input":"2025-04-10T05:34:11.405364Z","iopub.status.idle":"2025-04-10T05:34:11.441874Z","shell.execute_reply.started":"2025-04-10T05:34:11.405341Z","shell.execute_reply":"2025-04-10T05:34:11.440937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ------------------------------------------------------------------------------\n# Generate the Submission CSV File\n# ------------------------------------------------------------------------------\n\n# Define formatting helper functions.\narray_to_str = lambda array: ';'.join([f\"{x:.09f}\" for x in array])\nnone_to_str = lambda n: ';'.join(['nan'] * n)\n\n# Set submission file path (e.g., for Kaggle working directory)\nsubmission_file = '/kaggle/working/submission.csv'\nwith open(submission_file, 'w') as f:\n    # Write header line.\n    f.write('image_id, dataset,scene,image,rotation_matrix,translation_vector\\n')\n    # Write one line per prediction.\n    for prediction in predictions:\n        rotation = none_to_str(9) if prediction['rotation_matrix'] is None else array_to_str(prediction['rotation_matrix'].flatten())\n        translation = none_to_str(3) if prediction['translation_vector'] is None else array_to_str(prediction['translation_vector'])\n        f.write(f\"{prediction['image_id']},{prediction['dataset']},{prediction['scene']},{prediction['image']},{rotation},{translation}\\n\")\n        \n# Optionally, if using Kaggle Notebook, preview the file.\n!head {submission_file}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-09T19:44:50.913491Z","iopub.execute_input":"2025-04-09T19:44:50.913819Z","iopub.status.idle":"2025-04-09T19:44:51.110682Z","shell.execute_reply.started":"2025-04-09T19:44:50.913794Z","shell.execute_reply":"2025-04-09T19:44:51.109717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(descriptors)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import matplotlib.pyplot as plt\n# from PIL import Image\n\n# def visualize_similar_image_pairs(pairs, max_pairs=10, title=\"DINO Shortlisted Similar Images\"):\n#     plt.figure(figsize=(12, 3 * max_pairs))\n\n#     for idx, (img1_path, img2_path, score) in enumerate(pairs[:max_pairs]):\n#         img1 = Image.open(img1_path).convert(\"RGB\")\n#         img2 = Image.open(img2_path).convert(\"RGB\")\n\n#         # Show first image\n#         ax1 = plt.subplot(max_pairs, 2, 2 * idx + 1)\n#         ax1.imshow(img1)\n#         ax1.set_title(f\"Image 1\\n{os.path.basename(img1_path)}\")\n#         ax1.axis(\"off\")\n\n#         # Show second image\n#         ax2 = plt.subplot(max_pairs, 2, 2 * idx + 2)\n#         ax2.imshow(img2)\n#         ax2.set_title(f\"Image 2\\n{os.path.basename(img2_path)}\\nSimilarity: {score:.3f}\")\n#         ax2.axis(\"off\")\n\n#     plt.suptitle(title, fontsize=16)\n#     plt.tight_layout()\n#     plt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# visualize_similar_image_pairs(shortlisted_iamges_result, max_pairs=10)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def visualize_matches(img_fnames, index_pairs, feature_dir=feature_dir, device=torch.device('cpu')):\n#     with h5py.File(f'{feature_dir}/keypoints.h5', mode='r') as f_kp, \\\n#          h5py.File(f'{feature_dir}/descriptors.h5', mode='r') as f_desc, \\\n#          h5py.File(f'{feature_dir}/matches.h5', mode='r') as f_match:\n        \n#         for pair_idx in index_pairs:\n#             fname1, fname2, _ = pair_idx  # Ignore the third element (the score)\n#             key1, key2 = fname1.split('/')[-1], fname2.split('/')[-1]\n#             kp1 = f_kp[key1][...]\n#             kp2 = f_kp[key2][...]\n#             matches = f_match[key1][key2][...]  # Matches for the pair\n\n#             # Load the images\n#             img1 = cv2.imread(fname1)\n#             img2 = cv2.imread(fname2)\n\n#             # Convert images to RGB (OpenCV loads them in BGR)\n#             img1_rgb = cv2.cvtColor(img1, cv2.COLOR_BGR2RGB)\n#             img2_rgb = cv2.cvtColor(img2, cv2.COLOR_BGR2RGB)\n\n#             # Draw keypoints on the images\n#             img1_kp = img1_rgb.copy()\n#             img2_kp = img2_rgb.copy()\n\n#             for kp in kp1:\n#                 cv2.circle(img1_kp, tuple(kp.astype(int)), 5, (255, 0, 0), -1)  # Red keypoints\n#             for kp in kp2:\n#                 cv2.circle(img2_kp, tuple(kp.astype(int)), 5, (0, 0, 255), -1)  # Blue keypoints\n\n#             # Create a new image to concatenate both images side by side\n#             concat_img = np.concatenate([img1_rgb, img2_rgb], axis=1)\n\n#             # Draw lines between matched keypoints\n#             for match in matches:\n#                 pt1 = tuple(kp1[match[0]].astype(int))\n#                 pt2 = tuple(kp2[match[1]].astype(int))\n#                 pt2 = (pt2[0] + img1.shape[1], pt2[1])  # Adjust second image point position\n\n#                 # Draw a line between the points on the concatenated image\n#                 cv2.line(concat_img, pt1, pt2, (0, 255, 0), 2)  # Green lines for matches\n\n#             # Show the final image\n#             plt.figure(figsize=(12, 6))\n#             plt.imshow(concat_img)\n#             plt.axis('off')  # Turn off axis\n#             plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# visualize_matches(image_paths, shortlisted_iamges_result, device=torch.device('cpu'))","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}