{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91498,"databundleVersionId":11655853,"sourceType":"competition"},{"sourceId":7884485,"sourceType":"datasetVersion","datasetId":4628051},{"sourceId":11691214,"sourceType":"datasetVersion","datasetId":7338042},{"sourceId":11717468,"sourceType":"datasetVersion","datasetId":7355327},{"sourceId":11752663,"sourceType":"datasetVersion","datasetId":7378152},{"sourceId":11815060,"sourceType":"datasetVersion","datasetId":7421016},{"sourceId":11924468,"sourceType":"datasetVersion","datasetId":6988459},{"sourceId":12016115,"sourceType":"datasetVersion","datasetId":7559779},{"sourceId":4534,"sourceType":"modelInstanceVersion","modelInstanceId":3326,"modelId":986},{"sourceId":17191,"sourceType":"modelInstanceVersion","modelInstanceId":14317,"modelId":21716},{"sourceId":17555,"sourceType":"modelInstanceVersion","modelInstanceId":14611,"modelId":22086}],"dockerImageVersionId":31011,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Initial Setup (Imports + Installation)","metadata":{}},{"cell_type":"code","source":"import os\nos.environ[\"GLOG_minloglevel\"] = \"2\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:00:15.546702Z","iopub.execute_input":"2025-06-02T19:00:15.546958Z","iopub.status.idle":"2025-06-02T19:00:15.554893Z","shell.execute_reply.started":"2025-06-02T19:00:15.546932Z","shell.execute_reply":"2025-06-02T19:00:15.554285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\n\n# 1. Install pre-downloaded packages (no internet needed)\n!pip install --no-index /kaggle/input/kornia-0-8-0-wheel/kornia-0.8.0-py2.py3-none-any.whl --no-deps\n!pip install --no-index /kaggle/input/opencv-4-5-5-64-whl/opencv_python-4.5.5.64-cp36-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl --no-deps\n#!pip install --no-index /kaggle/input/pycolmap/numpy-2.2.5-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl --no-deps\n!pip install --no-index --force-reinstall /kaggle/input/pycolmap-0-5-0/pycolmap-0.5.0-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl --no-deps\n\n!pip install --no-index /kaggle/input/hdbscan-0-8-40-3/hdbscan-0.8.40-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl --no-deps\n\n# !pip install --no-index /kaggle/input/imc2024-packages-lightglue-rerun-kornia/kornia-0.7.2-py2.py3-none-any.whl --no-deps\n!pip install --no-index /kaggle/input/imc2024-packages-lightglue-rerun-kornia/lightglue-0.0-py3-none-any.whl --no-deps\n\n# Copy model weights\n!mkdir -p /root/.cache/torch/hub/checkpoints\n!cp /kaggle/input/aliked/pytorch/aliked-n16/1/aliked-n16.pth /root/.cache/torch/hub/checkpoints/\n!cp /kaggle/input/lightglue/pytorch/aliked/1/aliked_lightglue.pth /root/.cache/torch/hub/checkpoints/\n!mv /root/.cache/torch/hub/checkpoints/aliked_lightglue.pth /root/.cache/torch/hub/checkpoints/aliked_lightglue_v0-1_arxiv.pth\n\n# 3. Verify\nimport cv2, kornia, pycolmap\nprint(f\"Versions: OpenCV={cv2.__version__}, Kornia={kornia.__version__}, PyCOLMAP={pycolmap.__version__}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:00:15.556317Z","iopub.execute_input":"2025-06-02T19:00:15.556572Z","iopub.status.idle":"2025-06-02T19:00:35.170636Z","shell.execute_reply.started":"2025-06-02T19:00:15.556549Z","shell.execute_reply":"2025-06-02T19:00:35.169837Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport sys\nfrom pathlib import Path\nimport matplotlib.pyplot as plt\nimport kornia as K\nimport kornia.feature as KF\nfrom kornia.io import load_image, ImageLoadType\nfrom abc import ABC, abstractmethod\nimport h5py\nfrom tqdm import tqdm\nfrom tqdm.notebook import tqdm\nimport shutil\nimport random\nimport math\nfrom IPython.display import Markdown, display, clear_output\nfrom sklearn.neighbors import NearestNeighbors\nfrom sklearn.cluster import DBSCAN\nimport networkx as nx\nimport community as community_louvain\nfrom joblib import Parallel, delayed\nimport os\nimport pycolmap\nfrom collections import defaultdict\nfrom sklearn.decomposition import PCA\nfrom transformers import AutoImageProcessor, AutoModel\nimport torch.nn.functional as F\n\nfrom lightglue import ALIKED, LightGlue\nfrom lightglue.utils import load_image as load_lg_image, rbd\n\nprint(\" ALIKED and LightGlue imported successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:00:45.380010Z","iopub.execute_input":"2025-06-02T19:00:45.380677Z","iopub.status.idle":"2025-06-02T19:01:03.474930Z","shell.execute_reply.started":"2025-06-02T19:00:45.380651Z","shell.execute_reply":"2025-06-02T19:01:03.474155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_device():\n    \"\"\"Safer device detection\"\"\"\n    if torch.cuda.is_available():\n        try:\n            # Verify CUDA is actually usable\n            torch.zeros(1).cuda()\n            return torch.device('cuda')\n        except:\n            pass\n    return torch.device('cpu')\n\ndevice = get_device()\nprint(f\"Using device: {device}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:01:20.068529Z","iopub.execute_input":"2025-06-02T19:01:20.069455Z","iopub.status.idle":"2025-06-02T19:01:20.352988Z","shell.execute_reply.started":"2025-06-02T19:01:20.069396Z","shell.execute_reply":"2025-06-02T19:01:20.352208Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Universal Interfaces","metadata":{}},{"cell_type":"code","source":"class FeatureExtractor(ABC):\n    @abstractmethod\n    def extract_features(self, img_path: str) -> tuple[np.ndarray, np.ndarray]:\n        \"\"\"Returns (keypoints_Nx2, descriptors_Nx128)\"\"\"\n        pass\n\nclass FeatureMatcher(ABC):\n    @abstractmethod\n    def match(\n        self, \n        features1: dict,  # {'keypoints': np.array, 'descriptors': np.array}\n        features2: dict\n    ) -> np.ndarray:  # Return Mx2 array of matched indices\n        pass","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:01:21.278545Z","iopub.execute_input":"2025-06-02T19:01:21.279030Z","iopub.status.idle":"2025-06-02T19:01:21.283312Z","shell.execute_reply.started":"2025-06-02T19:01:21.279006Z","shell.execute_reply":"2025-06-02T19:01:21.282767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ALIKEDExtractor(FeatureExtractor):\n    def __init__(self, max_kpts=8000, resize=1024, device=None):\n        self.device = device or get_device()\n        self.model = ALIKED(\n            max_num_keypoints=max_kpts,\n            detection_threshold=0.02,\n            resize=resize\n        ).eval().to(self.device)\n\n    def extract_features(self, img_path):\n        with torch.no_grad():\n            img_tensor = load_lg_image(str(img_path), resize=getattr(self.model, \"resize\", None)).to(self.device)\n            feats = self.model.extract(img_tensor)\n    \n            keypoints = feats['keypoints'][0].cpu().numpy()\n            descriptors = feats['descriptors'][0].cpu().numpy()\n    \n        return keypoints, descriptors","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:01:22.263479Z","iopub.execute_input":"2025-06-02T19:01:22.263746Z","iopub.status.idle":"2025-06-02T19:01:22.269069Z","shell.execute_reply.started":"2025-06-02T19:01:22.263726Z","shell.execute_reply":"2025-06-02T19:01:22.268293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Verbus\nVERBOSE = True # or True if you want to enable logs\n\n#  Configuration\nIS_TRAIN = False\nDATA_DIR = Path('/kaggle/input/image-matching-challenge-2025')\nTEST_DIR = DATA_DIR / 'test'\nOUTPUT_FEATURES_DIR = Path('/kaggle/working/output_features')\n\n# Device setup\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nextractor = ALIKEDExtractor(device=device)\nprint(f\" ALIKED Extractor loaded on {next(extractor.model.parameters()).device}\")\n\n# Load sample_submission and dataset mapping\nsample_df = pd.read_csv('/kaggle/input/image-matching-challenge-2025/sample_submission.csv')\nimage_to_dataset = dict(zip(sample_df['image'], sample_df['dataset']))\nprint(\" Loaded dataset assignments from sample_submission.csv\")\n\n# 🔍 Find all image files with valid suffixes (jpg, png, jpeg, etc.)\nVALID_SUFFIXES = {'.png', '.jpg', '.jpeg', '.webp', '.bmp', '.tif', '.tiff'}\n\ntest_images_all = [\n    p for p in TEST_DIR.rglob(\"*\")\n    if p.suffix.lower() in VALID_SUFFIXES and p.is_file()\n]\n\n# 🔎 Keep only those that are listed in sample_submission.csv\ntest_images = []\nfor img_path in test_images_all:\n    img_name = img_path.name\n    if img_name in image_to_dataset:\n        test_images.append((img_name, img_path))  # (filename, full path)\n\nprint(f\" Found {len(test_images)} test images via sample_submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:04:12.943195Z","iopub.execute_input":"2025-06-02T19:04:12.943537Z","iopub.status.idle":"2025-06-02T19:04:13.022821Z","shell.execute_reply.started":"2025-06-02T19:04:12.943515Z","shell.execute_reply":"2025-06-02T19:04:13.022231Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def save_features_flat(image_path: Path, dataset_name: str, extractor: FeatureExtractor, output_dir: Path):\n    try:\n        h5_name = image_path.stem + \".h5\"\n        save_path = output_dir / dataset_name / h5_name\n        save_path.parent.mkdir(parents=True, exist_ok=True)\n\n        kpts, descs = extractor.extract_features(str(image_path))\n\n        with h5py.File(save_path, 'w') as f:\n            f.create_dataset('keypoints', data=kpts)\n            f.create_dataset('descriptors', data=descs)\n            f.attrs['image_path'] = str(image_path)\n\n    except Exception as e:\n        if VERBOSE:\n            print(f\" Failed on {image_path.name}: {str(e)}\")\n\n#  Run extraction\nloop = test_images if not VERBOSE else tqdm(test_images, desc=\" Extracting features\")\n\nfor img_name, img_path in loop:\n    dataset = image_to_dataset.get(img_name)\n    save_features_flat(img_path, dataset, extractor, OUTPUT_FEATURES_DIR)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:04:15.673774Z","iopub.execute_input":"2025-06-02T19:04:15.674525Z","iopub.status.idle":"2025-06-02T19:04:22.621212Z","shell.execute_reply.started":"2025-06-02T19:04:15.674501Z","shell.execute_reply":"2025-06-02T19:04:22.620303Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nimport h5py\n\n# Set this to match your output path\nfeature_dir = Path(\"/kaggle/working/output_features\")\n\n# Find all .h5 feature files\nfeature_files = sorted(feature_dir.rglob(\"*.h5\"))\nif VERBOSE:\n    print(f\" Found {len(feature_files)} feature files\")\n\n# Inspect first few\nfor f in feature_files[:3]:\n    if VERBOSE:\n        print(f\"\\n File: {f}\")\n    with h5py.File(f, 'r') as h5f:\n        if 'keypoints' not in h5f or 'descriptors' not in h5f:\n            if VERBOSE:\n                print(\" Missing keypoints or descriptors\")\n            continue\n\n        kpts = h5f['keypoints'][:]\n        descs = h5f['descriptors'][:]\n        if VERBOSE:\n            print(f\"  - Keypoints shape   : {kpts.shape}\")\n            print(f\"  - Descriptors shape : {descs.shape}\")\n            print(f\"  - Image path attr   : {h5f.attrs.get('image_path', ' missing')}\")\n\n        if kpts.shape[0] != descs.shape[0] and VERBOSE:\n            print(\" Mismatch between keypoints and descriptors count\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:04:22.622484Z","iopub.execute_input":"2025-06-02T19:04:22.622738Z","iopub.status.idle":"2025-06-02T19:04:22.635826Z","shell.execute_reply.started":"2025-06-02T19:04:22.622721Z","shell.execute_reply":"2025-06-02T19:04:22.635165Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 7. Feature Matching","metadata":{}},{"cell_type":"markdown","source":"### NN Matcher Implementation","metadata":{}},{"cell_type":"markdown","source":"# LightGlue","metadata":{}},{"cell_type":"code","source":"class LightGlueMatcher(FeatureMatcher):\n    def __init__(self, min_matches=20, device=None, ransac_thresh=4.0, max_iters=3000):\n        \"\"\"\n        Now requires:\n          • min_matches: minimum number of matches to retain\n          • ransac_thresh, max_iters: RANSAC parameters\n        \"\"\"\n        self.device = device or torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n        self.min_matches = min_matches\n        self.ransac_thresh = ransac_thresh\n        self.max_iters = max_iters\n\n        # -------------------------------------------------------------------\n        # Replace the default constructor with one that requests both\n        # confidence scores and mixed‐precision.\n        glue_cfg = {\n            \"width_confidence\": -1,\n            \"depth_confidence\": -1,\n            \"mp\": True\n        }\n        self.model = LightGlue('aliked', config=glue_cfg).eval().to(self.device)\n        # -------------------------------------------------------------------\n\n    def match(self, features1: dict, features2: dict) -> np.ndarray:\n        \"\"\"\n        features1, features2 each contain:\n          • 'keypoints': (N×2 array)\n          • 'descriptors': (N×128 array)\n\n        Returns an M×2 array of inlier index pairs, or empty if none pass min_matches.\n        \"\"\"\n        def _run_lightglue(f1: dict, f2: dict) -> np.ndarray:\n            \"\"\"Run LightGlue on two already‐extracted feature dicts, return raw ℕ×2 matches.\"\"\"\n            kpts1 = torch.from_numpy(f1['keypoints']).unsqueeze(0).to(self.device)\n            kpts2 = torch.from_numpy(f2['keypoints']).unsqueeze(0).to(self.device)\n            desc1 = torch.from_numpy(f1['descriptors']).unsqueeze(0).to(self.device)\n            desc2 = torch.from_numpy(f2['descriptors']).unsqueeze(0).to(self.device)\n            with torch.no_grad():\n                output = self.model({\n                    'image0': {'keypoints': kpts1, 'descriptors': desc1},\n                    'image1': {'keypoints': kpts2, 'descriptors': desc2},\n                })\n            m0 = output['matches0'][0].cpu().numpy()\n            valid = np.where(m0 != -1)[0]\n            return np.stack([valid, m0[valid]], axis=-1)  # raw matches array\n\n        def _ransac_filter(raw: np.ndarray, f1: dict, f2: dict) -> np.ndarray:\n            \"\"\"\n            Apply Fundamental‐matrix RANSAC to raw matches (array of shape M×2).\n            Returns only the inlier subset (K×2), or empty if <min_matches.\n            \"\"\"\n            if raw.shape[0] < self.min_matches:\n                return np.zeros((0, 2), dtype=np.int32)\n\n            src = f1['keypoints'][raw[:, 0]]\n            dst = f2['keypoints'][raw[:, 1]]\n            F, mask = cv2.findFundamentalMat(\n                src, dst,\n                method=cv2.FM_RANSAC,\n                ransacReprojThreshold=self.ransac_thresh,\n                confidence=0.999,\n                maxIters=self.max_iters\n            )\n            if mask is None:\n                return np.zeros((0, 2), dtype=np.int32)\n\n            keep = mask.flatten() == 1\n            inliers = raw[keep]\n            if inliers.shape[0] < self.min_matches:\n                return np.zeros((0, 2), dtype=np.int32)\n            return inliers\n\n        # 1) Run LightGlue and then RANSAC-filter the raw matches\n        raw_matches = _run_lightglue(features1, features2)\n        if VERBOSE:\n            print(f\"LightGlue raw matches: {len(raw_matches)}\", end=\"\")\n        inliers = _ransac_filter(raw_matches, features1, features2)\n        if VERBOSE:\n            print(f\"  → RANSAC inliers: {len(inliers)}\", \n                  \"kept\" if len(inliers) >= self.min_matches else \"skipped\")\n            print()\n        return inliers\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:04:48.556557Z","iopub.execute_input":"2025-06-02T19:04:48.557083Z","iopub.status.idle":"2025-06-02T19:04:48.567036Z","shell.execute_reply.started":"2025-06-02T19:04:48.557064Z","shell.execute_reply":"2025-06-02T19:04:48.566461Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Usage Example and Visualization","metadata":{}},{"cell_type":"markdown","source":"# 8. General Matching and Clustering","metadata":{}},{"cell_type":"markdown","source":"### Load All Feature Paths\n### Precompute Average Descriptors (For Efficient NN Search)\n### Find Approximate Nearest Neighbors","metadata":{}},{"cell_type":"code","source":"from scipy.spatial.distance import cdist\n\n\ndef get_image_feature_paths(base_dir=Path(\"/kaggle/working/output_features\")):\n    \"\"\"Get all .h5 feature file paths, including subfolders\"\"\"\n    return sorted(base_dir.rglob(\"*.h5\"))\n\n\n\ndef get_image_pairs_shortlist(\n    feature_paths,           # List[Path] to your .h5 files (each has f.attrs['image_path'])\n    sim_th=0.60,             # distance threshold on normalized DINOv2 descriptors\n    min_pairs=20,            # ensure each image has ≥ min_pairs neighbors\n    exhaustive_if_less=30,   # if dataset size ≤ this, return all (i,j)\n    device=torch.device('cuda')\n):\n    \"\"\"\n    For each HDF5 in feature_paths:\n      • read f.attrs['image_path'] → actual image file on disk\n      • run the DINOv2 processor→model to get one L2‐normalized 1×D vector\n    Then compute the full N×N distance matrix, keep pairs with dist ≤ sim_th, \n    and pad to min_pairs if needed. If N <= exhaustive_if_less, return all (i<j).\n    \"\"\"\n    N = len(feature_paths)\n    if N <= exhaustive_if_less:\n        return [(i, j) for i in range(N) for j in range(i+1, N)]\n\n    # 1) Build a list of actual image‐file Paths from each HDF5's 'image_path' attr\n    img_paths = []\n    for fp in feature_paths:\n        with h5py.File(fp, 'r') as f:\n            img_str = f.attrs.get('image_path')\n            if img_str is None:\n                raise KeyError(f\"No 'image_path' in {fp}\")\n            img_paths.append(Path(img_str))\n\n    # 2) Load DINOv2 processor & model (just once)\n    processor = AutoImageProcessor.from_pretrained('/kaggle/input/dinov2/pytorch/base/1')\n    model     = AutoModel.from_pretrained('/kaggle/input/dinov2/pytorch/base/1').to(device).eval()\n\n    # 3) Compute one global descriptor per image\n    all_descs = []\n    pbar = tqdm(img_paths, desc=\"DINOv2 descriptors\", disable=not VERBOSE)\n    for img_path in pbar:\n        # load image as 3×H×W float32 tensor, range [0,1]\n        img_tensor = K.io.load_image(\n            str(img_path),\n            K.io.ImageLoadType.RGB32,\n            device=device\n        )[None, ...]  # shape (1,3,H,W)\n\n        with torch.inference_mode():\n            inputs  = processor(images=img_tensor, return_tensors=\"pt\", do_rescale=False).to(device)\n            outputs = model(**inputs)\n            # max‐along‐spatial (skip CLS token index 0)\n            mac_desc = outputs.last_hidden_state[:, 1:].max(dim=1)[0]  # shape (1,D)\n            dino_desc = F.normalize(mac_desc, dim=1, p=2).cpu().numpy()  # (1,D) normalized\n        all_descs.append(dino_desc[0])\n\n    D = np.stack(all_descs, axis=0)  # shape (N, D)\n\n    # 4) Compute pairwise Euclidean distances (equal to cosine on normalized vectors)\n    dist_matrix = cdist(D, D, metric='euclidean')\n\n    # 5) For each image i, pick all j where dist[i,j] ≤ sim_th; if <min_pairs, pad from nearest\n    shortlist = []\n    for i in range(N):\n        # find all neighbors j ≠ i with dist ≤ sim_th\n        nbrs = np.where(dist_matrix[i] <= sim_th)[0].tolist()\n        if i in nbrs:\n            nbrs.remove(i)\n\n        if len(nbrs) < min_pairs:\n            # sort entire row by distance (excluding self), then take top min_pairs\n            sorted_idxs = np.argsort(dist_matrix[i])\n            padded = []\n            for idx in sorted_idxs:\n                if idx == i:\n                    continue\n                padded.append(int(idx))\n                if len(padded) >= min_pairs:\n                    break\n            nbrs = padded\n\n        # record (i,j) for every neighbor j>i (to avoid duplicates)\n        for j in nbrs:\n            if j > i:\n                shortlist.append((i, j))\n            else:\n                shortlist.append((j, i))\n\n    shortlist = sorted(set(shortlist))\n    return shortlist","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:05:11.664798Z","iopub.execute_input":"2025-06-02T19:05:11.665063Z","iopub.status.idle":"2025-06-02T19:05:11.675930Z","shell.execute_reply.started":"2025-06-02T19:05:11.665047Z","shell.execute_reply":"2025-06-02T19:05:11.675225Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Calculate Matches","metadata":{}},{"cell_type":"code","source":"def compute_lightglue_matches_for_clustering(feature_paths, top_k_indices, matcher, pbar=None):\n    from pathlib import Path\n    import h5py\n    import networkx as nx\n    import numpy as np\n    from tqdm import tqdm\n\n    matches_dir = Path('/kaggle/working/output_lightglue_matches')\n    matches_dir.mkdir(exist_ok=True)\n    match_file = matches_dir / 'all_matches.h5'\n\n    all_match_counts = []\n    match_graph = nx.Graph()\n\n    internal_bar = False\n    if pbar is None:\n        total = sum(len(nbrs) for nbrs in top_k_indices)\n        pbar = tqdm(total=total, desc=\"LightGlue Matching\", disable=not VERBOSE)\n        internal_bar = True\n\n    with h5py.File(match_file, 'w') as f_matches:\n        for i, path_i in enumerate(feature_paths):\n            with h5py.File(path_i, 'r') as f1:\n                feats_i = {\n                    'keypoints': f1['keypoints'][()],\n                    'descriptors': f1['descriptors'][()],\n                    'image_path': f1.attrs.get('image_path', str(path_i))\n                }\n\n            group_i = f_matches.require_group(path_i.stem)\n            match_graph.add_node(i, path=feats_i['image_path'])\n\n            for j in top_k_indices[i]:\n                if i == j:\n                    pbar.update(1)\n                    continue\n\n                path_j = feature_paths[j]\n                with h5py.File(path_j, 'r') as f2:\n                    feats_j = {\n                        'keypoints': f2['keypoints'][()],\n                        'descriptors': f2['descriptors'][()],\n                        'image_path': f2.attrs.get('image_path', str(path_j))\n                    }\n\n                matches = matcher.match(feats_i, feats_j)\n                all_match_counts.append(len(matches))\n\n                if len(matches) > 0:\n                    group_i.create_dataset(path_j.stem, data=matches)\n                    if len(matches) >= matcher.min_matches:\n                        match_graph.add_edge(i, j, weight=len(matches))\n\n                pbar.update(1)\n\n    if internal_bar:\n        pbar.close()\n\n    if VERBOSE:\n        print(\"\\nMatch statistics:\")\n        print(f\"  • Total pairs attempted: {len(all_match_counts)}\")\n        print(f\"  • Avg matches per pair: {np.mean(all_match_counts):.1f}\\n\")\n\n    return match_graph, all_match_counts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:05:22.053623Z","iopub.execute_input":"2025-06-02T19:05:22.054311Z","iopub.status.idle":"2025-06-02T19:05:22.062378Z","shell.execute_reply.started":"2025-06-02T19:05:22.054285Z","shell.execute_reply":"2025-06-02T19:05:22.061715Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Full Pipeline for Matching","metadata":{}},{"cell_type":"code","source":"# (Assume get_image_pairs_shortlist and get_device are already defined above.)\n# Also assume LightGlueMatcher and VERBOSE are in scope.\n\n# ─────────────────────────────────────────────────────────────────────────────\n# Paths & constants\n# ─────────────────────────────────────────────────────────────────────────────\n\nfeature_dir = Path(\"/kaggle/working/output_features\")\nsample_csv = Path(\"/kaggle/input/image-matching-challenge-2025/sample_submission.csv\")\n\n# Read sample_submission to map each image → dataset name\nsample_df = pd.read_csv(sample_csv)\nimage_to_dataset = dict(zip(sample_df['image'], sample_df['dataset']))\n\n# Discover all .h5 feature files\nall_feature_paths = list(feature_dir.rglob(\"*.h5\"))\ndataset_to_feature_paths = defaultdict(list)\n\nfor fp in all_feature_paths:\n    with h5py.File(fp, 'r') as f:\n        img_path = Path(f.attrs['image_path'])\n    dataset_name = image_to_dataset.get(img_path.name)\n    if dataset_name is not None:\n        dataset_to_feature_paths[dataset_name].append(fp)\n    elif VERBOSE:\n        print(f\"  • Could not assign dataset to '{img_path.name}', skipping.\")\n\nif VERBOSE:\n    print(f\"Found {len(dataset_to_feature_paths)} datasets in feature files.\\n\")\n\n# ─────────────────────────────────────────────────────────────────────────────\n# Step 3: Compute neighbors (using distance‐threshold + min_pairs) and prepare global lists\n# ─────────────────────────────────────────────────────────────────────────────\n\nall_paths = []\nall_indices = []\npair_counts = []\noffset = 0\n\nfor dataset_name, feature_paths in dataset_to_feature_paths.items():\n    if VERBOSE:\n        print(f\"Processing dataset: {dataset_name} ({len(feature_paths)} images)\")\n\n    N = len(feature_paths)\n\n    # If the dataset is very small, let “exhaustive” mean “use all pairs.”\n    exhaustive_if_less = 30\n\n    # Set a dataset‐size–dependent “minimum neighbors” (cap at 30, floor at 5).\n    min_pairs_for_dataset = min(30, max(8, int(N / 10)))\n\n    # Instead of “find_top_k_neighbors”, use distance‐threshold + min_pairs:\n    image_pairs = get_image_pairs_shortlist(\n        feature_paths,\n        sim_th=0.60,\n        min_pairs=min_pairs_for_dataset,\n        exhaustive_if_less=exhaustive_if_less,\n        device=get_device()\n    )\n\n    # Build a per-image neighbor list (local indices 0..N-1)\n    neighbors_per_image = [[] for _ in range(N)]\n    for (i_local, j_local) in image_pairs:\n        neighbors_per_image[i_local].append(j_local)\n        neighbors_per_image[j_local].append(i_local)\n\n    # Append these feature_paths to the global flat list\n    all_paths.extend(feature_paths)\n\n    # Shift local neighbor indices by 'offset' to get global indices\n    adjusted_top_k = [\n        [nbr + offset for nbr in neighbors_per_image[i_local]]\n        for i_local in range(N)\n    ]\n    all_indices.extend(adjusted_top_k)\n\n    # Count valid neighbor pairs (excluding any self‐match) for stats\n    for i_local, nbrs_global in enumerate(adjusted_top_k):\n        valid_count = len([j for j in nbrs_global if j != (i_local + offset)])\n        pair_counts.append(valid_count)\n\n    offset += N\n\n# Report totals\ntotal_pairs = sum(pair_counts)\nif VERBOSE:\n    print(f\"\\nTotal feature files: {len(all_paths)}\")\n    print(f\"Total match pairs to try: {total_pairs}\\n\")\n\n# ─────────────────────────────────────────────────────────────────────────────\n# Instantiate the LightGlue matcher (using the same min_matches)\n# ─────────────────────────────────────────────────────────────────────────────\n\nmatcher = LightGlueMatcher(device=get_device(), min_matches=20)\n\n# Create a progress bar for the entire matching process\npbar = tqdm(total=total_pairs, desc=\"Matching with LightGlue\", disable=not VERBOSE)\n\n# Compute matches across all datasets\nmatch_graph, match_counts = compute_lightglue_matches_for_clustering(\n    all_paths,\n    all_indices,\n    matcher=matcher,\n    pbar=pbar\n)\n\npbar.close()\nif VERBOSE:\n    print(\"All matching complete.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:05:33.520720Z","iopub.execute_input":"2025-06-02T19:05:33.521515Z","iopub.status.idle":"2025-06-02T19:11:32.298853Z","shell.execute_reply.started":"2025-06-02T19:05:33.521480Z","shell.execute_reply":"2025-06-02T19:11:32.298252Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Visualize Statistics","metadata":{}},{"cell_type":"code","source":"def plot_match_stats_per_dataset(match_counts, dataset_name, match_threshold=10):\n    if not VERBOSE:\n        return\n\n    print(f\"\\n Dataset: {dataset_name}\")\n    plt.figure(figsize=(15, 5))\n\n    # Full match count histogram\n    plt.subplot(1, 2, 1)\n    plt.hist(match_counts, bins=50, color='skyblue', edgecolor='black')\n    plt.axvline(match_threshold, color='red', linestyle='--', label=f'Threshold: {match_threshold}')\n    plt.title(f'{dataset_name}: Match Count Distribution')\n    plt.xlabel('Match Count')\n    plt.ylabel('Frequency')\n    plt.legend()\n\n    # Above-threshold matches\n    valid = [m for m in match_counts if m >= match_threshold]\n    plt.subplot(1, 2, 2)\n    plt.hist(valid, bins=30, color='lightgreen', edgecolor='darkgreen')\n    plt.title(f'{dataset_name}: Matches ≥ {match_threshold} (n={len(valid)})')\n    plt.xlabel('Match Count')\n    plt.ylabel('Frequency')\n\n    plt.tight_layout()\n    plt.show()\n\n    print(f\"Total pairs: {len(match_counts)}\")\n    print(f\"Above threshold: {len(valid)} ({len(valid)/len(match_counts):.1%})\")\n    print(f\"Min: {min(match_counts)} | Max: {max(match_counts)} | Median: {int(np.median(match_counts))} | Mean: {np.mean(match_counts):.1f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:15:13.833971Z","iopub.execute_input":"2025-06-02T19:15:13.834718Z","iopub.status.idle":"2025-06-02T19:15:13.841046Z","shell.execute_reply.started":"2025-06-02T19:15:13.834691Z","shell.execute_reply":"2025-06-02T19:15:13.840302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # STEP 2: Visualize stats & get exact median\n# median_threshold = int(np.median(match_counts))\n# plot_match_stats(match_counts, match_threshold=median_threshold)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:15:14.024583Z","iopub.execute_input":"2025-06-02T19:15:14.024793Z","iopub.status.idle":"2025-06-02T19:15:14.028237Z","shell.execute_reply.started":"2025-06-02T19:15:14.024776Z","shell.execute_reply":"2025-06-02T19:15:14.027484Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Clear Memory","metadata":{}},{"cell_type":"code","source":"import gc\n\ndel all_feature_paths\n\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:15:14.481080Z","iopub.execute_input":"2025-06-02T19:15:14.481611Z","iopub.status.idle":"2025-06-02T19:15:14.867215Z","shell.execute_reply.started":"2025-06-02T19:15:14.481587Z","shell.execute_reply":"2025-06-02T19:15:14.866474Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Build Match Graph","metadata":{}},{"cell_type":"code","source":"def build_graphs_from_all_matches(match_file, feature_root):\n    match_file = Path(match_file)\n    feature_root = Path(feature_root)\n\n    sample_df = pd.read_csv('/kaggle/input/image-matching-challenge-2025/sample_submission.csv')\n    image_to_dataset = dict(zip(sample_df['image'], sample_df['dataset']))\n\n    with h5py.File(match_file, 'r') as f_matches:\n        graphs = {}\n        match_stats = {}\n\n        # Group feature paths by dataset using image_path attr\n        dataset_paths = defaultdict(list)\n        for path in sorted(feature_root.rglob(\"*.h5\")):\n            with h5py.File(path, 'r') as f:\n                image_name = Path(f.attrs['image_path']).name\n                dataset = image_to_dataset.get(image_name, None)\n                if dataset is not None:\n                    dataset_paths[dataset].append(path)\n                elif VERBOSE:\n                    print(f\" No dataset found for {image_name}, skipping.\")\n\n        for dataset, paths in dataset_paths.items():\n            G = nx.Graph()\n            match_counts = []\n            name_to_index = {Path(p).stem: i for i, p in enumerate(paths)}\n            index_to_name = {v: k for k, v in name_to_index.items()}\n\n            N = len(paths)\n            # Compute a dataset-specific match threshold (floor=5, ceil=30, scale=N/10)\n            threshold = min(30, max(10, int(N / 10)))\n\n            for i, path_i in enumerate(paths):\n                name_i = Path(path_i).stem\n                if name_i not in f_matches:\n                    continue\n                group_i = f_matches[name_i]\n                G.add_node(i, path=str(path_i))\n\n                for j, path_j in enumerate(paths):\n                    if i >= j:\n                        continue\n                    name_j = Path(path_j).stem\n                    if name_j in group_i:\n                        matches = group_i[name_j][()]\n                        match_counts.append(len(matches))\n                        if len(matches) >= threshold:\n                            G.add_edge(i, j, weight=len(matches))\n\n            if match_counts:\n                plot_match_stats_per_dataset(match_counts, dataset, match_threshold=threshold)\n                graphs[dataset] = G\n                match_stats[dataset] = {\n                    'counts': match_counts,\n                    'threshold': threshold,\n                    'num_nodes': G.number_of_nodes(),\n                    'num_edges': G.number_of_edges()\n                }\n            elif VERBOSE:\n                print(f\" No matches for dataset: {dataset}\")\n\n    return graphs, match_stats","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:15:14.968760Z","iopub.execute_input":"2025-06-02T19:15:14.969181Z","iopub.status.idle":"2025-06-02T19:15:14.977783Z","shell.execute_reply.started":"2025-06-02T19:15:14.969163Z","shell.execute_reply":"2025-06-02T19:15:14.977163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === RUN IT ===\ngraphs_by_dataset, stats_by_dataset = build_graphs_from_all_matches(\n    match_file=\"/kaggle/working/output_lightglue_matches/all_matches.h5\",\n    feature_root=\"/kaggle/working/output_features\"\n)\n\n# Quick summary if verbose\nif VERBOSE:\n    print(\"\\n Final Graph Summary:\")\n    for dataset, G in graphs_by_dataset.items():\n        info = stats_by_dataset[dataset]\n        print(f\" {dataset}: threshold={info['threshold']}, {info['num_nodes']} nodes, {info['num_edges']} edges\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:15:17.409639Z","iopub.execute_input":"2025-06-02T19:15:17.409910Z","iopub.status.idle":"2025-06-02T19:15:18.407051Z","shell.execute_reply.started":"2025-06-02T19:15:17.409890Z","shell.execute_reply":"2025-06-02T19:15:18.406459Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Visualize the Graphs","metadata":{}},{"cell_type":"code","source":"def visualize_graph_kaggle_fixed(graph, dataset_name=None, min_weight=1, figsize=(12, 12)):\n    if not VERBOSE:\n        return\n\n    import matplotlib.pyplot as plt\n    import networkx as nx\n    import numpy as np\n\n    plt.figure(figsize=figsize)\n\n    all_weights = [d['weight'] for _, _, d in graph.edges(data=True)]\n    if not all_weights:\n        print(\" No edges in graph!\")\n        return\n\n    edges = [(u, v, d) for u, v, d in graph.edges(data=True) if d['weight'] >= min_weight]\n    if not edges:\n        print(f\" No edges meet weight threshold (≥{min_weight})\")\n        return\n\n    subG = nx.Graph()\n    subG.add_edges_from(edges)\n    weights = [d['weight'] for _, _, d in edges]\n\n    pos = nx.circular_layout(subG) if len(subG.nodes) < 10 else nx.spring_layout(subG, k=0.5)\n\n    nx.draw_networkx_edges(\n        subG, pos,\n        width=[0.5 + 4*(w - min(weights)) / (max(weights) - min(weights) + 1e-6) for w in weights],\n        edge_color='red', alpha=0.8\n    )\n    nx.draw_networkx_nodes(subG, pos, node_size=100, node_color='blue', alpha=1.0)\n\n    sm = plt.cm.ScalarMappable(cmap=plt.cm.Reds, norm=plt.Normalize(vmin=min(weights), vmax=max(weights)))\n    sm.set_array([])\n    cbar = plt.colorbar(sm, ax=plt.gca(), shrink=0.8)\n    cbar.set_label('Match Strength')\n\n    plt.title(f\"{dataset_name or 'Graph'}: {len(subG.nodes)} nodes | {len(edges)} edges (≥ {min_weight})\")\n    plt.axis('off')\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:15:18.408167Z","iopub.execute_input":"2025-06-02T19:15:18.408391Z","iopub.status.idle":"2025-06-02T19:15:18.416627Z","shell.execute_reply.started":"2025-06-02T19:15:18.408375Z","shell.execute_reply":"2025-06-02T19:15:18.415983Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize graphs if VERBOSE\nif VERBOSE:\n    for dataset, graph in graphs_by_dataset.items():\n        print(f\"\\n Dataset: {dataset}\")\n        print(f\" Match graph: {len(graph.nodes)} nodes, {len(graph.edges)} edges\")\n\n        print(\"\\n Sample nodes with neighbors:\")\n        for i in list(graph.nodes)[:5]:\n            neighbors = list(graph[i].keys())\n            weights = [graph[i][j]['weight'] for j in neighbors]\n            print(f\"  Node {i}: {len(neighbors)} neighbors → {neighbors[:3]}\")\n            if weights:\n                print(f\"    Sample weights: {weights[:3]}\")\n\n        print(\"Graph info:\")\n        print(f\"Nodes: {len(graph.nodes)}\")\n        print(f\"Edges: {len(graph.edges)}\")\n        if graph.edges:\n            edge_weights = [d['weight'] for _, _, d in graph.edges(data=True)]\n            sample_weights = edge_weights[:3]\n            print(f\"Sample edge weights: {sample_weights}\")\n            median_weight = int(np.median(edge_weights))\n            visualize_graph_kaggle_fixed(graph, dataset_name=dataset, min_weight=max(1, median_weight // 2))\n        else:\n            print(\"No edges to visualize!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:15:18.417267Z","iopub.execute_input":"2025-06-02T19:15:18.417508Z","iopub.status.idle":"2025-06-02T19:15:18.973300Z","shell.execute_reply.started":"2025-06-02T19:15:18.417485Z","shell.execute_reply":"2025-06-02T19:15:18.972477Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Cluster the Graphs","metadata":{}},{"cell_type":"code","source":"def enhanced_cluster_analysis(\n    cluster_results,\n    feature_paths,\n    original_image_root,\n    show_image_names=True,\n    show_images=True\n):\n    if not VERBOSE:\n        return\n\n    scenes = defaultdict(list)\n    for node, data in cluster_results.items():\n        scenes[data['label']].append((node, data['stability']))\n\n    sorted_scenes = sorted(scenes.items(), key=lambda x: (x[0] == -1, -len(x[1])))\n\n    if VERBOSE:\n        print(f\"\\n{' Cluster Statistics ':=^60}\")\n        print(f\"Total images: {len(cluster_results)}\")\n        num_scenes = len([x for x in sorted_scenes if x[0] != -1])\n        print(f\"Total scenes: {num_scenes}\")\n        num_outliers = len(scenes.get(-1, []))\n        print(f\"Outliers: {num_outliers}\\n\")\n\n        print(f\"\\n{' Scene Details ':=^60}\")\n    for label, members in sorted_scenes:\n        if label == -1:\n            if VERBOSE:\n                print(f\"\\n{' Outliers ':-^40}\")\n        else:\n            if VERBOSE:\n                print(f\"\\n{' Scene %d ' % label:-^40}\")\n                avg_stability = sum(s for _, s in members) / len(members)\n                print(f\"Images: {len(members)} | Avg Stability: {avg_stability:.2f}\")\n\n        if show_images:\n            fig, axes = plt.subplots(1, min(5, len(members)), figsize=(20, 4))\n            if len(members) == 1:\n                axes = [axes]\n            for ax, (node, stability) in zip(axes, members[:5]):\n                try:\n                    with h5py.File(feature_paths[node], 'r') as f:\n                        img_name = f.attrs['image_path']\n                    img_path = original_image_root / Path(img_name).name\n                    img = cv2.imread(str(img_path))\n                    if img is not None:\n                        ax.imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\n                        title = f\"Stab: {stability:.2f}\" if label != -1 else \"Outlier\"\n                        ax.set_title(title, fontsize=8)\n                    else:\n                        ax.text(0.5, 0.5, f\"Missing\\n{Path(img_name).name}\", ha='center')\n                    ax.axis('off')\n                except:\n                    ax.text(0.5, 0.5, f\"Error loading\", ha='center')\n                    ax.axis('off')\n            plt.tight_layout()\n            plt.show()\n\n    if show_image_names and VERBOSE:\n        print(f\"\\n{' Image Names per Scene ':=^60}\")\n        for label, members in sorted_scenes:\n            image_names = []\n            for node, _ in members:\n                with h5py.File(feature_paths[node], 'r') as f:\n                    img_name = f.attrs['image_path']\n                image_names.append(Path(img_name).name)\n            if label == -1:\n                print(f\"\\n{' Outliers ':-^40}\")\n            else:\n                print(f\"\\n{' Scene %d ' % label:-^40}\")\n            print(f\"Images: {', '.join(image_names)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:15:18.974725Z","iopub.execute_input":"2025-06-02T19:15:18.975310Z","iopub.status.idle":"2025-06-02T19:15:18.986754Z","shell.execute_reply.started":"2025-06-02T19:15:18.975286Z","shell.execute_reply":"2025-06-02T19:15:18.986023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─────────────────────────────────────────────────────────────────────────────\n# K-means splitter + full clustering pipeline (Option-1 strategy)\n# ─────────────────────────────────────────────────────────────────────────────\nimport numpy as np\nimport networkx as nx\nfrom sklearn.cluster import KMeans\nfrom collections import defaultdict\nimport h5py\nfrom pathlib import Path\n\n# PARAMETERS ---------------------------------------------------------------\nmin_size    = 5     # min images for a component to count as a “scene”\n# --------------------------------------------------------------------------\n\ndef split_graph_kmeans(G, min_size=3):\n    \"\"\"\n    Returns:\n        clusters : List[List[int]]     – each inner list is a component ≥ min_size\n        outliers : List[int]           – node indices belonging to tiny comps (<min_size)\n        cutoff   : int                 – weight threshold used\n    \"\"\"\n    # --- collect edge weights ---\n    weights = np.array([d['weight'] for _,_,d in G.edges(data=True)])\n    if len(weights) == 0:\n        return [], list(G.nodes()), 0   # everything is outlier\n\n    # --- 2-cluster K-means on log weights (weak vs strong) ---\n    km = KMeans(n_clusters=2, n_init=20, random_state=42).fit(np.log10(weights).reshape(-1,1))\n    weak_cluster = int(np.argmin(km.cluster_centers_.flatten()))\n    cutoff = weights[km.labels_ == weak_cluster].max()\n\n    # --- prune graph to strong edges only ---\n    pruned = nx.Graph()\n    pruned.add_nodes_from(G.nodes(data=True))\n    for (u,v,d), lbl in zip(G.edges(data=True), km.labels_):\n        if lbl != weak_cluster:                 # keep only strong edges\n            pruned.add_edge(u,v,weight=d['weight'])\n\n    # --- components ---\n    comps      = list(nx.connected_components(pruned))\n    clusters   = [sorted(list(c)) for c in comps if len(c) >= min_size]\n    tiny_comps = [c for c in comps if len(c) <  min_size]\n    outliers   = sorted([node for c in tiny_comps for node in c])\n    return clusters, outliers, cutoff","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:15:27.383306Z","iopub.execute_input":"2025-06-02T19:15:27.383909Z","iopub.status.idle":"2025-06-02T19:15:27.391383Z","shell.execute_reply.started":"2025-06-02T19:15:27.383888Z","shell.execute_reply":"2025-06-02T19:15:27.390715Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# TOGGLE OPTIONS\nSHOW_IMAGES    = False\nSHOW_FILENAMES = True\n\n# Step-0: feature-paths per dataset (unchanged)\nfeature_root = Path(\"/kaggle/working/output_features\")\ndataset_feature_paths = {\n    d.name: sorted(list(d.rglob(\"*.h5\"))) for d in feature_root.iterdir() if d.is_dir()\n}\n\nall_cluster_results      = {}          # dataset → {node:{label,stab}}\ncluster_results_original = {}\ncluster_results          = {}          # global cluster-id → list[global_node_ids]\ncluster_to_dataset       = {}\n\nfeature_paths = []\nglobal_offset     = 0\nglobal_cluster_id = 0\n\nfor dataset, G in graphs_by_dataset.items():\n    if VERBOSE:\n        print(f\"\\n=== Clustering for dataset: {dataset} ===\")\n\n    N = G.number_of_nodes()\n    if N == 0:\n        all_cluster_results[dataset] = {}\n        continue\n\n    # 1) K-MEANS SPLIT\n    clusters, outliers, cutoff = split_graph_kmeans(G, min_size=min_size)\n\n    if len(clusters) >= 2:\n        decision = \"MULTI-OBJECT\"\n    else:\n        decision = \"SINGLE-SCENE\"\n        # If single-scene, force exactly one cluster of all non-outlier nodes\n        if len(clusters) == 0:\n            clusters = [sorted(list(set(range(N)) - set(outliers)))]\n\n    if VERBOSE:\n        print(f\"  cutoff weight = {cutoff}\")\n        print(f\"  clusters ≥{min_size}: {len(clusters)}   •  outliers: {len(outliers)}\")\n        print(f\"  DECISION → {decision}\")\n\n    # 2) Build per-node label dict for this dataset\n    label_dict = {node:{'label':-1,'stability':0.0} for node in range(N)}\n    for local_lbl, comp in enumerate(clusters):\n        for node in comp:\n            label_dict[node] = {'label': local_lbl, 'stability':1.0}\n    all_cluster_results[dataset] = label_dict\n\n    # 3) Map local→global and populate global structures\n    paths = dataset_feature_paths[dataset]\n    feature_paths.extend(paths)\n    local_to_global = {i: global_offset+i for i in range(N)}\n\n    for local_lbl, comp in enumerate(clusters):\n        global_nodes = [local_to_global[n] for n in comp]\n        cluster_results[global_cluster_id] = global_nodes\n        cluster_to_dataset[global_cluster_id] = dataset\n\n        for g in global_nodes:\n            cluster_results_original[g] = {'label':global_cluster_id,'stability':1.0}\n        global_cluster_id += 1\n\n    for node in outliers:\n        g = local_to_global[node]\n        cluster_results_original[g] = {'label':-1,'stability':0.0}\n\n    global_offset += N\n\n# --------------------------------------------------------------------------\n# Visualise / print scene summaries (uses your existing helper)\nfor dataset, result in all_cluster_results.items():\n    if VERBOSE:\n        print(f\"\\n--- Scene Analysis: {dataset} ---\")\n    paths = dataset_feature_paths[dataset]\n    enhanced_cluster_analysis(\n        result,\n        paths,\n        original_image_root=Path(\"/kaggle/input/flat-522-imc25-images/Dataset-Modified_Flat/Dataset-Modified_Flat/test\"),\n        show_image_names=SHOW_FILENAMES,\n        show_images=SHOW_IMAGES\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:17:12.572640Z","iopub.execute_input":"2025-06-02T19:17:12.573245Z","iopub.status.idle":"2025-06-02T19:17:12.636535Z","shell.execute_reply.started":"2025-06-02T19:17:12.573220Z","shell.execute_reply":"2025-06-02T19:17:12.635806Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Split all_matches.h5 into cluster_*.h5 based on clustering results","metadata":{}},{"cell_type":"code","source":"# from path import pathlib\n# import gc\n\n# shutil.rmtree(\"/kaggle/working/pose_outputs\")\n# match_root = pathlib.Path(\"/kaggle/working/final_matches_subclusters\")\n# shutil.rmtree(match_root, ignore_errors=True)\n# gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:15:58.587062Z","iopub.execute_input":"2025-06-02T19:15:58.587332Z","iopub.status.idle":"2025-06-02T19:15:58.591008Z","shell.execute_reply.started":"2025-06-02T19:15:58.587311Z","shell.execute_reply":"2025-06-02T19:15:58.590173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def save_clusters_for_colmap(\n    all_match_file: str,\n    all_cluster_results: dict,\n    dataset_feature_paths: dict,\n    output_dir: str\n):\n    \"\"\"\n    For each dataset:\n      • Determine whether it’s multi-object (≥2 clusters of size ≥ 1) or single-scene (exactly 1 cluster).\n      • If multi-object: write each non-outlier cluster to its own HDF5; drop all outliers permanently.\n      • If single-scene: let N = total images. If (#outliers) > 5% of N, merge outliers back into the single cluster before writing.\n                         Otherwise, drop those outliers. Always write exactly one HDF5 for the (possibly merged) main cluster.\n\n    Returns\n    -------\n    cluster_to_dataset : dict\n        Maps each produced cluster_{id}.h5 → dataset name.\n    \"\"\"\n    output_dir = Path(output_dir)\n    output_dir.mkdir(parents=True, exist_ok=True)\n\n    cluster_to_dataset = {}\n    cluster_counter    = 0\n\n    with h5py.File(all_match_file, \"r\") as f_in:\n        for dataset, cluster_result in all_cluster_results.items():\n            if VERBOSE:\n                print(f\"\\nProcessing dataset: {dataset}\")\n\n            feature_paths = dataset_feature_paths[dataset]\n            N = len(feature_paths)\n            name_to_index = {Path(p).stem: i for i, p in enumerate(feature_paths)}\n            index_to_name = {i: Path(p).stem for i, p in enumerate(feature_paths)}\n\n            # Build local clusters vs outliers\n            clusters_local = defaultdict(list)  # label → list of local ids\n            outliers_local = []\n            for local_id, info in cluster_result.items():\n                lbl = info.get(\"label\", -1)\n                if lbl == -1:\n                    outliers_local.append(local_id)\n                else:\n                    clusters_local[lbl].append(local_id)\n\n            # Count real clusters (labels ≥ 0)\n            num_real_clusters = len([lbl for lbl in clusters_local.keys() if clusters_local[lbl]])\n\n            # --- if multi-object: write each cluster; drop all outliers ---\n            if num_real_clusters >= 2:\n                if VERBOSE:\n                    print(\"  Detected MULTI-OBJECT → writing each cluster, dropping outliers\")\n\n                for _, local_indices in sorted(clusters_local.items()):\n                    out_path = output_dir / f\"cluster_{cluster_counter}.h5\"\n                    _write_subh5(out_path, local_indices, index_to_name, f_in)\n                    if VERBOSE:\n                        print(f\"   • cluster_{cluster_counter}.h5   (size {len(local_indices)})\")\n                    cluster_to_dataset[cluster_counter] = dataset\n                    cluster_counter += 1\n\n            # --- else single-scene: decide whether to merge outliers ---\n            else:\n                if VERBOSE:\n                    print(\"  Detected SINGLE-SCENE\")\n\n                # Single main cluster (take label = next existing key if any, else all nodes)\n                if clusters_local:\n                    main_local = next(iter(clusters_local.values()))\n                else:\n                    main_local = []\n\n                num_out = len(outliers_local)\n                # If outliers > 20% of N, merge them back\n                if num_out > 0.2 * N:\n                    if VERBOSE:\n                        print(f\"   • Outliers ({num_out}) > 5% of {N} → merging back into main cluster\")\n                    merged = sorted(set(main_local + outliers_local))\n                    out_path = output_dir / f\"cluster_{cluster_counter}.h5\"\n                    _write_subh5(out_path, merged, index_to_name, f_in)\n                    if VERBOSE:\n                        print(f\"   • cluster_{cluster_counter}.h5   (merged size {len(merged)})\")\n                    cluster_to_dataset[cluster_counter] = dataset\n                    cluster_counter += 1\n                else:\n                    if VERBOSE:\n                        print(f\"   • Outliers ({num_out}) ≤ 5% of {N} → dropping them\")\n                    # write only the main cluster (if empty, skip)\n                    if main_local:\n                        out_path = output_dir / f\"cluster_{cluster_counter}.h5\"\n                        _write_subh5(out_path, main_local, index_to_name, f_in)\n                        if VERBOSE:\n                            print(f\"   • cluster_{cluster_counter}.h5   (size {len(main_local)})\")\n                        cluster_to_dataset[cluster_counter] = dataset\n                        cluster_counter += 1\n\n    return cluster_to_dataset\n\n\ndef _write_subh5(out_path, local_indices, index_to_name, h5_matches_in):\n    \"\"\"\n    Helper: create cluster_{id}.h5 that contains only the image–pairs\n    whose two stems BOTH appear in `local_indices`.\n    \"\"\"\n    stems_in_cluster = {index_to_name[i] for i in local_indices}\n    with h5py.File(out_path, \"w\") as f_out:\n        for stem0 in stems_in_cluster:\n            if stem0 not in h5_matches_in:\n                continue\n            g_in = h5_matches_in[stem0]\n            g_out = f_out.create_group(stem0)\n            for stem1 in g_in:\n                if stem1 in stems_in_cluster:\n                    matches = g_in[stem1][()]\n                    g_out.create_dataset(stem1, data=matches)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:15:59.013953Z","iopub.execute_input":"2025-06-02T19:15:59.014173Z","iopub.status.idle":"2025-06-02T19:15:59.027301Z","shell.execute_reply.started":"2025-06-02T19:15:59.014156Z","shell.execute_reply":"2025-06-02T19:15:59.026660Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Run it (replacing the old save_clusters_and_outliers_for_colmap) ──\ncluster_to_dataset = save_clusters_for_colmap(\n    all_match_file=\"/kaggle/working/output_lightglue_matches/all_matches.h5\",\n    all_cluster_results=all_cluster_results,\n    dataset_feature_paths=dataset_feature_paths,\n    output_dir=\"/kaggle/working/final_matches_subclusters\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:15:59.256116Z","iopub.execute_input":"2025-06-02T19:15:59.256323Z","iopub.status.idle":"2025-06-02T19:15:59.342482Z","shell.execute_reply.started":"2025-06-02T19:15:59.256307Z","shell.execute_reply":"2025-06-02T19:15:59.341907Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Split Matches","metadata":{}},{"cell_type":"code","source":"import gc\nimport os\nimport h5py\nfrom pathlib import Path\n\ndef split_cluster_h5_for_colmap(cluster_h5_path, feature_root=\"/kaggle/working/output_features\"):\n    \"\"\"\n    Given cluster_{id}.h5 (keys are image stems), create:\n      • keypoints.h5  (original keypoints for each full‐filename image)\n      • matches.h5    (pairwise matches between those images)\n    Saves both under cluster_{id}/ folder.\n    \"\"\"\n    cluster_id  = cluster_h5_path.stem\n    output_dir  = cluster_h5_path.parent / cluster_id\n    output_dir.mkdir(exist_ok=True)\n\n    keypoints_out = output_dir / \"keypoints.h5\"\n    matches_out   = output_dir / \"matches.h5\"\n\n    # Remove existing files if any\n    for f in (keypoints_out, matches_out):\n        if f.exists():\n            f.unlink()\n\n    # Map stems to feature‐file paths, and cache extensions\n    feature_dir     = Path(feature_root)\n    stem_to_feature = {f.stem: f for f in feature_dir.rglob(\"*.h5\")}\n    ext_cache       = {}  # stem ➜ \".jpg\", \".png\", etc.\n\n    def stem_to_fullname(stem: str) -> str:\n        \"\"\"Return '<stem><ext>' with the real extension on disk.\"\"\"\n        if stem in ext_cache:\n            return stem + ext_cache[stem]\n\n        if stem not in stem_to_feature:\n            raise FileNotFoundError(f\"No feature file for image stem '{stem}'\")\n\n        with h5py.File(stem_to_feature[stem], \"r\") as f_feat:\n            ext = Path(f_feat.attrs[\"image_path\"]).suffix.lower()\n        ext_cache[stem] = ext\n        return stem + ext\n\n    with h5py.File(cluster_h5_path, \"r\") as f_in, \\\n         h5py.File(keypoints_out,   \"w\") as f_kp, \\\n         h5py.File(matches_out,     \"w\") as f_mt:\n\n        seen_kpts = set()\n\n        for img0_stem in f_in:\n            g_in  = f_in[img0_stem]\n            img0  = stem_to_fullname(img0_stem)\n            g_out = f_mt.require_group(img0)\n\n            for img1_stem in g_in:\n                img1    = stem_to_fullname(img1_stem)\n                matches = g_in[img1_stem][()]\n\n                # Write keypoints for each new image\n                for stem in (img0_stem, img1_stem):\n                    full = stem_to_fullname(stem)\n                    if full not in seen_kpts:\n                        with h5py.File(stem_to_feature[stem], \"r\") as f_feat:\n                            kpts = f_feat[\"keypoints\"][()]\n                        grp = f_kp.require_group(full)\n                        grp.create_dataset(\"keypoints\", data=kpts)\n                        seen_kpts.add(full)\n\n                # Write matches\n                g_out.create_dataset(img1, data=matches)\n\n    if VERBOSE:\n        print(f\"[{cluster_id}] keypoints & matches written to {output_dir}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:16:04.220035Z","iopub.execute_input":"2025-06-02T19:16:04.220297Z","iopub.status.idle":"2025-06-02T19:16:04.229188Z","shell.execute_reply.started":"2025-06-02T19:16:04.220278Z","shell.execute_reply":"2025-06-02T19:16:04.228570Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Run this for every cluster_*.h5 that save_clusters_and_outliers_for_colmap created ──\nsubcluster_root = Path(\"/kaggle/working/final_matches_subclusters\")\nfor cluster_file in subcluster_root.rglob(\"cluster_*.h5\"):\n    split_cluster_h5_for_colmap(cluster_file)\n\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:16:04.458302Z","iopub.execute_input":"2025-06-02T19:16:04.458794Z","iopub.status.idle":"2025-06-02T19:16:05.004934Z","shell.execute_reply.started":"2025-06-02T19:16:04.458772Z","shell.execute_reply":"2025-06-02T19:16:05.004384Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Delete Feature Files","metadata":{}},{"cell_type":"code","source":"# ------------------------------------------------------ build once, keep in RAM\nimport h5py, pandas as pd\nfrom pathlib import Path\n\n# we already have this from earlier\nimage_to_dataset = dict(zip(\n    pd.read_csv(\"/kaggle/input/image-matching-challenge-2025/sample_submission.csv\")[\"image\"],\n    pd.read_csv(\"/kaggle/input/image-matching-challenge-2025/sample_submission.csv\")[\"dataset\"]\n))\n\nimage_to_featidx = {}                 # key: (dataset, image_name) -> global idx\n\nfor idx, h5_path in enumerate(feature_paths):        # 'feature_paths' already exists\n    with h5py.File(h5_path, \"r\") as f:\n        img_name = Path(f.attrs[\"image_path\"]).name   # real filename\n    dataset = image_to_dataset[img_name]              # always defined\n    image_to_featidx[(dataset, img_name)] = idx","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T19:16:06.355358Z","iopub.execute_input":"2025-06-02T19:16:06.355647Z","iopub.status.idle":"2025-06-02T19:16:06.403043Z","shell.execute_reply.started":"2025-06-02T19:16:06.355627Z","shell.execute_reply":"2025-06-02T19:16:06.402553Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\nimport gc\n\ntry:\n    shutil.rmtree(\"/kaggle/working/output_features\")\n    print(\" Deleted output_features to save space.\")\nexcept Exception as e:\n    print(f\" Failed to delete output_features: {e}\")\n    \ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T18:09:40.442389Z","iopub.execute_input":"2025-06-01T18:09:40.442680Z","iopub.status.idle":"2025-06-01T18:09:40.853318Z","shell.execute_reply.started":"2025-06-01T18:09:40.442631Z","shell.execute_reply":"2025-06-01T18:09:40.852766Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Patch functions","metadata":{}},{"cell_type":"code","source":"# Add competition utils\nsys.path.append('/kaggle/input/imc25-utils')\n\nimport pycolmap\nfrom pathlib import Path\nimport os\nimport gc\n\n\nimport os\nimport h5py\nimport cv2\nfrom collections import defaultdict\n\nfrom database import COLMAPDatabase  # from /kaggle/input/imc25-utils","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T11:32:03.630842Z","iopub.execute_input":"2025-06-02T11:32:03.631350Z","iopub.status.idle":"2025-06-02T11:32:03.635180Z","shell.execute_reply.started":"2025-06-02T11:32:03.631328Z","shell.execute_reply":"2025-06-02T11:32:03.634453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from h5_to_db import add_matches, import_into_colmap\n\nVALID_IMG_SUFFIXES = {'.png', '.jpg', '.jpeg', '.bmp', '.tif', '.tiff', '.webp'}\n\ndef _has_valid_suffix(name: str) -> bool:\n    return Path(name).suffix.lower() in VALID_IMG_SUFFIXES\n\n\ndef add_keypoints_patched(db: COLMAPDatabase,\n                          h5_dir: str,\n                          image_dir: str,\n                          img_ext: str,\n                          camera_model: int = 2,      # SIMPLE_RADIAL\n                          single_camera: bool = False):\n    \"\"\"\n    Import keypoints from `<h5_dir>/keypoints.h5` into *db*.\n\n    If the key stored in the HDF5 already has an extension (.jpg/.png/…),\n    that exact name is used. Otherwise we append *img_ext* for backward\n    compatibility.\n    \"\"\"\n    key_f = h5py.File(os.path.join(h5_dir, \"keypoints.h5\"), \"r\")\n    camera_id = None\n    fname_to_id = {}\n\n    for raw_name in tqdm(list(key_f.keys()),\n                         desc=\"Adding keypoints\", disable=not VERBOSE):\n\n        # ---------------------------------------------------------------- name\n        if _has_valid_suffix(raw_name):\n            fname = raw_name\n        else:\n            fname = Path(raw_name).stem + img_ext          # legacy behaviour\n\n        img_path = os.path.join(image_dir, fname)\n        if not os.path.isfile(img_path):\n            raise IOError(f\"Image not found on disk: {img_path}\")\n\n        img = cv2.imread(img_path)\n        if img is None:\n            raise IOError(f\"Failed to read image: {img_path}\")\n        H, W = img.shape[:2]\n\n        # ------------------------------------------------------------- camera\n        fov_deg = 45\n        fx = fy = W / (2 * np.tan(np.radians(fov_deg) / 2))\n        cx, cy = W/2, H/2\n        camera_id = db.add_camera(model=camera_model,\n                                  width=W, height=H,\n                                  params=[fx, cx, cy])\n\n        # -------------------------------------------------------------- save\n        image_id = db.add_image(fname, camera_id)\n        keypoints = key_f[raw_name][\"keypoints\"][()]\n        db.add_keypoints(image_id, keypoints)\n        fname_to_id[fname] = image_id\n\n    return fname_to_id","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T11:32:03.822772Z","iopub.execute_input":"2025-06-02T11:32:03.823019Z","iopub.status.idle":"2025-06-02T11:32:03.831867Z","shell.execute_reply.started":"2025-06-02T11:32:03.822999Z","shell.execute_reply":"2025-06-02T11:32:03.831079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from database import image_ids_to_pair_id\n\ndef add_matches_patched(db: COLMAPDatabase, h5_dir: str, fname_to_id: dict):\n    match_f = h5py.File(os.path.join(h5_dir, \"matches.h5\"), \"r\")\n    seen_pairs = set()\n\n    for name0 in match_f:\n        for name1 in match_f[name0]:\n            if name0 not in fname_to_id or name1 not in fname_to_id:\n                if VERBOSE:\n                    print(f\"Skipping unmatched names: {name0}, {name1}\")\n                continue\n\n            matches = match_f[name0][name1][()]\n            if matches.shape[1] != 2:\n                if VERBOSE:\n                    print(f\"Invalid match shape {matches.shape} ({name0}-{name1})\")\n                continue\n\n            id0, id1 = fname_to_id[name0], fname_to_id[name1]\n            pair_id  = image_ids_to_pair_id(id0, id1)\n            if pair_id in seen_pairs:\n                continue\n\n            seen_pairs.add(pair_id)\n            db.add_matches(id0, id1, matches)\n            db.add_two_view_geometry(id0, id1, matches)   # let COLMAP filter","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T11:32:06.674747Z","iopub.execute_input":"2025-06-02T11:32:06.675032Z","iopub.status.idle":"2025-06-02T11:32:06.681938Z","shell.execute_reply.started":"2025-06-02T11:32:06.675012Z","shell.execute_reply":"2025-06-02T11:32:06.680864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Patch the loader functions\nimport h5_to_db\nh5_to_db.add_keypoints = add_keypoints_patched\nh5_to_db.add_matches = add_matches_patched","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T11:32:06.881293Z","iopub.execute_input":"2025-06-02T11:32:06.881902Z","iopub.status.idle":"2025-06-02T11:32:06.885633Z","shell.execute_reply.started":"2025-06-02T11:32:06.881878Z","shell.execute_reply":"2025-06-02T11:32:06.884858Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# AAA","metadata":{}},{"cell_type":"code","source":"# ── Cell: Compute how many pre-COLMAP clusters each dataset has ──\n\nfrom collections import defaultdict\n\n# For every dataset, collect the set of distinct K-means labels in all_cluster_results\ndataset_precolmap_labels = defaultdict(set)\n\n# all_cluster_results: {dataset: {local_node_idx: {\"label\": <local_label> or –1, …}}, …}\n\nfor dataset, label_dict in all_cluster_results.items():\n    for local_idx, info in label_dict.items():\n        lbl = info.get(\"label\", -1)\n        if lbl != -1:\n            dataset_precolmap_labels[dataset].add(lbl)\n\n# Count how many clusters per dataset\ndataset_precolmap_count = {ds: len(sorted(lbls)) for ds, lbls in dataset_precolmap_labels.items()}\n\nif VERBOSE:\n    print(\"Pre-COLMAP clusters per dataset:\", dataset_precolmap_count)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T11:32:07.335335Z","iopub.execute_input":"2025-06-02T11:32:07.335965Z","iopub.status.idle":"2025-06-02T11:32:07.342641Z","shell.execute_reply.started":"2025-06-02T11:32:07.335936Z","shell.execute_reply":"2025-06-02T11:32:07.342015Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# COLMAP","metadata":{}},{"cell_type":"code","source":"import os\nimport sys\nimport ctypes\nimport contextlib\n\n@contextlib.contextmanager\ndef suppress_cpp_output():\n    devnull_fd      = os.open(os.devnull, os.O_WRONLY)\n    old_stdout_fd   = os.dup(1)\n    old_stderr_fd   = os.dup(2)\n    try:\n        os.dup2(devnull_fd, 1)\n        os.dup2(devnull_fd, 2)\n        yield\n    finally:\n        os.dup2(old_stdout_fd, 1)\n        os.dup2(old_stderr_fd, 2)\n        os.close(devnull_fd)\n        os.close(old_stdout_fd)\n        os.close(old_stderr_fd)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T11:32:07.905420Z","iopub.execute_input":"2025-06-02T11:32:07.906140Z","iopub.status.idle":"2025-06-02T11:32:07.911266Z","shell.execute_reply.started":"2025-06-02T11:32:07.906105Z","shell.execute_reply":"2025-06-02T11:32:07.910379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reconstruct_colmap_per_cluster_efficient(\n    cluster_results,\n    cluster_to_dataset,\n    dataset_feature_paths,\n    match_root,\n    image_root,\n    dataset_precolmap_count,\n    output_pose_dir=Path(\"/kaggle/working/pose_outputs\")\n):\n    \"\"\"\n    Incrementally run COLMAP for each cluster and save recovered poses.\n    Uses 'dataset_precolmap_count' to offset new labels for single-scene clusters.\n    Returns colmap_labels: {\"dataset|image\": new_label_int}.\n    \"\"\"\n    output_pose_dir.mkdir(parents=True, exist_ok=True)\n\n    # Map filename → full Path\n    filename_to_path = {}\n    for p in image_root.rglob(\"*\"):\n        if p.suffix.lower() in {\".png\", \".jpg\", \".jpeg\", \".bmp\", \n                                \".tif\", \".tiff\", \".webp\"}:\n            filename_to_path[p.name] = p\n\n    colmap_labels = {}\n    # Initialize next free index per dataset from precolmap count\n    next_colmap_idx = {ds: dataset_precolmap_count.get(ds, 0) for ds in dataset_precolmap_count}\n\n    # Collect all cluster IDs from disk\n    cluster_ids = sorted(int(p.stem.split(\"_\")[1])\n                         for p in Path(match_root).glob(\"cluster_*.h5\"))\n    \n    for cluster_id in cluster_ids:\n        try:\n            if VERBOSE:\n                print(f\"\\n[Cluster {cluster_id}] running COLMAP on cluster_{cluster_id}\")\n\n            feature_dir  = Path(match_root) / f\"cluster_{cluster_id}\"\n            keypoints_h5 = feature_dir / \"keypoints.h5\"\n            if not keypoints_h5.exists():\n                raise FileNotFoundError(f\"{keypoints_h5} is missing\")\n\n            # Read full filenames from keypoints.h5\n            with h5py.File(keypoints_h5, \"r\") as f_kp:\n                filenames = list(f_kp.keys())\n\n            # Determine dataset for this cluster\n            dataset = cluster_to_dataset[cluster_id]\n\n            # Build full image Paths for COLMAP\n            cluster_image_paths = []\n            for fname in filenames:\n                if fname not in filename_to_path:\n                    raise FileNotFoundError(f\"{fname} not found under {image_root}\")\n                cluster_image_paths.append(filename_to_path[fname])\n\n            if len(cluster_image_paths) < 3:\n                raise ValueError(\"Need ≥3 images for COLMAP\")\n\n            # Create a scene directory and symlink images\n            scene_dir = Path(f\"/kaggle/working/scenes/scene_{cluster_id}\")\n            scene_dir.mkdir(parents=True, exist_ok=True)\n            for src in cluster_image_paths:\n                dst = scene_dir / src.name\n                if not dst.exists():\n                    os.symlink(src, dst)\n\n            database_path = scene_dir / \"database.db\"\n            if database_path.exists():\n                database_path.unlink()\n            (scene_dir / \"colmap_output\").mkdir(exist_ok=True)\n\n            if VERBOSE:\n                print(\" • importing keypoints & matches\")\n                pycolmap_import = import_into_colmap\n                pycolmap_import(\n                    img_dir=str(scene_dir),\n                    feature_dir=str(feature_dir),\n                    database_path=str(database_path),\n                    img_ext=\"\"\n                )\n            else:\n                with suppress_cpp_output():\n                    import_into_colmap(\n                        img_dir=str(scene_dir),\n                        feature_dir=str(feature_dir),\n                        database_path=str(database_path),\n                        img_ext=\"\"\n                    )\n\n            \n            # ------------------------------------------------------ exhaustive match (original)\n            if VERBOSE:\n                print(\" • running COLMAP exhaustive two‐view matching\")\n                pycolmap.match_exhaustive(database_path=str(database_path))\n            else:\n                with suppress_cpp_output():\n                    pycolmap.match_exhaustive(database_path=str(database_path))\n                    \n            # Incremental mapping options\n            mapper_opts = pycolmap.IncrementalPipelineOptions()\n            mapper_opts.min_model_size               = 6\n            mapper_opts.max_num_models               = 50\n            mapper_opts.num_threads                  = 2\n            mapper_opts.ba_global_function_tolerance = 1e-6\n\n            if VERBOSE:\n                maps = pycolmap.incremental_mapping(\n                    database_path=str(database_path),\n                    image_path=str(scene_dir),\n                    output_path=str(scene_dir / \"colmap_output\"),\n                    options=mapper_opts,\n                )\n            else:\n                with suppress_cpp_output():\n                    maps = pycolmap.incremental_mapping(\n                        database_path=str(database_path),\n                        image_path=str(scene_dir),\n                        output_path=str(scene_dir / \"colmap_output\"),\n                        options=mapper_opts,\n                    )\n            if not maps:\n                raise RuntimeError(\"COLMAP produced no model\")\n\n            # Record new labels & poses\n            models = maps if isinstance(maps, list) else list(maps.values())\n            for submodel_idx, model in enumerate(models):\n                for img in model.images.values():\n                    key = f\"{dataset}|{img.name}\"\n                    base = next_colmap_idx.setdefault(dataset, 0)\n                    new_label = base + submodel_idx\n                    colmap_labels[key] = new_label\n\n            # Advance the next free index by how many submodels were created\n            next_colmap_idx[dataset] += len(models)\n\n            # Save camera poses\n            pose_file = output_pose_dir / f\"poses_cluster_{cluster_id}.npz\"\n            poses = {}\n            for model in models:\n                for img in model.images.values():\n                    key = f\"{dataset}|{img.name}\"\n                    R = img.cam_from_world.rotation.matrix().astype(\"float32\")\n                    t = img.cam_from_world.translation.astype(\"float32\")\n                    poses[f\"{key}_R\"] = R\n                    poses[f\"{key}_t\"] = t\n\n            np.savez_compressed(pose_file, **poses)\n            if VERBOSE:\n                print(f\" • saved {len(poses)//2} poses → {pose_file}\")\n\n        except Exception as e:\n            print(f\"[Cluster {cluster_id}] failed: {e}\")\n            # Ensure we create an empty file so downstream sees missing poses\n            np.savez_compressed(output_pose_dir / f\"poses_cluster_{cluster_id}.npz\")\n\n        finally:\n            # Cleanup scene_dir and feature_dir\n            try:\n                shutil.rmtree(scene_dir)\n            except Exception:\n                pass\n            try:\n                shutil.rmtree(feature_dir)\n            except Exception:\n                pass\n            gc.collect()\n\n    if VERBOSE:\n        print(\"\\nAll clusters processed → poses stored in\", output_pose_dir)\n\n    return colmap_labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T11:32:08.265896Z","iopub.execute_input":"2025-06-02T11:32:08.266655Z","iopub.status.idle":"2025-06-02T11:32:08.282332Z","shell.execute_reply.started":"2025-06-02T11:32:08.266633Z","shell.execute_reply":"2025-06-02T11:32:08.281585Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Implementation","metadata":{}},{"cell_type":"code","source":"\n# ── Run the COLMAP reconstruction step ──\nscene_recons = reconstruct_colmap_per_cluster_efficient(\n    cluster_results       = cluster_results,\n    cluster_to_dataset    = cluster_to_dataset,\n    dataset_feature_paths = dataset_feature_paths,\n    match_root            = Path(\"/kaggle/working/final_matches_subclusters\"),\n    image_root            = Path(\"/kaggle/input/image-matching-challenge-2025/test\"),\n    dataset_precolmap_count = dataset_precolmap_count,\n    output_pose_dir       = Path(\"/kaggle/working/pose_outputs\")\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T11:32:08.917296Z","iopub.execute_input":"2025-06-02T11:32:08.917804Z","iopub.status.idle":"2025-06-02T11:33:57.320450Z","shell.execute_reply.started":"2025-06-02T11:32:08.917779Z","shell.execute_reply":"2025-06-02T11:33:57.319908Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Extract all_poses","metadata":{}},{"cell_type":"code","source":"# ─────────────────────────────────────────────────────────────────────────────\n# 1. Load poses from COLMAP output\n# ─────────────────────────────────────────────────────────────────────────────\nimport numpy as np, pandas as pd, torch, h5py, gc\nfrom pathlib import Path\nfrom sklearn.neighbors import NearestNeighbors\nfrom tqdm import tqdm\nimport scipy\n\npose_dir  = Path(\"/kaggle/working/pose_outputs\")\nall_poses = {}  # maps \"dataset|image\" ➜ (R, t)\n\nfor f in pose_dir.glob(\"poses_cluster_*.npz\"):\n    data = np.load(f)\n    for k in data.files:\n        if k.endswith(\"_R\"):\n            base = k[:-2]  # \"<dataset>|<image>\"\n            all_poses[base] = (\n                data[k],\n                data[f\"{base}_t\"].astype(np.float32)\n            )\n\nif VERBOSE:\n    print(\"Loaded poses for images:\", len(all_poses))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T11:41:31.983077Z","iopub.execute_input":"2025-06-02T11:41:31.983650Z","iopub.status.idle":"2025-06-02T11:41:32.005085Z","shell.execute_reply.started":"2025-06-02T11:41:31.983629Z","shell.execute_reply":"2025-06-02T11:41:32.004491Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Fill Unregistered but Labeled Images","metadata":{}},{"cell_type":"code","source":"# ─────────────────────────────────────────────────────────────────────────────\n# 2. Mark missing poses as outliers (skip synthetic fill)\n# ─────────────────────────────────────────────────────────────────────────────\ndevice = get_device()                           # defined earlier\nextractor_net = extractor.model.eval().to(device)\nVALID_SUFFIXES = {\".png\", \".jpg\", \".jpeg\", \".bmp\", \".tif\", \".tiff\", \".webp\"}\n\nfilename_to_path = {\n    p.name: p for p in TEST_DIR.rglob(\"*\")\n    if p.suffix.lower() in VALID_SUFFIXES\n}\nfeatidx_to_name = {\n    idx: (ds, img) for (ds, img), idx in image_to_featidx.items()\n}\n\ndef global_desc(img_path: Path) -> np.ndarray:\n    with torch.no_grad():\n        img  = load_lg_image(str(img_path), resize=getattr(extractor_net, \"resize\", None)).to(device)\n        feats = extractor_net.extract(img)\n        desc  = feats[\"descriptors\"][0].mean(dim=0).cpu().numpy()\n    return desc\n\nfor cid, feat_idx_list in cluster_results.items():\n    missing = [\n        i for i in feat_idx_list\n        if f\"{featidx_to_name[i][0]}|{featidx_to_name[i][1]}\" not in all_poses\n    ]\n    # Move all missing images to outliers\n    for i in missing:\n        cluster_results_original[i][\"label\"] = -1\n    # Skip any further synthetic pose filling\n    continue","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T11:41:32.409779Z","iopub.execute_input":"2025-06-02T11:41:32.410001Z","iopub.status.idle":"2025-06-02T11:41:32.442922Z","shell.execute_reply.started":"2025-06-02T11:41:32.409987Z","shell.execute_reply":"2025-06-02T11:41:32.442332Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"from collections import defaultdict\n\ndef write_submission_csv_custom(\n    SUBMIT_CSV: str,\n    sample_submission_path: str,\n    image_to_featidx: dict,\n    cluster_results_original: dict,\n    all_poses: dict,\n    colmap_labels: dict,\n    dataset_precolmap_count: dict\n):\n    \"\"\"\n    • Multi-object datasets ( ≥2 pre-COLMAP clusters ):\n        scene = cluster<ID> from HDBSCAN/K-means.\n        If an image was an outlier but COLMAP registered it, give it a new cluster<ID>.\n    • Single-scene datasets ( exactly 1 pre-COLMAP cluster ):\n        *ignore* the HDB label; use COLMAP sub-model index instead.\n        Unregistered images → outliers.\n    • Every new scene key gets the next global integer ID 0,1,2,… in first-appearance order.\n    \"\"\"\n    tpl        = pd.read_csv(sample_submission_path)\n    use_img_id = \"image_id\" in tpl.columns\n\n    fmt9 = lambda M: \";\".join(f\"{x:.9f}\" for x in M.flatten())\n    fmt3 = lambda v: \";\".join(f\"{x:.9f}\" for x in v.flatten())\n    nan9, nan3 = \";\".join([\"nan\"]*9), \";\".join([\"nan\"]*3)\n\n    scene_key_to_idx = {}\n    next_idx = 0\n\n    with open(SUBMIT_CSV, \"w\") as f:\n        f.write(\",\".join(tpl.columns) + \"\\n\")\n\n        for _, row in tpl.iterrows():\n            ds, img = row[\"dataset\"], row[\"image\"]\n            img_id  = row[\"image_id\"] if use_img_id else \"\"\n\n            feat_idx = image_to_featidx.get((ds, img))\n            scene_name = \"outliers\"\n            scene_key  = None                    # stay None for true outliers\n\n            single_scene = (dataset_precolmap_count.get(ds, 0) == 1)\n\n            if feat_idx is not None:\n                info      = cluster_results_original.get(feat_idx, {})\n                hdb_label = info.get(\"label\", -1)\n\n                key_pose  = f\"{ds}|{img}\"\n                has_pose  = key_pose in colmap_labels\n\n                if single_scene:\n                    # Ignore any HDB label; rely only on COLMAP\n                    if has_pose:                        # reconstructed\n                        scene_key = (\"colmap\", ds, colmap_labels[key_pose])\n                else:\n                    # Multi-object path\n                    if hdb_label >= 0:\n                        scene_key = (\"hdb\", ds, hdb_label)\n                    elif has_pose:                      # outlier but registered\n                        scene_key = (\"colmap\", ds, colmap_labels[key_pose])\n\n            # Assign global index if this is a real scene\n            if scene_key is not None:\n                if scene_key not in scene_key_to_idx:\n                    scene_key_to_idx[scene_key] = next_idx\n                    next_idx += 1\n                scene_name = f\"cluster{scene_key_to_idx[scene_key]}\"\n\n            # ─── pose or NaNs ────────────────────────────────────────────\n            key_pose = f\"{ds}|{img}\"\n            if key_pose in all_poses:\n                R, t = all_poses[key_pose]\n                R_str, t_str = fmt9(R), fmt3(t)\n            else:\n                R_str, t_str = nan9, nan3\n\n            # ─── write row ───────────────────────────────────────────────\n            if use_img_id:\n                f.write(f\"{img_id},{ds},{scene_name},{img},{R_str},{t_str}\\n\")\n            else:\n                f.write(f\"{ds},{scene_name},{img},{R_str},{t_str}\\n\")\n\n    print(\"Submission file written to\", SUBMIT_CSV)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T11:41:34.332654Z","iopub.execute_input":"2025-06-02T11:41:34.333285Z","iopub.status.idle":"2025-06-02T11:41:34.343812Z","shell.execute_reply.started":"2025-06-02T11:41:34.333262Z","shell.execute_reply":"2025-06-02T11:41:34.343077Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─────────────────────────────────────────────────────────────────────────────\n# 4. Run the submission writer\n# ─────────────────────────────────────────────────────────────────────────────\nwrite_submission_csv_custom(\n    SUBMIT_CSV                = \"/kaggle/working/submission.csv\",\n    sample_submission_path    = \"/kaggle/input/image-matching-challenge-2025/sample_submission.csv\",\n    image_to_featidx          = image_to_featidx,\n    cluster_results_original  = cluster_results_original,\n    all_poses                 = all_poses,\n    colmap_labels             = scene_recons,\n    dataset_precolmap_count   = dataset_precolmap_count\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T11:41:34.539842Z","iopub.execute_input":"2025-06-02T11:41:34.540403Z","iopub.status.idle":"2025-06-02T11:41:34.653051Z","shell.execute_reply.started":"2025-06-02T11:41:34.540375Z","shell.execute_reply":"2025-06-02T11:41:34.652253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}