{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91498,"databundleVersionId":11655853,"sourceType":"competition"},{"sourceId":5632975,"sourceType":"datasetVersion","datasetId":3238926},{"sourceId":7884485,"sourceType":"datasetVersion","datasetId":4628051},{"sourceId":11556081,"sourceType":"datasetVersion","datasetId":7245970},{"sourceId":11628614,"sourceType":"datasetVersion","datasetId":7259927},{"sourceId":11924468,"sourceType":"datasetVersion","datasetId":6988459},{"sourceId":4534,"sourceType":"modelInstanceVersion","modelInstanceId":3326,"modelId":986},{"sourceId":17191,"sourceType":"modelInstanceVersion","modelInstanceId":14317,"modelId":21716},{"sourceId":17555,"sourceType":"modelInstanceVersion","modelInstanceId":14611,"modelId":22086},{"sourceId":27885,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":8020,"modelId":9362},{"sourceId":4557,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":3349,"modelId":1030},{"sourceId":4554,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":3346,"modelId":1030}],"dockerImageVersionId":30919,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install --no-index /kaggle/input/imc2024-packages-lightglue-rerun-kornia/* --no-deps\n!mkdir -p /root/.cache/torch/hub/checkpoints\n!cp /kaggle/input/clip/keras/clip-vit-base-patch32/6/model.weights.h5 /root/.cache/torch/hub/checkpoints/\n!cp /kaggle/input/aliked/pytorch/aliked-n16/1/aliked-n16.pth /root/.cache/torch/hub/checkpoints/\n!cp /kaggle/input/lightglue/pytorch/aliked/1/aliked_lightglue.pth /root/.cache/torch/hub/checkpoints/\n!cp /kaggle/input/lightglue/pytorch/aliked/1/aliked_lightglue.pth /root/.cache/torch/hub/checkpoints/aliked_lightglue_v0-1_arxiv-pth\n!pip install -U /kaggle/input/faiss-gpu-173-python310/faiss_gpu-1.7.2-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T01:26:23.697075Z","iopub.execute_input":"2025-06-02T01:26:23.697387Z","iopub.status.idle":"2025-06-02T01:26:30.522414Z","shell.execute_reply.started":"2025-06-02T01:26:23.697363Z","shell.execute_reply":"2025-06-02T01:26:30.521557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !pip install umap-learn","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn.functional as F\nimport kornia as K\nimport kornia.feature as KF\nimport h5py\nimport dataclasses\nfrom IPython.display import clear_output\nfrom collections import defaultdict\nfrom copy import deepcopy\nfrom lightglue import match_pair\nfrom lightglue import ALIKED, LightGlue\nfrom lightglue.utils import load_image, rbd\nfrom transformers import AutoImageProcessor, AutoModel\nfrom transformers import CLIPProcessor, CLIPModel\nfrom PIL import Image\nimport os\nimport cv2 as cv\nfrom tqdm import tqdm\nfrom time import time, sleep\n# import umap\nimport gc\nimport pandas as pd\nimport numpy as np\nimport faiss\nimport matplotlib.pyplot as plt\n\nimport pycolmap\nimport sys\nsys.path.append('/kaggle/input/imc25-utils')\nfrom database import *\nfrom h5_to_db import *\nimport metric","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T01:26:30.523886Z","iopub.execute_input":"2025-06-02T01:26:30.524229Z","iopub.status.idle":"2025-06-02T01:26:37.170852Z","shell.execute_reply.started":"2025-06-02T01:26:30.524195Z","shell.execute_reply":"2025-06-02T01:26:37.169947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = K.utils.get_cuda_device_if_available(0)\nprint(f'{device=}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T01:26:37.172554Z","iopub.execute_input":"2025-06-02T01:26:37.173160Z","iopub.status.idle":"2025-06-02T01:26:37.221041Z","shell.execute_reply.started":"2025-06-02T01:26:37.173135Z","shell.execute_reply":"2025-06-02T01:26:37.220338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dir = \"/kaggle/input/image-matching-challenge-2025\"\ntrain_labels = pd.read_csv(f\"{data_dir}/train_labels.csv\")\ntrain_labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T01:26:37.221983Z","iopub.execute_input":"2025-06-02T01:26:37.222326Z","iopub.status.idle":"2025-06-02T01:26:37.258591Z","shell.execute_reply.started":"2025-06-02T01:26:37.222293Z","shell.execute_reply":"2025-06-02T01:26:37.257902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def kmeans(X, cluster_num):\n    print(\"Perform K-means clustering...\")\n    d = X.shape[1]\n    X = X.astype(np.float32)\n    kmeans = faiss.Kmeans(d, cluster_num, gpu=True, spherical=True, niter=300, nredo=10)\n    kmeans.train(X)\n    D, I = kmeans.index.search(X, 1)\n    I = I.reshape(-1)\n    print(\"K-means clustering done.\")\n    return I\n    \ndef get_image_embeddings(image_paths, clip_model, processor, device):\n    batch_size = 2048\n    num_images = len(image_paths)\n    features = []\n    for i in range(num_images // batch_size + 1):\n        start = i * batch_size\n        end = start + batch_size\n        if end > num_images:\n            end = num_images\n        images_batch = []\n        for image_path in image_paths[start:end]:\n            images_batch.append(Image.open(image_path))  \n\n        with torch.no_grad():\n            inputs = processor(images=images_batch, return_tensors=\"pt\", padding=True).to(device)\n            feature = clip_model.get_image_features(**inputs)\n            features.append(feature)\n\n        if i % 50 == 0:\n            print(f\"[Completed {i * batch_size}/{num_images}]\")\n            \n    features = torch.cat(features)\n    return features\n\ndef construct_text_counterparts(samples, device, is_train=True):\n    # Load Pretrained CLIP\n    # model_name = \"openai/clip-vit-base-patch32\"\n    model_name = \"/kaggle/input/clip-vit/pytorch/b-32-laion2b-s34b-b79k/1\"\n    clip_model = CLIPModel.from_pretrained(model_name).to(device)\n    processor = CLIPProcessor.from_pretrained(model_name)\n    clip_model.eval()\n\n    # Get image paths from all datasets\n    image_paths = []\n\n    \n    for dataset, predictions in samples.items():\n        images_dir = os.path.join(data_dir, 'train' if is_train else 'test', dataset)\n        if not is_train and not os.path.isdir(images_dir):\n            continue\n        paths = [os.path.join(images_dir, p.filename) for p in predictions]\n        image_paths.extend(paths)\n\n    def get_text_embeddings():\n        nouns = pd.read_csv(f\"/kaggle/input/wordnetnouns/WordNetNouns.csv\").values\n        nouns_num = nouns.shape[0]\n        batch_size = 2048\n        features = []\n        for i in range(nouns_num // batch_size + 1):\n            start = i * batch_size\n            end = start + batch_size\n            if end > nouns_num:\n                end = nouns_num\n            nouns_batch = nouns[start:end]\n            with torch.no_grad():\n                prompt = [f\"a photo of a {word}\" for word in nouns_batch[:, 0]]\n                text = processor(text=prompt, return_tensors=\"pt\", padding=True, truncation=True).to(device)\n                feature = clip_model.get_text_features(**text)\n                features.append(feature)\n            if i % 50 == 0:\n                print(f\"[Completed {i * batch_size}/{nouns_num}]\")\n        features = torch.cat(features)\n        return features, nouns\n\n    # Get text embeddings using CLIP\n    text_embeddings, nouns = get_text_embeddings()\n    \n    # Get image embeddings using CLIP\n    image_embeddings = get_image_embeddings(image_paths, clip_model, processor, device)\n\n    # Normalize embeddings\n    text_embeddings = text_embeddings / text_embeddings.norm(dim=1, keepdim=True)\n    image_embeddings = image_embeddings / image_embeddings.norm(dim=1, keepdim=True)\n\n    text_embeddings = text_embeddings.half()\n    image_embeddings = image_embeddings.half()\n\n    n_nouns = text_embeddings.shape[0]\n    n_images = image_embeddings.shape[0]\n    \n    # Find cluster_num semantic clusters in image embeddings based on text embeddings\n    cluster_num = n_images//40\n    preds = kmeans(image_embeddings.cpu().numpy(), cluster_num)\n\n    image_centers = torch.zeros((cluster_num, 512), dtype=torch.float16).cuda()\n    for k in range(cluster_num):\n        image_centers[k] = image_embeddings[preds == k].mean(dim=0)\n    image_centers = F.normalize(image_centers, dim=1)\n\n    # Match nouns to image centers\n    similarity = torch.matmul(image_centers, text_embeddings.T)\n    softmax_nouns = torch.softmax(similarity, dim=0).cpu().float()\n    class_pred = torch.argmax(softmax_nouns, dim=0).long()\n\n    # Identify highly distinguishable nouns by choosing topK most confident nouns for each image cluster\n    topK = 5\n    selected_idx = torch.zeros_like(class_pred, dtype=torch.bool)\n    for k in range(cluster_num):\n        if (class_pred == k).sum() == 0:\n            continue\n        class_index = torch.where(class_pred == k)[0]\n        softmax_class = softmax_nouns[:, class_index]\n        confidence = softmax_class.max(dim=0)[0]\n        rank = torch.argsort(confidence, descending=True)\n        selected_idx[class_index[rank[:topK]]] = True\n    selected_idx = selected_idx.cpu().numpy()\n\n    print(selected_idx.sum(), \"nouns selected.\")\n    text_embeddings_selected = text_embeddings[selected_idx]\n\n    # Use selected nouns for zero-shot classification\n    tau = 0.005\n    retrieval_embeddings = []\n    batch_size = 8192\n    for i in range(n_images // batch_size + 1):\n        start = i * batch_size\n        end = start + batch_size\n        if end > n_images:\n            end = n_images\n            images_batch = image_embeddings[start:end]\n        similarity = torch.matmul(image_embeddings[start:end], text_embeddings_selected.T)\n        similarity = torch.softmax(similarity / tau, dim=1)\n        retrieval_embedding = (similarity @ text_embeddings_selected).cpu()\n        retrieval_embeddings.append(retrieval_embedding)\n        if i % 50 == 0:\n            print(f\"[Completed {i * batch_size}/{n_images}]\")\n    retrieval_embedding = torch.cat(retrieval_embeddings, dim=0).cuda().half()\n    retrieval_embedding = F.normalize(retrieval_embedding, dim=1)\n    concat_embeddings = torch.cat([image_embeddings, retrieval_embedding], axis=1)\n\n    # Cleanup\n    del clip_model\n    torch.cuda.empty_cache()\n    gc.collect()\n    selected_nouns = [noun for noun, is_selected in zip(nouns[:, 0], selected_idx) if is_selected]\n    return concat_embeddings, text_embeddings_selected, selected_nouns\n\ndef get_dataset_embeddings(samples, dataset, concat_embeddings):\n    # Util to select the right embeddings from concat embeddings by dataset name\n    start = 0\n    count = 0\n    for ds, predictions in samples.items():\n        if ds != dataset:\n            start += len(predictions)\n        else:\n            count = len(predictions)\n            break\n    end = start + count\n    embeddings = concat_embeddings[start:end].cpu().numpy()\n    return embeddings\n\ndef cluster_images(embeddings, n_neighbors, min_dist, metric='cosine'):\n    # Dimensionality reduction\n    umap_model = umap.UMAP(\n        n_neighbors = n_neighbors,\n        min_dist = min_dist,\n        n_components = 2,\n        metric = metric\n    )\n    reduced_embeddings = umap_model.fit_transform(embeddings)\n    return reduced_embeddings","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T01:27:53.759885Z","iopub.execute_input":"2025-06-02T01:27:53.760231Z","iopub.status.idle":"2025-06-02T01:27:53.778903Z","shell.execute_reply.started":"2025-06-02T01:27:53.760204Z","shell.execute_reply":"2025-06-02T01:27:53.778049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Code provided by Octavi Grau https://www.kaggle.com/code/octaviograu/baseline-dinov2-aliked-lightglue\ndef load_torch_image(fname, device=torch.device('cpu')):\n    img = K.io.load_image(fname, K.io.ImageLoadType.RGB32, device=device)[None, ...]\n    return img\n\ndef get_global_desc(fnames, device = torch.device('cpu')):\n    processor = AutoImageProcessor.from_pretrained('/kaggle/input/dinov2/pytorch/base/1')\n    model = AutoModel.from_pretrained('/kaggle/input/dinov2/pytorch/base/1')\n    model = model.eval()\n    model = model.to(device)\n    global_descs_dinov2 = []\n    for i, img_fname_full in tqdm(enumerate(fnames),total= len(fnames)):\n        key = os.path.splitext(os.path.basename(img_fname_full))[0]\n        timg = load_torch_image(img_fname_full)\n        with torch.inference_mode():\n            inputs = processor(images=timg, return_tensors=\"pt\", do_rescale=False).to(device)\n            outputs = model(**inputs)\n            dino_mac = F.normalize(outputs.last_hidden_state[:,1:].max(dim=1)[0], dim=1, p=2)\n        global_descs_dinov2.append(dino_mac.detach().cpu())\n    global_descs_dinov2 = torch.cat(global_descs_dinov2, dim=0)\n    return global_descs_dinov2\n    \ndef get_img_pairs_exhaustive(img_fnames):\n    index_pairs = []\n    for i in range(len(img_fnames)):\n        for j in range(i+1, len(img_fnames)):\n            index_pairs.append((i,j))\n    return index_pairs\n\n\ndef get_image_pairs_shortlist(fnames,\n                              descs,\n                              sim_th = 0.6, # should be strict\n                              min_pairs = 30,\n                              exhaustive_if_less = 20,\n                              device=torch.device('cpu')):\n    num_imgs = len(fnames)\n    if num_imgs <= exhaustive_if_less:\n        return get_img_pairs_exhaustive(fnames)\n    \n    dm = torch.cdist(descs, descs, p=2).numpy()\n    # removing half\n    mask = dm <= sim_th\n    total = 0\n    matching_list = []\n    ar = np.arange(num_imgs)\n    already_there_set = []\n    for st_idx in range(num_imgs-1):\n        mask_idx = mask[st_idx]\n        to_match = ar[mask_idx]\n        if len(to_match) < min_pairs:\n            to_match = np.argsort(dm[st_idx])[:min_pairs]  \n        for idx in to_match:\n            if st_idx == idx:\n                continue\n            if dm[st_idx, idx] < 1000:\n                matching_list.append(tuple(sorted((st_idx, idx.item()))))\n                total+=1\n    matching_list = sorted(list(set(matching_list)))\n    return matching_list\n\ndef detect_aliked(img_fnames,\n                  feature_dir = '.featureout',\n                  num_features = 4096,\n                  resize_to = 1024,\n                  device=torch.device('cpu')):\n    dtype = torch.float32 # ALIKED has issues with float16\n    extractor = ALIKED(max_num_keypoints=num_features, detection_threshold=0.01, resize=resize_to).eval().to(device, dtype)\n    if not os.path.isdir(feature_dir):\n        os.makedirs(feature_dir)\n    with h5py.File(f'{feature_dir}/keypoints.h5', mode='w') as f_kp, \\\n         h5py.File(f'{feature_dir}/descriptors.h5', mode='w') as f_desc:\n        for img_path in tqdm(img_fnames):\n            img_fname = img_path.split('/')[-1]\n            key = img_fname\n            with torch.inference_mode():\n                image0 = load_torch_image(img_path, device=device).to(dtype)\n                feats0 = extractor.extract(image0)  # auto-resize the image, disable with resize=None\n                kpts = feats0['keypoints'].reshape(-1, 2).detach().cpu().numpy()\n                descs = feats0['descriptors'].reshape(len(kpts), -1).detach().cpu().numpy()\n                f_kp[key] = kpts\n                f_desc[key] = descs\n    return\n\ndef match_with_lightglue(img_fnames,\n                   index_pairs,\n                   feature_dir = '.featureout',\n                   device=torch.device('cpu'),\n                   min_matches=25,verbose=True):\n    lg_matcher = KF.LightGlueMatcher(\"aliked\", {\"width_confidence\": -1,\n                                                \"depth_confidence\": -1,\n                                                 \"mp\": True if 'cuda' in str(device) else False}).eval().to(device)\n    with h5py.File(f'{feature_dir}/keypoints.h5', mode='r') as f_kp, \\\n        h5py.File(f'{feature_dir}/descriptors.h5', mode='r') as f_desc, \\\n        h5py.File(f'{feature_dir}/matches.h5', mode='w') as f_match:\n        for pair_idx in tqdm(index_pairs):\n            idx1, idx2 = pair_idx\n            fname1, fname2 = img_fnames[idx1], img_fnames[idx2]\n            key1, key2 = fname1.split('/')[-1], fname2.split('/')[-1]\n            kp1 = torch.from_numpy(f_kp[key1][...]).to(device)\n            kp2 = torch.from_numpy(f_kp[key2][...]).to(device)\n            desc1 = torch.from_numpy(f_desc[key1][...]).to(device)\n            desc2 = torch.from_numpy(f_desc[key2][...]).to(device)\n            with torch.inference_mode():\n                dists, idxs = lg_matcher(desc1,\n                                         desc2,\n                                         KF.laf_from_center_scale_ori(kp1[None]),\n                                         KF.laf_from_center_scale_ori(kp2[None]))\n            if len(idxs)  == 0:\n                continue\n            n_matches = len(idxs)\n            if verbose:\n                print (f'{key1}-{key2}: {n_matches} matches')\n            group  = f_match.require_group(key1)\n            if n_matches >= min_matches:\n                 group.create_dataset(key2, data=idxs.detach().cpu().numpy().reshape(-1, 2))\n    return\n\ndef import_into_colmap(img_dir, feature_dir ='.featureout', database_path = 'colmap.db'):\n    db = COLMAPDatabase.connect(database_path)\n    db.create_tables()\n    single_camera = False\n    fname_to_id = add_keypoints(db, feature_dir, img_dir, '', 'simple-pinhole', single_camera)\n    add_matches(\n        db,\n        feature_dir,\n        fname_to_id,\n    )\n    db.commit()\n    return\n\n@dataclasses.dataclass\nclass Prediction:\n    image_id: str | None  # A unique identifier for the row -- unused otherwise. Used only on the hidden test set.\n    dataset: str\n    filename: str\n    cluster_index: int | None = None\n    rotation: np.ndarray | None = None\n    translation: np.ndarray | None = None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T01:26:40.430011Z","iopub.execute_input":"2025-06-02T01:26:40.430353Z","iopub.status.idle":"2025-06-02T01:26:40.448317Z","shell.execute_reply.started":"2025-06-02T01:26:40.430318Z","shell.execute_reply":"2025-06-02T01:26:40.447406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"is_train = False # Set to False if submitting to contest\nworkdir = '/kaggle/working/result/'\nos.makedirs(workdir, exist_ok=True)\n\nif is_train:\n    sample_submission_csv = os.path.join(data_dir, 'train_labels.csv')\nelse:\n    sample_submission_csv = os.path.join(data_dir, 'sample_submission.csv')\n\nsamples = {}\ncompetition_data = pd.read_csv(sample_submission_csv)\nfor _, row in competition_data.iterrows():\n    # Note: For the test data, the \"scene\" column has no meaning, and the rotation_matrix and translation_vector columns are random.\n    if row.dataset not in samples:\n        samples[row.dataset] = []\n    samples[row.dataset].append(\n        Prediction(\n            image_id=None if is_train else row.image_id,\n            dataset=row.dataset,\n            filename=row.image\n        )\n    )\n\nfor dataset in samples:\n    print(f'Dataset \"{dataset}\" -> num_images={len(samples[dataset])}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T01:26:46.473767Z","iopub.execute_input":"2025-06-02T01:26:46.474070Z","iopub.status.idle":"2025-06-02T01:26:46.617321Z","shell.execute_reply.started":"2025-06-02T01:26:46.474047Z","shell.execute_reply":"2025-06-02T01:26:46.616460Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualize text counterparts","metadata":{}},{"cell_type":"code","source":"# Set to 0 to use DINOv2 embeddings\n# Set to 1 to construct text counterparts; \n# Set to 2 to only get image embeddings using CLIP\nclip_option = 1\n\nif clip_option == 1:\n    # Construct text counterparts on all images (zero-shot classification)\n    concat_embeddings, text_embeddings_selected, nouns = construct_text_counterparts(samples, device, is_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T01:27:58.928278Z","iopub.execute_input":"2025-06-02T01:27:58.928607Z","iopub.status.idle":"2025-06-02T01:28:03.719666Z","shell.execute_reply.started":"2025-06-02T01:27:58.928582Z","shell.execute_reply":"2025-06-02T01:28:03.718372Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def visualize_text_counterparts(example_dataset, nouns, samples, concat_embeddings):\n    example_embeddings = get_dataset_embeddings(samples, example_dataset, concat_embeddings)\n    embedding_dim = 512 #Default embedding dim of CLIP\n    image_embeddings = example_embeddings[:, :embedding_dim]\n    best_indices = torch.argmax(torch.tensor(image_embeddings).to(device) @ text_embeddings_selected.T, axis=1).cpu().numpy()\n    best_nouns = []\n    for idx in best_indices:\n        best_nouns.append(nouns[idx])\n        \n    fig, axs = plt.subplots(2, 5, figsize=(10, 4))\n    \n    # Display 5 sample images with their best-matching noun\n    for i in range(5):\n        index = i\n        prediction = samples[example_dataset][index]\n        image_path = f\"{data_dir}/train/{example_dataset}/{prediction.filename}\"\n        image = Image.open(image_path)\n        image = image.resize((300, 300))\n    \n        axs[0][i].imshow(image)\n        gt_label = train_labels[train_labels[\"image\"] == prediction.filename][\"scene\"].item()\n        axs[0][i].set_title(f\"{best_nouns[index]}\\n({gt_label})\", fontsize=12)\n        axs[0][i].axis('off')\n\n        prediction2 = samples[example_dataset][-index-1]\n        image_path2 = f\"{data_dir}/train/{example_dataset}/{prediction2.filename}\"\n        image2 = Image.open(image_path2)\n        image2 = image2.resize((300, 300))\n        axs[1][i].imshow(image2)\n        gt_label2 = train_labels[train_labels[\"image\"] == prediction2.filename][\"scene\"].item()\n        axs[1][i].set_title(f\"{best_nouns[-index-1]}\\n({gt_label2})\", fontsize=12, fontweight=15)\n        axs[1][i].axis('off')\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if is_train:\n    visualize_text_counterparts(\"imc2023_heritage\", nouns, samples, concat_embeddings)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if is_train:\n    visualize_text_counterparts(\"pt_stpeters_stpauls\", nouns, samples, concat_embeddings)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if is_train:\n    visualize_text_counterparts(\"stairs\", nouns, samples, concat_embeddings)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualize embeddings","metadata":{}},{"cell_type":"code","source":"# Collect embeddings for each method\nif is_train:\n    model_name = \"openai/clip-vit-base-patch32\"\n    clip_model = CLIPModel.from_pretrained(model_name).to(device)\n    processor = CLIPProcessor.from_pretrained(model_name)\n    clip_model.eval()\n    ndatasets = len(train_labels[\"dataset\"].unique())\n\n    # Collect embeddings for all datasets\n    umap_embeddings = [[] for i in range(ndatasets)]\n    for i, (dataset, group) in enumerate(train_labels.groupby('dataset', sort=False)):\n        predictions = samples[dataset] \n        images_dir = os.path.join(data_dir, 'train' if is_train else 'test', dataset)\n        image_paths = [os.path.join(images_dir, p.filename) for p in predictions]\n        \n        # Get DINOv2 embeddings\n        embeddings1 = get_global_desc(image_paths, device)\n\n        # Get CLIP vision encoder embeddings\n        embeddings2 = get_image_embeddings(image_paths, clip_model, processor, device).cpu().numpy()\n    \n        # Get CLIP simularity matrix embeddings with text counterparts\n        embeddings3 = get_dataset_embeddings(samples, dataset, concat_embeddings)\n\n        n_neighbors = 15\n        min_dist = 0.5\n    \n        umap_embeddings1 = cluster_images(embeddings1, n_neighbors, min_dist)\n        umap_embeddings2 = cluster_images(embeddings2, n_neighbors, min_dist)\n        umap_embeddings3 = cluster_images(embeddings3, n_neighbors, min_dist)\n        umap_embeddings[i].append(umap_embeddings1)\n        umap_embeddings[i].append(umap_embeddings2)\n        umap_embeddings[i].append(umap_embeddings3)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot UMAP visualizations of embeddings \ndatasets_to_plot = [\n    \"ETs\",\n    \"stairs\",\n    \"pt_stpeters_stpauls\",\n    \"pt_sacrecoeur_trevi_tajmahal\",\n    \"pt_brandenburg_british_buckingham\",\n    \"imc2023_heritage\"\n]\nif is_train:\n    n = len(datasets_to_plot)\n    fig, axs = plt.subplots(n, 4, figsize=(15, 3*n), sharex=True, sharey=True, constrained_layout=True)\n    axs[0][0].set_title(\"DINOv2 Embeddings\")\n    axs[0][1].set_title(\"CLIP Vision Embeddings\")\n    axs[0][2].set_title(\"CLIP Cross-Modal Embeddings\")\n    plt_idx = 0\n    for i, (dataset, group) in enumerate(train_labels.groupby('dataset', sort=False)):\n        if dataset not in datasets_to_plot:\n            continue\n            \n        # Get the scenes for this dataset, in order of appearance\n        scene_order = group['scene'].drop_duplicates().tolist()\n        \n        # Count number of images per scene\n        scene_counts = group['scene'].value_counts().loc[scene_order].tolist()\n\n        labels = [j for j, count in enumerate(scene_counts) for _ in range(count)]\n        \n        axs[plt_idx][0].scatter(\n            umap_embeddings[i][0][:,0], \n            umap_embeddings[i][0][:,1],\n            c=labels,\n            cmap='viridis'\n        )\n        axs[plt_idx][1].scatter(\n            umap_embeddings[i][1][:,0], \n            umap_embeddings[i][1][:,1],\n            c=labels,\n            cmap='viridis'\n        )\n        axs[plt_idx][2].scatter(\n            umap_embeddings[i][2][:,0], \n            umap_embeddings[i][2][:,1],\n            c=labels,\n            cmap='viridis'\n        )\n\n        axs[plt_idx][0].set_ylabel(dataset, fontsize=10)\n        \n        for j in range(3):\n            if j > 0:\n                axs[plt_idx][j].set_ylabel('')\n                axs[plt_idx][j].set_yticklabels([])\n            if i < ndatasets - 1:\n                axs[plt_idx][j].set_xlabel('')\n                axs[plt_idx][j].set_xticklabels([])\n\n        # Show legend in 4th column\n        handles = []\n        nscenes = len(scene_order)\n        for t, label in enumerate(scene_order):\n            color_range = max(1, nscenes-1)\n            handle = plt.Line2D([0], [0], marker='o', color='w', markerfacecolor=plt.cm.viridis(t/color_range), markersize=10, label=label)\n            handles.append(handle)\n\n        axs[plt_idx][3].axis('off')\n        axs[plt_idx][3].legend(handles=handles, title=\"Scenes\", loc='center left', bbox_to_anchor=(0, 0.5))\n        plt_idx += 1\n        \n    plt.suptitle(\"UMAP Visualization of Image Clusters\", x=0.38, fontsize=16)\n    plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Run COLMAP for Camera Pose Estimation","metadata":{}},{"cell_type":"code","source":"# Code adapted from Octavi Grau https://www.kaggle.com/code/octaviograu/baseline-dinov2-aliked-lightglue\ndatasets_to_process = None #Run on all test datasets\nif is_train:\n    # Note: When running on the training dataset, the notebook will hit the time limit and die. Use this filter to run on a few specific datasets.\n    datasets_to_process = [\n    \t# New data.\n    \t# 'amy_gardens',\n    \t'ETs',\n    \t# 'fbk_vineyard',\n    \t# 'stairs',\n    \t# Data from IMC 2023 and 2024.\n        # 'imc2024_dioscuri_baalshamin',\n    \t# 'imc2023_theather_imc2024_church',\n    \t# 'imc2023_heritage',\n    \t# 'imc2023_haiper',\n    \t# 'imc2024_lizard_pond',\n    \t# Crowdsourced PhotoTourism data.\n    \t# 'pt_stpeters_stpauls',\n    \t# 'pt_brandenburg_british_buckingham',\n    \t# 'pt_piazzasanmarco_grandplace',\n    \t# 'pt_sacrecoeur_trevi_tajmahal',\n    ]\n    \ntimings = {\n    \"shortlisting\":[],\n    \"feature_detection\": [],\n    \"feature_matching\":[],\n    \"RANSAC\": [],\n    \"Reconstruction\": [],\n}\nmapping_result_strs = []\n\nstart = 0\nprint (f\"Extracting on device {device}\")\nfor dataset, predictions in samples.items():\n    if datasets_to_process and dataset not in datasets_to_process:\n        print(f'Skipping \"{dataset}\"')\n        continue\n    \n    images_dir = os.path.join(data_dir, 'train' if is_train else 'test', dataset)\n    images = [os.path.join(images_dir, p.filename) for p in predictions]\n\n    print(f'\\nProcessing dataset \"{dataset}\": {len(images)} images')\n\n    filename_to_index = {p.filename: idx for idx, p in enumerate(predictions)}\n\n    feature_dir = os.path.join(workdir, 'featureout', dataset)\n    os.makedirs(feature_dir, exist_ok=True)\n\n    # Select the correct embeddings based on clip_option\n    if clip_option == 0:\n        embeddings = get_global_desc(images, device)\n    if clip_option == 1:\n        embeddings = torch.tensor(get_dataset_embeddings(samples, dataset, concat_embeddings)).float()\n    elif clip_option == 2:\n        model_name = \"openai/clip-vit-base-patch32\"\n        clip_model = CLIPModel.from_pretrained(model_name).to(device)\n        processor = CLIPProcessor.from_pretrained(model_name)\n        clip_model.eval()\n        embeddings = get_image_embeddings(images, clip_model, processor, device).cpu().float()\n\n    # Wrap algos in try-except blocks so we can populate a submission even if one scene crashes.\n    try:\n        t = time()\n        index_pairs = get_image_pairs_shortlist(\n            images,\n            embeddings,\n            sim_th = 0.3, # should be strict\n            min_pairs = 20, # we should select at least min_pairs PER IMAGE with biggest similarity\n            exhaustive_if_less = 20,\n            device=device\n        )\n        timings['shortlisting'].append(time() - t)\n        print (f'Shortlisting. Number of pairs to match: {len(index_pairs)}. Done in {time() - t:.4f} sec')\n        gc.collect()\n    \n        t = time()\n\n        detect_aliked(images, feature_dir, 4096, device=device)\n        gc.collect()\n        timings['feature_detection'].append(time() - t)\n        print(f'Features detected in {time() - t:.4f} sec')\n        \n        t = time()\n        match_with_lightglue(images, index_pairs, feature_dir=feature_dir, device=device, verbose=False)\n        timings['feature_matching'].append(time() - t)\n        print(f'Features matched in {time() - t:.4f} sec')\n\n        database_path = os.path.join(feature_dir, 'colmap.db')\n        if os.path.isfile(database_path):\n            os.remove(database_path)\n        gc.collect()\n        sleep(1)\n        import_into_colmap(images_dir, feature_dir=feature_dir, database_path=database_path)\n        output_path = f'{feature_dir}/colmap_rec_aliked'\n        \n        t = time()\n        pycolmap.match_exhaustive(database_path)\n        timings['RANSAC'].append(time() - t)\n        print(f'Ran RANSAC in {time() - t:.4f} sec')\n        \n        # By default colmap does not generate a reconstruction if less than 10 images are registered.\n        # Lower it to 3.\n        mapper_options = pycolmap.IncrementalPipelineOptions()\n        mapper_options.min_model_size = 3\n        mapper_options.max_num_models = 25\n        os.makedirs(output_path, exist_ok=True)\n        t = time()\n        maps = pycolmap.incremental_mapping(\n            database_path=database_path, \n            image_path=images_dir,\n            output_path=output_path,\n            options=mapper_options)\n        sleep(1)\n        timings['Reconstruction'].append(time() - t)\n        print(f'Reconstruction done in  {time() - t:.4f} sec')\n        print(maps)\n\n        clear_output(wait=False)\n    \n        registered = 0\n        for map_index, cur_map in maps.items():\n            for index, image in cur_map.images.items():\n                prediction_index = filename_to_index[image.name]\n                predictions[prediction_index].cluster_index = map_index\n                predictions[prediction_index].rotation = deepcopy(image.cam_from_world.rotation.matrix())\n                predictions[prediction_index].translation = deepcopy(image.cam_from_world.translation)\n                registered += 1\n        mapping_result_str = f'Dataset \"{dataset}\" -> Registered {registered} / {len(images)} images with {len(maps)} clusters'\n        mapping_result_strs.append(mapping_result_str)\n        print(mapping_result_str)\n        gc.collect()\n    except Exception as e:\n        print(e)\n        # raise e\n        mapping_result_str = f'Dataset \"{dataset}\" -> Failed!'\n        mapping_result_strs.append(mapping_result_str)\n        print(mapping_result_str)\n\nprint('\\nResults')\nfor s in mapping_result_strs:\n    print(s)\n\nprint('\\nTimings')\nfor k, v in timings.items():\n    print(f'{k} -> total={sum(v):.02f} sec.')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Save data to CSV and Evaluate","metadata":{}},{"cell_type":"code","source":"# Must Create a submission file.\n\narray_to_str = lambda array: ';'.join([f\"{x:.09f}\" for x in array])\nnone_to_str = lambda n: ';'.join(['nan'] * n)\n\nsubmission_file = '/kaggle/working/submission.csv'\nwith open(submission_file, 'w') as f:\n    if is_train:\n        f.write('dataset,scene,image,rotation_matrix,translation_vector\\n')\n        for dataset in samples:\n            for prediction in samples[dataset]:\n                cluster_name = 'outliers' if prediction.cluster_index is None else f'cluster{prediction.cluster_index}'\n                rotation = none_to_str(9) if prediction.rotation is None else array_to_str(prediction.rotation.flatten())\n                translation = none_to_str(3) if prediction.translation is None else array_to_str(prediction.translation)\n                f.write(f'{prediction.dataset},{cluster_name},{prediction.filename},{rotation},{translation}\\n')\n    else:\n        f.write('image_id,dataset,scene,image,rotation_matrix,translation_vector\\n')\n        for dataset in samples:\n            for prediction in samples[dataset]:\n                cluster_name = 'outliers' if prediction.cluster_index is None else f'cluster{prediction.cluster_index}'\n                rotation = none_to_str(9) if prediction.rotation is None else array_to_str(prediction.rotation.flatten())\n                translation = none_to_str(3) if prediction.translation is None else array_to_str(prediction.translation)\n                f.write(f'{prediction.image_id},{prediction.dataset},{cluster_name},{prediction.filename},{rotation},{translation}\\n')\n\n!head {submission_file}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Optional: Load a complete submission.csv for testing\n# Here, we use a custom dataset called old-submission to upload a submission file\n# submission_file = \"/kaggle/input/old-submission/submission.csv\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if is_train:\n    t = time()\n    final_score, dataset_scores = metric.score(\n        gt_csv='/kaggle/input/image-matching-challenge-2025/train_labels.csv',\n        user_csv=submission_file,\n        thresholds_csv='/kaggle/input/image-matching-challenge-2025/train_thresholds.csv',\n        mask_csv=None if is_train else os.path.join(data_dir, 'mask.csv'),\n        inl_cf=0,\n        strict_cf=-1,\n        verbose=True,\n    )\n    print(f'Computed metric in: {time() - t:.02f} sec.')","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}