{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91498,"databundleVersionId":11655853,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":7884485,"sourceType":"datasetVersion","datasetId":4628051},{"sourceId":11924468,"sourceType":"datasetVersion","datasetId":6988459},{"sourceId":12162657,"sourceType":"datasetVersion","datasetId":7652929},{"sourceId":247616341,"sourceType":"kernelVersion"},{"sourceId":4534,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":3326,"modelId":986}],"dockerImageVersionId":31154,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"f9d66a43-4681-4164-a33c-7cfa1308c907","cell_type":"markdown","source":"# Flow 1","metadata":{}},{"id":"5a5a0391-7051-40b7-9cb9-564d6f575244","cell_type":"markdown","source":"## Resize Images","metadata":{}},{"id":"88c34eb4-e360-4e03-bf93-67cf38af1019","cell_type":"code","source":"import cv2\nimport os\nimport numpy as np\n\ndef resize(folder_path, output_folder_name):\n    # Create output folder if not exists\n    os.makedirs(output_folder, exist_ok=True)\n\n    # Loop through all files in input folder\n    for filename in os.listdir(folder_path):\n        img_path = os.path.join(folder_path, filename)\n\n        # Load image (support PNG with alpha)\n        image = cv2.imread(img_path, -1)\n        if image is None:\n            print(f\"⚠️ Skipping {filename} (not a valid image)\")\n            continue\n\n        # Resize to 1280 (width) x 1600 (height)\n        resized = cv2.resize(image, (1280, 1600))\n\n        # Save to output folder (same filename)\n        out_path = os.path.join(output_folder, filename)\n        cv2.imwrite(out_path, resized)\n        print(f\"✅ Saved resized: {out_path}\")\n    \n# overwrites\ndef resize(folder_path):\n    # Loop through all files in the folder\n    for filename in os.listdir(folder_path):\n        img_path = os.path.join(folder_path, filename)\n\n        # Only process images\n        if not filename.lower().endswith((\".png\", \".jpg\", \".jpeg\")):\n            continue\n\n        # Load image (preserve alpha for PNGs)\n        image = cv2.imread(img_path, -1)\n        if image is None:\n            print(f\"⚠️ Skipping {filename} (not a valid image)\")\n            continue\n\n        # Resize to 1280 (width) x 1600 (height)\n        resized = cv2.resize(image, (1280, 1600))\n\n        # Overwrite original file\n        cv2.imwrite(img_path, resized)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:03:46.142278Z","iopub.execute_input":"2025-10-24T09:03:46.142501Z","iopub.status.idle":"2025-10-24T09:03:46.380590Z","shell.execute_reply.started":"2025-10-24T09:03:46.142479Z","shell.execute_reply":"2025-10-24T09:03:46.379869Z"}},"outputs":[],"execution_count":null},{"id":"99a426e3-f8cb-41a6-b59e-844085ca3395","cell_type":"code","source":"import cv2\nimport os\nimport subprocess\n\ndef resize_with_subfolders(input_root, output_root):\n    # Create main output folder\n    os.makedirs(output_root, exist_ok=True)\n\n    # Loop through subfolders in input_root\n    for subfolder in os.listdir(input_root):\n        subfolder_path = os.path.join(input_root, subfolder)\n        if not os.path.isdir(subfolder_path):\n            continue  # skip files at root level\n\n        # Name the output subfolder\n        out_subfolder_name = f\"{subfolder}_resized\"\n        out_subfolder_path = os.path.join(output_root, out_subfolder_name)\n        os.makedirs(out_subfolder_path, exist_ok=True)\n\n        # Loop through images in subfolder\n        for filename in os.listdir(subfolder_path):\n            img_path = os.path.join(subfolder_path, filename)\n\n            # Load image (support PNG with alpha)\n            image = cv2.imread(img_path, -1)\n            if image is None:\n                print(f\"⚠️ Skipping {filename} (not a valid image)\")\n                continue\n\n            # Resize to 1280 (width) x 1600 (height)\n            resized = cv2.resize(image, (1280, 1600))\n\n            # Save to output subfolder\n            out_path = os.path.join(out_subfolder_path, filename)\n            cv2.imwrite(out_path, resized)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:03:46.382407Z","iopub.execute_input":"2025-10-24T09:03:46.382604Z","iopub.status.idle":"2025-10-24T09:03:46.388279Z","shell.execute_reply.started":"2025-10-24T09:03:46.382588Z","shell.execute_reply":"2025-10-24T09:03:46.387509Z"}},"outputs":[],"execution_count":null},{"id":"22a77c62-e2ff-4da6-bf16-9d10441ce67e","cell_type":"markdown","source":"## Convert Raw Pictures --> Vectors ( DINOv2 )","metadata":{}},{"id":"a57183c7-dc31-4322-ba31-9eb6cab700b4","cell_type":"code","source":"import torch\nimport numpy as np\nfrom pathlib import Path\nfrom transformers import AutoImageProcessor, AutoModel\nfrom PIL import Image, ImageFile\nimport torchvision.transforms as T\n\n# allow PIL to load slightly truncated/partial PNGs instead of throwing\nImageFile.LOAD_TRUNCATED_IMAGES = True\n\nDINOV2_DIR = Path(\"/kaggle/input/dinov2/pytorch/base/1\")\n\ndef load_dinov2_from_dir(local_dir: Path):\n    \"\"\"\n    local_dir must contain:\n      - config.json\n      - preprocessor_config.json\n      - pytorch_model.bin\n\n    Returns (model, preprocess, device).\n    \"\"\"\n    device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n\n    # You instantiate processor here mostly for correctness, e.g. size info.\n    # You don't strictly need to return it since we build our own torchvision pipeline.\n    _ = AutoImageProcessor.from_pretrained(str(local_dir))\n\n    model = AutoModel.from_pretrained(str(local_dir))\n    model = model.eval().to(device)\n\n    preprocess = T.Compose([\n        T.Resize(518, interpolation=T.InterpolationMode.BICUBIC),\n        T.CenterCrop(518),\n        T.ToTensor(),\n        T.Normalize(mean=(0.485, 0.456, 0.406),\n                    std=(0.229, 0.224, 0.225)),\n    ])\n    \n    return model, preprocess, device\n\n\nmodel, preprocess, device = load_dinov2_from_dir(DINOV2_DIR)\nprint(\"Model loaded on:\", device)\n\n\n@torch.no_grad()\ndef embed_image(\n    img_path: str,\n    model,\n    preprocess,\n    device: str = \"cpu\",\n    l2_normalize: bool = True\n) -> np.ndarray:\n    \"\"\"\n    Load image from img_path, force RGB (so RGBA / grayscale won't break),\n    preprocess -> model -> 1D descriptor, L2-normalize, return np.float32.\n\n    This is robust to:\n    - alpha channel PNGs\n    - grayscale images\n    - slightly truncated PNGs\n    - HF model outputs that aren't just raw tensors\n    \"\"\"\n    # 1. Load image and force exactly 3 channels\n    img = Image.open(img_path).convert(\"RGB\")\n\n    # 2. Preprocess into model input tensor\n    x = preprocess(img).unsqueeze(0).to(device)  # [1, 3, H, W]\n\n    # 3. Forward pass\n    feats = model(x)\n\n    # 4. Handle different return types:\n    #    - Some HF vision models return BaseModelOutput with .last_hidden_state\n    #    - Some return just a tensor\n    if isinstance(feats, dict):\n        # e.g. {'last_hidden_state': ..., 'pooler_output': ...}\n        if \"pooler_output\" in feats and feats[\"pooler_output\"] is not None:\n            tensor_out = feats[\"pooler_output\"]          # [1, D]\n        elif \"last_hidden_state\" in feats:\n            # take CLS token [0,0,:] or mean pool\n            lh = feats[\"last_hidden_state\"]              # [1, seq, dim]\n            tensor_out = lh.mean(dim=1)                  # simple global pool -> [1, D]\n        else:\n            # fallback: grab first value in dict\n            tensor_out = list(feats.values())[0]\n    else:\n        # Could be a tuple or a plain tensor\n        if isinstance(feats, (list, tuple)):\n            tensor_out = feats[0]\n        else:\n            tensor_out = feats\n\n    # 5. Flatten to shape [1, D] if needed\n    if tensor_out.ndim > 2:\n        # e.g. [1, C, h, w] -> global average pool then flatten\n        tensor_out = tensor_out.mean(dim=list(range(2, tensor_out.ndim)))\n        # now [1, C]\n\n    # 6. Squeeze batch to get [D]\n    vec = tensor_out.squeeze(0).detach().cpu().numpy().astype(np.float32)\n\n    # 7. Optional L2 norm\n    if l2_normalize:\n        denom = np.linalg.norm(vec) + 1e-12\n        vec = vec / denom\n\n    return vec\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:03:46.389040Z","iopub.execute_input":"2025-10-24T09:03:46.389239Z","iopub.status.idle":"2025-10-24T09:04:20.220511Z","shell.execute_reply.started":"2025-10-24T09:03:46.389224Z","shell.execute_reply":"2025-10-24T09:04:20.219721Z"}},"outputs":[],"execution_count":null},{"id":"ecaceb68-0aac-4675-bfb3-8f3bb5b336ec","cell_type":"code","source":"from pathlib import Path\nimport csv\nimport numpy as np\n\nIMAGE_EXTS = {\".jpg\", \".jpeg\", \".png\", \".bmp\", \".webp\", \".tiff\"}\n\ndef list_images_under(roots):\n    imgs = []\n    for root in roots:\n        r = Path(root)\n        if not r.exists():\n            print(f\"[WARN] Missing: {r}\")\n            continue\n        for p in r.rglob(\"*\"):\n            if p.is_file() and p.suffix.lower() in IMAGE_EXTS:\n                imgs.append(p)\n    return imgs\n\ndef build_vectors_csv(\n    roots=(\"test_resized\",),\n    l2_normalize=True,\n    out_csv=\"test_vectors.csv\",\n):\n    all_imgs = list_images_under(roots)\n    print(f\"Found {len(all_imgs)} images under {roots}.\")\n\n    with open(out_csv, \"w\", newline=\"\", encoding=\"utf-8\") as f:\n        writer = csv.writer(f)\n        writer.writerow([\"folder\", \"file\", \"vector\"])  # header\n\n        for p in all_imgs:\n            try:\n                vec = embed_image(str(p), model, preprocess, device, l2_normalize=l2_normalize)\n                folder = p.parent.name\n                file = p.stem\n                # store vector as string of floats\n                writer.writerow([folder, file, vec.tolist()])\n            except Exception as e:\n                print(f\"[ERROR] {p}: {e}\")\n                continue\n\n    print(f\"[OK] Saved CSV: {out_csv} with {len(all_imgs)} rows\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:04:20.221478Z","iopub.execute_input":"2025-10-24T09:04:20.221731Z","iopub.status.idle":"2025-10-24T09:04:20.228903Z","shell.execute_reply.started":"2025-10-24T09:04:20.221703Z","shell.execute_reply":"2025-10-24T09:04:20.228062Z"}},"outputs":[],"execution_count":null},{"id":"a3b990f8-4e84-40a3-acda-cc247f871cb0","cell_type":"markdown","source":"## Model","metadata":{}},{"id":"8bf34a16-9bde-45f1-bd5f-d0a4728c2e4d","cell_type":"code","source":"from tqdm.auto import tqdm    # nice in-notebook progress bars\ntqdm.pandas()                 # adds .progress_apply if you need it","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:04:20.229728Z","iopub.execute_input":"2025-10-24T09:04:20.229990Z","iopub.status.idle":"2025-10-24T09:04:20.243404Z","shell.execute_reply.started":"2025-10-24T09:04:20.229948Z","shell.execute_reply":"2025-10-24T09:04:20.242706Z"}},"outputs":[],"execution_count":null},{"id":"64538c28-8208-4c1b-b0f1-915ed77accd3","cell_type":"code","source":"# === IMC2025 — MASt3R-style clustering for Task 1 (scene grouping) ===\n# With tqdm progress bars\nimport os, math, cv2, numpy as np, pandas as pd, networkx as nx\nfrom typing import Optional, List, Tuple, Dict\nfrom tqdm.auto import tqdm\ntqdm.pandas()  # optional: enables .progress_apply on pandas\n\n# ---------- file resolver ----------\n_EXTS = [\".png\",\".jpg\",\".jpeg\",\".JPG\",\".PNG\",\".JPEG\",\".webp\",\".bmp\",\".tif\",\".tiff\"]\ndef resolve_path(root: str, folder: str, fname: str) -> Optional[str]:\n    base = os.path.join(root, folder, fname)\n    if os.path.exists(base): return base\n    for e in _EXTS:\n        p = base + e\n        if os.path.exists(p): return p\n    return None\n\n# ---------- global retrieval (cosine on your embeddings) ----------\ndef shortlist_indices(X: np.ndarray, k: int) -> Tuple[np.ndarray, np.ndarray]:\n    # X assumed L2-normalized\n    sims = X @ X.T                              # cosine similarity\n    np.fill_diagonal(sims, -1.0)\n    idx = np.argpartition(-sims, kth=np.minimum(k, X.shape[0]-1), axis=1)[:, :k]\n    # keep strongest neighbors sorted desc\n    row = np.arange(X.shape[0])[:, None]\n    s = sims[row, idx]\n    order = np.argsort(-s, axis=1)\n    return idx[row, order], sims\n\n# ---------- geometric verification backends ----------\ndef _try_mast3r():\n    try:\n        # If you have MASt3R installed, adapt the import/API below to your env.\n        from mast3r.matcher import FastMatch  # placeholder; adjust to your local import\n        return FastMatch()\n    except Exception:\n        return None\n\n_MAST3R = _try_mast3r()\n\n# ORB fallback (fast, CPU)\n_ORB = cv2.ORB_create(nfeatures=2000)\ndef _orb_desc(gray: np.ndarray):\n    return _ORB.detectAndCompute(gray, None)\n\ndef _load_gray(path: str, max_side: int = 960) -> Optional[np.ndarray]:\n    if path is None: return None\n    img = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n    if img is None: return None\n    h, w = img.shape[:2]; m = max(h, w)\n    if m > max_side:\n        s = max_side / m\n        img = cv2.resize(img, (int(w*s), int(h*s)), interpolation=cv2.INTER_AREA)\n    return img\n\ndef verify_pair_mast3r(pA: str, pB: str, mast3r, max_side=960) -> Tuple[int, float]:\n    # returns (n_inliers_like, score) — adapt to your mast3r binding\n    try:\n        n_inliers, score = mast3r.match(pA, pB, max_side=max_side)  # pseudo-API\n        return int(n_inliers), float(score)\n    except Exception:\n        return 0, 0.0\n\ndef _robust_fundamental(ptsA, ptsB, ransac_px=2.5):\n    \"\"\"Try USAC_MAGSAC, then FM_RANSAC. Returns (n_inliers, mask) or (0, None).\"\"\"\n    # OpenCV likes float64 here\n    A = np.asarray(ptsA, np.float64)\n    B = np.asarray(ptsB, np.float64)\n    if len(A) < 8:\n        return 0, None\n    # basic sanity: remove exact duplicates\n    AB = np.hstack([A, B])\n    _, uniq_idx = np.unique(AB, axis=0, return_index=True)\n    A = A[uniq_idx]; B = B[uniq_idx]\n    if len(A) < 8:\n        return 0, None\n    # spread check (avoid all points on a pixel or line)\n    if (A.std(axis=0).max() < 1e-3) or (B.std(axis=0).max() < 1e-3):\n        return 0, None\n    # 1) USAC_MAGSAC\n    try:\n        F, m = cv2.findFundamentalMat(A, B, cv2.USAC_MAGSAC, float(max(0.5, ransac_px)), 0.999, 2000)\n        if F is not None and m is not None:\n            n_in = int(m.sum())\n            if n_in >= 8:\n                return n_in, m\n    except cv2.error:\n        pass\n    # 2) FM_RANSAC (classic)\n    try:\n        F, m = cv2.findFundamentalMat(A, B, cv2.FM_RANSAC, float(max(0.5, ransac_px)), 0.999)\n        if F is not None and m is not None:\n            n_in = int(m.sum())\n            if n_in >= 8:\n                return n_in, m\n    except cv2.error:\n        pass\n    return 0, None\n\ndef _robust_homography(ptsA, ptsB, ransac_px=3.0):\n    \"\"\"Homography fallback for near-planar pairs.\"\"\"\n    A = np.asarray(ptsA, np.float64)\n    B = np.asarray(ptsB, np.float64)\n    if len(A) < 4:\n        return 0, None\n    try:\n        H, m = cv2.findHomography(A, B, cv2.USAC_MAGSAC, float(max(0.5, ransac_px)), 0.999, 2000)\n        if H is not None and m is not None:\n            return int(m.sum()), m\n    except cv2.error:\n        pass\n    try:\n        H, m = cv2.findHomography(A, B, cv2.RANSAC, float(max(0.5, ransac_px)))\n        if H is not None and m is not None:\n            return int(m.sum()), m\n    except cv2.error:\n        pass\n    return 0, None\n\ndef verify_pair_orb(pA: str, pB: str, ratio=0.8, ransac_px=2.5, max_side=960) -> Tuple[int, float]:\n    A = _load_gray(pA, max_side=max_side); B = _load_gray(pB, max_side=max_side)\n    if A is None or B is None: \n        return 0, 0.0\n    kpA, desA = _ORB.detectAndCompute(A, None)\n    kpB, desB = _ORB.detectAndCompute(B, None)\n    if desA is None or desB is None or len(kpA) < 8 or len(kpB) < 8:\n        return 0, 0.0\n\n    # KNN + ratio\n    bf = cv2.BFMatcher(cv2.NORM_HAMMING)\n    knn = bf.knnMatch(desA, desB, k=2)\n    good = []\n    for pair in knn:\n        if len(pair) != 2: \n            continue\n        m, n = pair\n        if m.distance < ratio * n.distance:\n            good.append(m)\n    if len(good) < 8:\n        return 0, 0.0\n\n    # dedup by query AND train\n    best_by_q = {}\n    for m in good:\n        q = m.queryIdx\n        if (q not in best_by_q) or (m.distance < best_by_q[q].distance):\n            best_by_q[q] = m\n    # also dedup by train to avoid many-to-one\n    best_by_t = {}\n    for m in best_by_q.values():\n        t = m.trainIdx\n        if (t not in best_by_t) or (m.distance < best_by_t[t].distance):\n            best_by_t[t] = m\n    good = list(best_by_t.values())\n    if len(good) < 8:\n        return 0, 0.0\n\n    ptsA = np.float32([kpA[m.queryIdx].pt for m in good])\n    ptsB = np.float32([kpB[m.trainIdx].pt for m in good])\n\n    # fundamental with robust fallbacks\n    try:\n        nF, _ = _robust_fundamental(ptsA, ptsB, ransac_px=ransac_px)\n    except Exception:\n        nF = 0\n    if nF >= 8:\n        # geometric score in [0,1], cap at 80 inliers\n        return nF, min(1.0, nF / 80.0)\n\n    # homography fallback (planar scenes, walls, posters, etc.)\n    try:\n        nH, _ = _robust_homography(ptsA, ptsB, ransac_px=max(3.0, ransac_px))\n    except Exception:\n        nH = 0\n    if nH >= 10:\n        return nH, min(1.0, nH / 80.0)\n\n    return 0, 0.0\n\ndef verify_pair(pA: str, pB: str, use_mast3r=True, **kw) -> Tuple[int, float]:\n    if use_mast3r and (_MAST3R is not None):\n        n_in, score = verify_pair_mast3r(pA, pB, _MAST3R, max_side=kw.get(\"max_side\", 960))\n        if n_in > 0:\n            return n_in, min(1.0, score)\n    return verify_pair_orb(pA, pB,\n                           ratio=kw.get(\"ratio\", 0.8),\n                           ransac_px=kw.get(\"ransac_px\", 2.5),\n                           max_side=kw.get(\"max_side\", 960))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:04:20.244196Z","iopub.execute_input":"2025-10-24T09:04:20.244451Z","iopub.status.idle":"2025-10-24T09:04:20.436437Z","shell.execute_reply.started":"2025-10-24T09:04:20.244426Z","shell.execute_reply":"2025-10-24T09:04:20.435706Z"}},"outputs":[],"execution_count":null},{"id":"d50501b7-8a50-4971-a85a-5357398197db","cell_type":"code","source":"def cluster_mast3r_style(\n    df_vectors: pd.DataFrame,\n    folder_id: str,\n    image_root: str,\n    # retrieval\n    k_retrieve: int = 80,\n    t_accept: float = 0.85,\n    t_verify: float = 0.25,\n    t_reject: float = 0.10,\n    # geometry\n    use_mast3r: bool = True,\n    inlier_min: int = 12,\n    max_side: int = 960,\n    ratio: float = 0.80,\n    ransac_px: float = 2.5,\n    # graph + clustering\n    alpha: float = 0.8,\n    min_degree: int = 1,\n    use_mutual: bool = True,\n    per_node_cap: int = 8,\n    # NEW: post rules\n    rel_drop_ratio: float = 0.60,     # drop clusters ≤ 0.60 * largest\n    pop_drop_ratio: float = 0.10,     # drop clusters ≤ 0.10 * population\n    # NEW: outlier rescue\n    rescue_outliers: bool = True,\n    rescue_top_clusters: int = 3,     # search refs only in top-K clusters\n    topk_refs: int = 10,              # per candidate outlier, per cluster\n    rescue_inlier_avg: int = 20,      # avg inliers threshold\n    rescue_success_frac: float = 0.30,# fraction of refs with >0 inliers\n    # progress\n    show_progress: bool = True\n) -> pd.DataFrame:\n\n    sub = df_vectors[df_vectors[\"folder\"] == folder_id].reset_index(drop=True)\n    n = len(sub)\n    if n == 0:\n        return pd.DataFrame(columns=[\"folder\",\"file\",\"cluster_label\"])\n    if n == 1:\n        out = sub[[\"folder\",\"file\"]].copy(); out[\"cluster_label\"] = 0; return out\n\n#    if n <= 5 : \n#        return out\n\n    # --- vectors (L2) ---\n    X = np.stack(sub[\"vector\"].to_list()).astype(\"float32\")\n    X /= np.maximum(1e-8, np.linalg.norm(X, axis=1, keepdims=True))\n    k = max(1, min(k_retrieve, n-1))\n\n    nbrs, sims_full = shortlist_indices(X, k)\n    sims_full = np.clip(sims_full, -1.0, 1.0)\n    cos01 = 0.5 * (sims_full + 1.0)  # map to [0,1]\n\n    # paths\n    paths = [resolve_path(image_root, folder_id, f) for f in sub[\"file\"]]\n\n    # ------- candidate pairs (only j>i; optional mutual) -------\n    pairs = []\n    if use_mutual:\n        # build a quick set view for mutual test\n        nbr_sets = [set(map(int, nbrs[i].tolist())) for i in range(n)]\n    for i in range(n):\n        for j_raw in nbrs[i]:\n            j = int(j_raw)\n            if j <= i: \n                continue\n            if use_mutual and (i not in nbr_sets[j]): \n                continue\n            pairs.append((i, j, float(cos01[i, j])))\n\n    pbar = tqdm(total=len(pairs), desc=\"build+verify pairs\", unit=\"pair\",\n                disable=not show_progress)\n\n    # collect edges per node to cap degree later\n    edges_by_node: Dict[int, List[Tuple[int,float]]] = {u: [] for u in range(n)}\n\n    for i, j, c in pairs:\n        if c >= t_accept:\n            w = alpha*c + (1-alpha)*1.0\n            edges_by_node[i].append((j, w)); edges_by_node[j].append((i, w))\n            pbar.update(1); continue\n        if c < t_reject:\n            pbar.update(1); continue\n\n        n_in, g = verify_pair(paths[i], paths[j], use_mast3r=use_mast3r,\n                              max_side=max_side, ratio=ratio, ransac_px=ransac_px)\n        if n_in >= inlier_min:\n            w = alpha*c + (1-alpha)*g\n            edges_by_node[i].append((j, w)); edges_by_node[j].append((i, w))\n        pbar.update(1)\n\n    pbar.close()\n\n    # ------- per-node cap (keep strongest M) -------\n    G = nx.Graph(); G.add_nodes_from(range(n))\n    if per_node_cap and per_node_cap > 0:\n        for u in range(n):\n            if not edges_by_node[u]: continue\n            edges_by_node[u].sort(key=lambda t: -t[1])\n            edges_by_node[u] = edges_by_node[u][:per_node_cap]\n\n    for u in range(n):\n        for v, w in edges_by_node[u]:\n            a, b = (u, v) if u < v else (v, u)\n            if G.has_edge(a, b):\n                if w > G[a][b]['weight']: G[a][b]['weight'] = float(w)\n            else:\n                G.add_edge(a, b, weight=float(max(1e-6, w)))\n\n    # ------- connected components → labels -------\n    if G.number_of_edges() == 0:\n        labels = -1 * np.ones(n, np.int32)\n    else:\n        comps = list(nx.connected_components(G))\n        labels = -1 * np.ones(n, np.int32)\n        for cid, nodes in enumerate(comps):\n            for u in nodes: labels[u] = cid\n\n    # degree-based outliers\n    deg = np.array([d for _, d in G.degree()])\n    labels[deg < max(0, min_degree)] = -1\n\n    # ------- POST-RULE: drop clusters by size (absolute or relative) -------\n    valid = labels != -1\n    if valid.any():\n        # (a) absolute 10% of population\n        abs_thresh = int(np.floor( pop_drop_ratio * n))  # <= 10% of total\n        vals, counts = np.unique(labels[valid], return_counts=True)\n        # largest cluster size\n        Nmax = counts.max() if counts.size > 0 else 0\n        # (b) relative ≤ 60% of the largest\n        rel_thresh = int(np.floor(rel_drop_ratio * Nmax))\n\n        to_drop = set()\n        for v, c in zip(vals, counts):\n            if c <= abs_thresh or c <= rel_thresh:\n                to_drop.add(v)\n\n        if to_drop:\n            labels[np.isin(labels, list(to_drop))] = -1\n\n    # ------- outlier rescue (fast, top clusters only) -------\n    if rescue_outliers:\n        labels = _rescue_outliers_topk(\n            labels, X, cos01, paths,\n        )\n\n    # -------- Safe recursive version --------\n    outliers_file = sub.loc[labels == -1, \"file\"]\n    params_fixed = dict(\n                    k_retrieve=50,\n                    t_accept=0.85, t_verify=0.50, t_reject=0.10,\n                    use_mast3r=True, inlier_min=20, max_side=960, ratio=0.80, ransac_px=2.5,\n                    alpha=0.7, min_degree=2,\n                    use_mutual=True, per_node_cap=8,\n                    # post-rule: drop clusters ≤ 60% of largest\n                    rel_drop_ratio=0.6,\n                    pop_drop_ratio=0.1,\n                    # one-pass outlier rescue against top-K largest clusters\n                    rescue_outliers=True, rescue_top_clusters=5,\n                    topk_refs=20, rescue_inlier_avg=20, rescue_success_frac=0.30,\n                    show_progress=True\n                )\n    print(\"Size of outlier cluster: \" , len(outliers_file) )\n    print(\"Size of parent dataset: \" , n )\n\n    if len(outliers_file) == n:\n        out = sub[[\"folder\", \"file\"]].copy()\n        out[\"cluster_label\"] = -1\n        print(f\"[warn] All {n} images are outliers for folder '{folder_id}' — no clusters formed.\")\n        return out\n    \n    if len(outliers_file) >= 0.5 * n:\n        print(\"Outlier group is larger than 50% of population --> Clustering this seperated cluster !!!\")\n        outliers_group = df_vectors[df_vectors[\"file\"].isin(outliers_file)].reset_index(drop=True)\n        \n        df_cluster_out = cluster_mast3r_style(outliers_group, folder_id, image_root, **params_fixed)\n        outliers_group = df_vectors[df_vectors[\"file\"].isin(\n            df_cluster_out.loc[df_cluster_out[\"cluster_label\"] == -1, \"file\"]\n        )].reset_index(drop=True)\n    \n        # Merge new clusters back to main df\n        if (df_cluster_out[\"cluster_label\"] != -1).any():\n            offset = labels[labels != -1].max() + 1 if (labels != -1).any() else 0\n            for _, row in df_cluster_out.iterrows():\n                if row[\"cluster_label\"] != -1:\n                    labels[sub[\"file\"] == row[\"file\"]] = offset + row[\"cluster_label\"]\n\n        if rescue_outliers:\n            remaining_outliers = (labels == -1).sum()\n            if remaining_outliers > 0:\n                labels = _rescue_outliers_topk(labels, X, cos01, paths,\n                                               sim_high_thr = 0.8, sim_mid_low = 0.7, \n                                               inlier_cond = 20, success_cond = 0.50 )\n            \n    elif len(outliers_file) >= 0.3 * n:\n        print(\"Outlier group is between 30% and 50% of population  --> Clustering this seperated cluster !!!\")\n        outliers_group = df_vectors[df_vectors[\"file\"].isin(outliers_file)].reset_index(drop=True)\n        \n        df_cluster_out = cluster_mast3r_style(outliers_group, folder_id, image_root, **params_fixed)\n        outliers_group = df_vectors[df_vectors[\"file\"].isin(\n            df_cluster_out.loc[df_cluster_out[\"cluster_label\"] == -1, \"file\"]\n        )].reset_index(drop=True)\n    \n        # Merge new clusters back to main df\n        if (df_cluster_out[\"cluster_label\"] != -1).any():\n            offset = labels[labels != -1].max() + 1 if (labels != -1).any() else 0\n            for _, row in df_cluster_out.iterrows():\n                if row[\"cluster_label\"] != -1:\n                    labels[sub[\"file\"] == row[\"file\"]] = offset + row[\"cluster_label\"]\n\n        if rescue_outliers:\n            remaining_outliers = (labels == -1).sum()\n            if remaining_outliers > 0:\n                labels = _rescue_outliers_topk(labels, X, cos01, paths,\n                                               sim_high_thr = 0.8, sim_mid_low = 0.65, \n                                               inlier_cond = 9, success_cond = 0.50 )\n\n    out = sub[[\"folder\",\"file\"]].copy()\n    out[\"cluster_label\"] = labels\n    return out","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:04:20.438576Z","iopub.execute_input":"2025-10-24T09:04:20.439203Z","iopub.status.idle":"2025-10-24T09:04:20.462601Z","shell.execute_reply.started":"2025-10-24T09:04:20.439184Z","shell.execute_reply":"2025-10-24T09:04:20.461793Z"}},"outputs":[],"execution_count":null},{"id":"8d09c5d5-1996-4a9b-95ab-12c9a6b58667","cell_type":"code","source":"def _rescue_outliers_topk(\n    labels: np.ndarray,\n    X: np.ndarray,\n    cos01: np.ndarray,\n    paths: List[Optional[str]],\n    rescue_top_clusters: int = 3,\n    topk_refs: int = 10,\n    # legacy knobs (not used by the new rule but kept for signature compatibility)\n    inlier_avg_thr: int = 20,\n    success_frac_thr: float = 0.30,\n    # geometry params\n    ratio: float = 0.80,\n    ransac_px: float = 2.5,\n    max_side: int = 960,\n    # --- NEW rule thresholds (tunable) ---\n    sim_high_thr: float = 0.80,\n    sim_mid_low: float = 0.70,\n    inlier_cond: float = 9.8,\n    success_cond: float = 0.50,\n    verbose: bool = False,\n) -> np.ndarray:\n    \"\"\"\n    Relabel selected outliers into best cluster using your two-stage rule:\n      1) avg_sim >= sim_high_thr -> choose highest avg_sim (tie by avg_inliers)\n      2) else among mid band [sim_mid_low, sim_high_thr):\n           require avg_inliers >= inlier_cond and success_frac >= success_cond\n           choose highest avg_sim (tie by avg_inliers)\n      3) else keep as -1\n    \"\"\"\n    \n    n = len(labels)\n    out = labels.copy()\n\n    # which nodes are in some cluster\n    valid = out != -1\n    if not valid.any():\n        return out\n\n    # top-K (largest) clusters\n    vals, counts = np.unique(out[valid], return_counts=True)\n    order = np.argsort(-counts)\n    top_clusters = vals[order[:rescue_top_clusters]]\n    if len(top_clusters) == 0:\n        return out\n\n    # members dict for quick retrieval (updated when we reassign)\n    members = {c: np.where(out == c)[0] for c in top_clusters}\n\n    # precompute once\n    # X is L2-normalized; cos01 already provided\n    for i in np.where(out == -1)[0]:\n        # collect per-candidate stats\n        cand_stats = []  # list of (cluster_id, avg_sim, avg_inliers, success_frac)\n\n        for c in top_clusters:\n            idxs = members.get(c, None)\n            if idxs is None or idxs.size == 0:\n                continue\n\n            # top-k most similar refs from this cluster\n            sims = cos01[i, idxs]\n            take = min(topk_refs, idxs.size)\n            order_refs = np.argsort(-sims)[:take]\n            ref_idx = idxs[order_refs]\n            ref_sims = sims[order_refs]\n\n            if take == 0:\n                continue\n\n            # run geometry to compute inliers\n            inliers = []\n            p_i = paths[i]\n            if p_i is None:\n                continue\n            for j in ref_idx:\n                p_j = paths[j]\n                if p_j is None:\n                    inliers.append(0)\n                    continue\n                n_in, _ = verify_pair(\n                    p_i, p_j,\n                    use_mast3r=False,         # keep ORB+MAGSAC here for speed & stability\n                    max_side=max_side,\n                    ratio=ratio,\n                    ransac_px=ransac_px\n                )\n                inliers.append(int(max(0, n_in)))\n\n            if len(inliers) == 0:\n                continue\n\n            avg_sim     = float(np.mean(ref_sims))\n            avg_inliers = float(np.mean(inliers))\n            success_frac = float(np.mean([x > 0 for x in inliers]))\n\n            cand_stats.append((c, avg_sim, avg_inliers, success_frac))\n\n            if verbose:\n                print(f\"[rescue] outlier idx={i} vs cluster {c}: \"\n                      f\"avg_sim={avg_sim:.3f}, avg_in={avg_inliers:.2f}, succ={success_frac:.2f}, \"\n                      f\"size={len(idxs)}\")\n\n        if not cand_stats:\n            continue\n\n        # --- Apply your selection rule ---\n        # 1) High similarity bucket\n        high = [t for t in cand_stats if t[1] >= sim_high_thr]\n        choice = None\n        if high:\n            # pick by highest avg_sim, tie by avg_inliers\n            high.sort(key=lambda x: (x[1], x[2]), reverse=True)\n            choice = high[0]\n        else:\n            # 2) Mid similarity with geometric evidence\n            mid = [\n                t for t in cand_stats\n                if (sim_mid_low <= t[1] < sim_high_thr) and (t[2] >= inlier_cond) and (t[3] >= success_cond)\n            ]\n            if mid:\n                mid.sort(key=lambda x: (x[1], x[2]), reverse=True)\n                choice = mid[0]\n\n        # 3) If a choice exists, reassign this outlier\n        if choice is not None:\n            chosen_cluster = int(choice[0])\n            out[i] = chosen_cluster\n            # update members so subsequent outliers can also consider this\n            if chosen_cluster in members:\n                members[chosen_cluster] = np.append(members[chosen_cluster], i)\n            else:\n                members[chosen_cluster] = np.array([i], dtype=int)\n\n            if verbose:\n                _, asim, ain, sfrac = choice\n                print(f\"[rescue] ✅ reassigned idx={i} to cluster {chosen_cluster} \"\n                      f\"(avg_sim={asim:.3f}, avg_in={ain:.1f}, succ={sfrac:.2f})\")\n\n    return out","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:04:20.463727Z","iopub.execute_input":"2025-10-24T09:04:20.463996Z","iopub.status.idle":"2025-10-24T09:04:20.479583Z","shell.execute_reply.started":"2025-10-24T09:04:20.463964Z","shell.execute_reply":"2025-10-24T09:04:20.478911Z"}},"outputs":[],"execution_count":null},{"id":"a7843907-0284-434e-9dfd-4a400468ef22","cell_type":"markdown","source":"## Submission Output","metadata":{}},{"id":"6bffb7bc-cee9-4127-bf2c-329cd88800d5","cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom typing import Dict, List\n\n_IMAGE_EXTS = {\".png\", \".jpg\", \".jpeg\", \".bmp\", \".webp\", \".tif\", \".tiff\",\n               \".PNG\", \".JPG\", \".JPEG\"}\n\ndef build_base_from_test_root(test_root: str) -> pd.DataFrame:\n    \"\"\"Scan test_root/<dataset>/* and return ['dataset','image'].\"\"\"\n    rows = []\n    for dataset in sorted(os.listdir(test_root)):\n        dpath = os.path.join(test_root, dataset)\n        if not os.path.isdir(dpath):\n            continue\n        for fn in sorted(os.listdir(dpath)):\n            _, ext = os.path.splitext(fn)\n            if ext in _IMAGE_EXTS:\n                rows.append((dataset, fn))\n    base = pd.DataFrame(rows, columns=[\"dataset\",\"image\"])\n    if base.empty:\n        raise RuntimeError(f\"No images found under: {test_root}\")\n    return base\n\n\ndef build_submission(\n    test_root: str,\n    # Map: dataset(folder) -> dataframe with columns ['file','cluster_label']\n    predictions_by_dataset: Dict[str, pd.DataFrame],\n    output_csv: str\n) -> pd.DataFrame:\n    \"\"\"\n    - Scans test_root.\n    - Left-joins per-dataset predictions by filename (handles missing extensions).\n    - Preserves your labels exactly, including -1.\n    - scene = f\"cluster_{label}\"\n    - image_id = image + \"_private\"\n    \"\"\"\n    base = build_base_from_test_root(test_root)  # ['dataset','image']\n\n    all_rows: List[pd.DataFrame] = []\n    for ds, sub_df in base.groupby(\"dataset\"):\n        sub = sub_df.copy()\n\n        pred = predictions_by_dataset.get(ds, None)\n        if pred is None or pred.empty:\n            sub[\"cluster_label\"] = -1\n        else:\n            p = pred.copy()\n            if \"file\" not in p.columns or \"cluster_label\" not in p.columns:\n                raise ValueError(f\"Predictions for '{ds}' must have ['file','cluster_label']\")\n\n            # normalize names for merge\n            p[\"file_noext\"] = p[\"file\"].apply(lambda s: os.path.splitext(str(s))[0])\n            sub[\"image_noext\"] = sub[\"image\"].apply(lambda s: os.path.splitext(str(s))[0])\n\n            sub = sub.merge(\n                p[[\"file_noext\", \"cluster_label\"]],\n                left_on=\"image_noext\", right_on=\"file_noext\",\n                how=\"left\"\n            )\n\n            sub[\"cluster_label\"] = sub[\"cluster_label\"].fillna(-1).astype(int)\n            sub.drop(columns=[\"image_noext\",\"file_noext\"], inplace=True)\n\n        all_rows.append(sub)\n\n    merged = pd.concat(all_rows, ignore_index=True)\n\n    # Preserve labels exactly\n    merged[\"scene\"] = merged[\"cluster_label\"].apply(lambda v: f\"cluster_{int(v)}\")\n\n    # image_id per rule\n    merged[\"image_id\"] = merged[\"image\"].astype(str) + \"_private\"\n\n    # Final submission columns\n    subm = merged[[\"image_id\",\"dataset\",\"scene\",\"image\"]].copy()\n\n    os.makedirs(os.path.dirname(output_csv) or \".\", exist_ok=True)\n    subm.to_csv(output_csv, index=False)\n    return subm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:04:20.480347Z","iopub.execute_input":"2025-10-24T09:04:20.480893Z","iopub.status.idle":"2025-10-24T09:04:20.494989Z","shell.execute_reply.started":"2025-10-24T09:04:20.480866Z","shell.execute_reply":"2025-10-24T09:04:20.494225Z"}},"outputs":[],"execution_count":null},{"id":"0e0952ef-5d56-4bf1-9296-74b13fd26053","cell_type":"markdown","source":"# Pipeline","metadata":{}},{"id":"c5f0945c-a5a3-4759-b7e3-ac8eb8982e6f","cell_type":"code","source":"resize_with_subfolders(\"/kaggle/input/image-matching-challenge-2025/test\", \"/kaggle/working/test_resized\")\nprint(\"Resize Done\")\n\n# ---- Run it\nbuild_vectors_csv( roots=(\"/kaggle/working/test_resized\",),l2_normalize=True, out_csv=\"/kaggle/working/test_vectors.csv\")\nprint(\"Converting Vector Done\")\n\ndf_vectors = pd.read_csv(\"/kaggle/working/test_vectors.csv\")   # folder, file, vector\n\ndef parse_vec(x):\n    if isinstance(x, (list, np.ndarray)):\n        return np.asarray(x, dtype=np.float32)\n    try:\n        v = np.asarray(ast.literal_eval(x), dtype=np.float32)\n    except Exception:\n        # fallback: remove brackets and split by space or comma\n        x = x.strip(\"[] \")\n        parts = [float(p) for p in x.replace(\",\", \" \").split()]\n        v = np.asarray(parts, dtype=np.float32)\n    n = np.linalg.norm(v)\n    return v / n if n > 0 else v\n\ndf_vectors[\"vector\"] = df_vectors[\"vector\"].apply(parse_vec)\n\nlist_of_dataset = list( df_vectors['folder'].unique() )\n\nfor idx in range( len(list_of_dataset) ):\n    if list_of_dataset[idx][-7:] == \"resized\" :\n        list_of_dataset[idx] = list_of_dataset[idx][:-8]\n    else: \n        list_of_dataset.pop(idx)\n        idx -= 1\n\npred = {}\n\nimage_root = \"/kaggle/working/test_resized\"\nparams = dict(\n    k_retrieve=50,\n    t_accept=0.85, t_verify=0.50, t_reject=0.10,\n    use_mast3r=True, inlier_min=20, max_side=960, ratio=0.80, ransac_px=2.5,\n    alpha=0.7, min_degree=2,\n    use_mutual=True, per_node_cap=8,\n    # post-rule: drop clusters ≤ 60% of largest\n    rel_drop_ratio=0.6,\n    pop_drop_ratio=0.1,\n    # one-pass outlier rescue against top-K largest clusters\n    rescue_outliers=True, rescue_top_clusters=5,\n    topk_refs=20, rescue_inlier_avg=20, rescue_success_frac=0.30,\n    show_progress=True\n)\n\nfor dataset in list_of_dataset:\n    folder = dataset + \"_resized\"\n    df_cluster_one = cluster_mast3r_style(df_vectors, folder, image_root, **params)\n    pred[dataset] = df_cluster_one\n\nprint(\"Clustering Done\")\n\n\n# 2) Build submission\ntest_root = \"/kaggle/input/image-matching-challenge-2025/test\"   # top-level directory with subfolders per dataset\nsubmission = build_submission(test_root, pred, \"submission.csv\")\nprint(\"Output Done\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:04:20.495830Z","iopub.execute_input":"2025-10-24T09:04:20.496126Z","iopub.status.idle":"2025-10-24T09:06:42.684975Z","shell.execute_reply.started":"2025-10-24T09:04:20.496110Z","shell.execute_reply":"2025-10-24T09:06:42.684295Z"}},"outputs":[],"execution_count":null},{"id":"8430b07b-1fe3-4455-9416-88760cbdc9e6","cell_type":"code","source":"pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:06:42.685651Z","iopub.execute_input":"2025-10-24T09:06:42.685906Z","iopub.status.idle":"2025-10-24T09:06:42.696627Z","shell.execute_reply.started":"2025-10-24T09:06:42.685887Z","shell.execute_reply":"2025-10-24T09:06:42.696069Z"}},"outputs":[],"execution_count":null},{"id":"2ad62ad7-a445-453b-99f6-1fa3b40801f2","cell_type":"code","source":"df = pd.read_csv(\"/kaggle/working/test_vectors.csv\")\ndf.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:06:42.697391Z","iopub.execute_input":"2025-10-24T09:06:42.697789Z","iopub.status.idle":"2025-10-24T09:06:42.724830Z","shell.execute_reply.started":"2025-10-24T09:06:42.697754Z","shell.execute_reply":"2025-10-24T09:06:42.724199Z"}},"outputs":[],"execution_count":null},{"id":"659fad71-15ef-4784-ace6-0fb002f0384e","cell_type":"code","source":"test_root = \"/kaggle/input/image-matching-challenge-2025/test\"   # top-level directory with subfolders per dataset\nsubmission = build_submission(test_root, pred, \"submission.csv\")\nprint(\"Output Done\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:06:42.725419Z","iopub.execute_input":"2025-10-24T09:06:42.725582Z","iopub.status.idle":"2025-10-24T09:06:42.739937Z","shell.execute_reply.started":"2025-10-24T09:06:42.725569Z","shell.execute_reply":"2025-10-24T09:06:42.739332Z"}},"outputs":[],"execution_count":null},{"id":"3096f9f1-2636-43b1-811d-4e6c0a0a1519","cell_type":"code","source":"submission.head(25)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:09:01.030613Z","iopub.execute_input":"2025-10-24T09:09:01.031217Z","iopub.status.idle":"2025-10-24T09:09:01.039644Z","shell.execute_reply.started":"2025-10-24T09:09:01.031194Z","shell.execute_reply":"2025-10-24T09:09:01.039068Z"}},"outputs":[],"execution_count":null},{"id":"52511995-0a03-42bb-b0cb-f8b18dc644c4","cell_type":"code","source":"with pd.option_context('display.max_rows', None, 'display.max_columns', None):  # more options can be specified also\n    data = submission.sort_values(\"scene\", ascending=True).reset_index(drop=True)\n    print(data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:08:21.884262Z","iopub.execute_input":"2025-10-24T09:08:21.884855Z","iopub.status.idle":"2025-10-24T09:08:21.892587Z","shell.execute_reply.started":"2025-10-24T09:08:21.884828Z","shell.execute_reply":"2025-10-24T09:08:21.891724Z"}},"outputs":[],"execution_count":null},{"id":"d35d8aa9-d28c-4d5d-9162-a56f8447b799","cell_type":"markdown","source":"# COLMAP ","metadata":{}},{"id":"1504cbd2-fcec-43b8-9bd6-e31239c15720","cell_type":"code","source":"##!pip -q install --upgrade pip\n##!pip -q install pycolmap","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:38:32.056412Z","iopub.execute_input":"2025-10-24T09:38:32.057164Z","iopub.status.idle":"2025-10-24T09:38:39.917657Z","shell.execute_reply.started":"2025-10-24T09:38:32.057126Z","shell.execute_reply":"2025-10-24T09:38:39.916614Z"}},"outputs":[],"execution_count":null},{"id":"477401f9-e669-4bc7-a479-7b0167037798","cell_type":"code","source":"import pycolmap","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:38:47.531836Z","iopub.execute_input":"2025-10-24T09:38:47.532320Z","iopub.status.idle":"2025-10-24T09:38:47.535527Z","shell.execute_reply.started":"2025-10-24T09:38:47.532296Z","shell.execute_reply":"2025-10-24T09:38:47.534878Z"}},"outputs":[],"execution_count":null},{"id":"7b58fe75-a1d5-445b-b65d-9bbe6f653435","cell_type":"markdown","source":"import pandas as pd\nimport numpy as np\nimport glob\nimport os\nimport shutil\nfrom pathlib import Path\nfrom scipy.spatial.transform import Rotation as R\nimport itertools\nimport subprocess # <-- Make sure this is imported\nimport sys\n\n\nROOT_IMAGE_DIR = \"/kaggle/input/image-matching-challenge-2025/test\" \nWORKSPACE_DIR_BASE = \"colmap_workspace\"\n# TOGGLE this path based on your colmap directory\nCOLMAP_EXE = \"colmap\"\n\ntry:\n    import pycolmap\nexcept ImportError:\n    print(\"CRITICAL: pycolmap library not found. Please install it.\")\n\n    class MockReconstruction:\n        def __init__(self, model_dir): pass\n        def num_reg_images(self): return 0\n        def reg_image_ids(self): return []\n        @property\n        def images(self): return {}\n    pycolmap.Reconstruction = MockReconstruction\n\n\ndef average_quaternions(quaternions: np.ndarray) -> np.ndarray:\n    if len(quaternions) == 0:\n        return np.array([1.0, 0.0, 0.0, 0.0]) # Identity\n    first_q = quaternions[0]\n    for i in range(1, len(quaternions)):\n        if np.dot(first_q, quaternions[i]) < 0:\n            quaternions[i] = -quaternions[i]\n    mean_q = np.mean(quaternions, axis=0)\n    return mean_q / np.linalg.norm(mean_q)\n\n\ndef run_colmap_pipeline(scene_image_dir: Path, workspace_path: Path) -> Path | None:\n    db_path = workspace_path / \"database.db\"\n    sparse_dir = workspace_path / \"sparse\"\n    os.makedirs(sparse_dir, exist_ok=True)\n    \n\n    print(f\"   [COLMAP] Running feature_extractor on {scene_image_dir}\")\n    cmd = [\n        COLMAP_EXE, \"feature_extractor\",\n        \"--database_path\", str(db_path),\n        \"--image_path\", str(scene_image_dir),\n        \"--ImageReader.camera_model\", \"SIMPLE_RADIAL\",\n        \"--ImageReader.single_camera_per_folder\", \"1\",\n        \"--SiftExtraction.use_gpu\", \"true\"\n    ]\n    try:\n        subprocess.run(cmd, check=True, capture_output=True, text=True)\n    except subprocess.CalledProcessError as e:\n        print(f\"   [COLMAP ERROR] feature_extractor failed:\\n{e.stderr}\")\n        return None\n\n\n    print(\"   [COLMAP] Running exhaustive_matcher...\")\n    cmd = [\n        COLMAP_EXE, \"exhaustive_matcher\", \n        \"--database_path\", str(db_path),\n        \"--SiftMatching.use_gpu\", \"true\",\n    ]\n    try:\n        subprocess.run(cmd, check=True, capture_output=True, text=True)\n    except subprocess.CalledProcessError as e:\n        print(f\"   [COLMAP ERROR] exhaustive_matcher failed:\\n{e.stderr}\")\n        return None\n\n\n    print(\"   [COLMAP] Running mapper...\")\n    cmd = [\n        COLMAP_EXE, \"mapper\",\n        \"--database_path\", str(db_path),\n        \"--image_path\", str(scene_image_dir),\n        \"--output_path\", str(sparse_dir),\n        \"--Mapper.init_min_num_inliers\", \"5\", \n        \"--Mapper.abs_pose_min_num_inliers\", \"5\",\n        \"--Mapper.filter_max_reproj_error\", \"10.0\",\n        \"--Mapper.min_num_matches\", \"5\",\n    ]\n    try:\n        subprocess.run(cmd, check=True, capture_output=True, text=True)\n    except subprocess.CalledProcessError as e:\n        print(f\"   [COLMAP ERROR] mapper failed. This scene likely has no matches.\\n{e.stderr}\")\n        return None\n\n    model_dir = sparse_dir / \"0\"\n    if not model_dir.exists():\n        print(f\"   [COLMAP] Mapper ran but produced no model at {model_dir}\")\n        return None\n        \n    print(f\"   [COLMAP] Success! Model created at {model_dir}\")\n    return model_dir\n\n\n######↓↓↓↓↓USE THIS FUNCTION FOR ROTATION/TRANSALTION MATRIX CALCULATION↓↓↓↓↓#####\ndef find_poses_for_scene(df: pd.DataFrame) -> pd.DataFrame:\n######↑↑↑↑↑THIS THE MAIN FUNCTION↑↑↑↑↑#####\n    \n    df_filled = df.copy()\n    \n    for col in ['rotation_matrix', 'translation_vector']:\n        if col not in df_filled.columns:\n            df_filled[col] = None\n        df_filled[col] = df_filled[col].astype(object)\n\n    for (dataset, scene), scene_group in df_filled.groupby(['dataset', 'scene']):\n        print(f\"\\n=======================================================\")\n        print(f\"| PROCESSING SCENE: {scene.upper()} (Dataset: {dataset})\")\n        print(f\"=======================================================\")\n        \n        if scene == \"cluster_-1\":\n            print(\"   [INFO] Scene is 'cluster_-1'. Skipping COLMAP and writing NaN values.\")\n            dummy_rot_str = \";\".join([\"NaN\"] * 9)\n            dummy_trans_str = \";\".join([\"NaN\"] * 3)\n            for index, row in scene_group.iterrows():\n                df_filled.loc[index, 'rotation_matrix'] = dummy_rot_str\n                df_filled.loc[index, 'translation_vector'] = dummy_trans_str\n            continue\n\n        dataset_image_dir = Path(ROOT_IMAGE_DIR) / dataset \n        workspace_path = Path(f\"{WORKSPACE_DIR_BASE}_{dataset}_{scene}\")\n        temp_scene_image_dir = workspace_path / \"temp_scene_images\"\n        \n        if not dataset_image_dir.exists():\n            print(f\"⚠️ Error: Dataset directory not found, skipping: {dataset_image_dir}\")\n            continue\n            \n        if workspace_path.exists():\n            shutil.rmtree(workspace_path)\n        os.makedirs(temp_scene_image_dir)\n        \n        print(f\"   [INFO] Copying {len(scene_group)} images for scene '{scene}' to temporary folder...\")\n        images_copied = 0\n        for index, row in scene_group.iterrows():\n            image_name = row['image']\n            src_path = dataset_image_dir / image_name\n            dst_path = temp_scene_image_dir / image_name\n            if src_path.exists():\n                shutil.copy(str(src_path), str(dst_path))\n                images_copied += 1\n            else:\n                print(f\"   [WARN] Image file not found at source: {src_path}\")\n        \n        if images_copied == 0:\n            print(f\"   [ERROR] No images were found/copied for scene '{scene}'. Skipping COLMAP.\")\n            shutil.rmtree(workspace_path)\n            continue\n        \n        successful_quats = []\n        successful_trans = []\n        pose_results = {}\n        \n\n        mean_R, mean_t = None, None \n\n        try:\n            model_dir = run_colmap_pipeline(temp_scene_image_dir, workspace_path)\n\n            if model_dir:\n                try:\n                    reconstruction = pycolmap.Reconstruction(str(model_dir))\n                    if reconstruction.num_reg_images() > 0:\n                        for image_id in reconstruction.reg_image_ids():\n                            image = reconstruction.images[image_id]\n                            pose = image.cam_from_world()\n                            \n                            q, t = pose.rotation.quat, pose.translation\n                            successful_quats.append(q)\n                            successful_trans.append(t)\n                            \n                            q_scipy = [q[1], q[2], q[3], q[0]]\n                            rot_matrix = R.from_quat(q_scipy).as_matrix()\n                            pose_results[image.name] = (rot_matrix, t)\n                        \n                        # --- START OF LOGIC CHANGE ---\n                        # Calculate mean pose only if successful\n                        mean_q = average_quaternions(np.array(successful_quats))\n                        mean_t = np.mean(successful_trans, axis=0)\n                        mean_q_scipy = [mean_q[1], mean_q[2], mean_q[3], mean_q[0]]\n                        mean_R = R.from_quat(mean_q_scipy).as_matrix()\n                        # --- END OF LOGIC CHANGE ---\n                        \n                        print(f\"   [PyColmap] Successfully read {len(pose_results)} poses.\")\n                        print(f\"   [PyColmap] Calculated Mean Translation: {mean_t}\")\n                        print(f\"   [PyColmap] Calculated Mean Rotation:\\n{mean_R}\")\n                    else:\n                        print(\"   [PyColmap] COLMAP ran but registered 0 images.\")\n                        \n                except Exception as e:\n                    print(f\"❌ CRITICAL ERROR during reconstruction loading: {e}\")\n            else:\n                print(\"   [PyColmap] COLMAP pipeline failed to produce a model. All poses will be fallback.\")\n\n            # if mean_R is still None, it means COLMAP failed completely\n            # generate a random pose to use as the fallback\n            if mean_R is None:\n                print(\"   [INFO] COLMAP failed for all images. Generating random fallback pose.\")\n                mean_R = R.random().as_matrix()  # random 3x3 rotation matrix\n                mean_t = np.random.rand(3)       # random 3x1 translation vector\n                print(f\"   [INFO] Random fallback Translation: {mean_t}\")\n                print(f\"   [INFO] Random fallback Rotation:\\n{mean_R}\")\n\n\n            # update dataframe\n            mean_R_str = \";\".join(map(str, mean_R.flatten()))\n            mean_t_str = \";\".join(map(str, mean_t.flatten()))\n            \n            for index, row in scene_group.iterrows():\n                image_name = row['image']\n                \n                if image_name in pose_results:\n                    rot_matrix, t = pose_results[image_name]\n                    rot_str = \";\".join(map(str, rot_matrix.flatten()))\n                    trans_str = \";\".join(map(str, t.flatten()))\n                    df_filled.loc[index, 'rotation_matrix'] = rot_str\n                    df_filled.loc[index, 'translation_vector'] = trans_str\n                else:\n                    # use the mean pose (which is random if COLMAP failed)\n                    df_filled.loc[index, 'rotation_matrix'] = mean_R_str\n                    df_filled.loc[index, 'translation_vector'] = mean_t_str\n\n        except Exception as e:\n            print(f\"❌ UNHANDLED CRITICAL ERROR during processing scene '{scene}': {e}\")\n        finally:\n            if workspace_path.exists():\n                shutil.rmtree(workspace_path)\n            \n    return df_filled","metadata":{"execution":{"iopub.status.busy":"2025-10-24T09:45:13.265097Z","iopub.execute_input":"2025-10-24T09:45:13.265387Z","iopub.status.idle":"2025-10-24T09:45:13.286481Z","shell.execute_reply.started":"2025-10-24T09:45:13.265366Z","shell.execute_reply":"2025-10-24T09:45:13.285718Z"}}},{"id":"1aabbfac-a699-4e33-a702-1bd1f9c5477a","cell_type":"code","source":"df_result=find_poses_for_scene(submission)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:45:17.311840Z","iopub.execute_input":"2025-10-24T09:45:17.312390Z","iopub.status.idle":"2025-10-24T09:45:17.524341Z","shell.execute_reply.started":"2025-10-24T09:45:17.312369Z","shell.execute_reply":"2025-10-24T09:45:17.523570Z"}},"outputs":[],"execution_count":null},{"id":"71f7ef24-f3eb-48ad-8d3f-46446b6ca704","cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport shutil\nimport types\nfrom pathlib import Path\nfrom scipy.spatial.transform import Rotation as R\nimport subprocess\nimport sys\n\n# -------------------------------------------------\n# CONFIG\n# -------------------------------------------------\nROOT_IMAGE_DIR = \"/kaggle/input/image-matching-challenge-2025/test\"\nWORKSPACE_DIR_BASE = \"colmap_workspace\"\n\n# We'll CALL this string as a binary. We assume the runtime has 'colmap' on PATH.\nCOLMAP_EXE = \"colmap\"\n\n\n# -------------------------------------------------\n# pycolmap import (with safe fallback)\n# -------------------------------------------------\ntry:\n    import pycolmap\nexcept ImportError:\n    print(\"CRITICAL: pycolmap library not found. Using mock. Pose extraction will be mostly empty.\")\n\n    class MockReconstruction:\n        def __init__(self, model_dir):\n            pass\n        def num_reg_images(self):\n            return 0\n        def reg_image_ids(self):\n            return []\n        @property\n        def images(self):\n            return {}\n\n    pycolmap = types.SimpleNamespace(Reconstruction=MockReconstruction)\n\n\n# -------------------------------------------------\n# Helpers\n# -------------------------------------------------\ndef average_quaternions(quaternions: np.ndarray) -> np.ndarray:\n    \"\"\"\n    Average quaternions while handling antipodal symmetry.\n    Returns unit quaternion [w, x, y, z].\n    \"\"\"\n    if len(quaternions) == 0:\n        # identity rotation\n        return np.array([1.0, 0.0, 0.0, 0.0], dtype=np.float64)\n\n    # Flip signs to make them all \"point the same way\" before averaging\n    first_q = quaternions[0]\n    aligned = []\n    for q in quaternions:\n        if np.dot(first_q, q) < 0:\n            aligned.append(-q)\n        else:\n            aligned.append(q)\n    aligned = np.stack(aligned, axis=0)\n\n    mean_q = np.mean(aligned, axis=0)\n    norm = np.linalg.norm(mean_q) + 1e-12\n    return mean_q / norm\n\n\ndef _colmap_available() -> bool:\n    \"\"\"\n    Quick sanity check: can we even run the colmap binary in this environment?\n    If not, we bail early and let caller fall back to random pose.\n    \"\"\"\n    try:\n        test = subprocess.run(\n            [COLMAP_EXE, \"--help\"],\n            capture_output=True,\n            text=True\n        )\n        # returncode 0 or 1 is fine (colmap tends to exit 1 after printing help)\n        return test.returncode in (0, 1)\n    except FileNotFoundError:\n        return False\n    except PermissionError:\n        return False\n\n\ndef run_colmap_pipeline(scene_image_dir: Path, workspace_path: Path):\n    \"\"\"\n    Run COLMAP (feature_extractor -> exhaustive_matcher -> mapper)\n    on scene_image_dir.\n    Returns Path to sparse model dir (workspace_path/sparse/0) or None.\n    \"\"\"\n\n    if not _colmap_available():\n        print(\"   [COLMAP ERROR] 'colmap' binary not available in this environment. Skipping SfM.\")\n        return None\n\n    db_path = workspace_path / \"database.db\"\n    sparse_dir = workspace_path / \"sparse\"\n    os.makedirs(sparse_dir, exist_ok=True)\n\n    # 1. feature_extractor\n    print(f\"   [COLMAP] Running feature_extractor on {scene_image_dir}\")\n    cmd = [\n        COLMAP_EXE, \"feature_extractor\",\n        \"--database_path\", str(db_path),\n        \"--image_path\", str(scene_image_dir),\n        \"--ImageReader.camera_model\", \"SIMPLE_RADIAL\",\n        \"--ImageReader.single_camera_per_folder\", \"1\",\n        \"--SiftExtraction.use_gpu\", \"true\",\n    ]\n    try:\n        subprocess.run(cmd, check=True, capture_output=True, text=True)\n    except subprocess.CalledProcessError as e:\n        print(f\"   [COLMAP ERROR] feature_extractor failed:\\n{e.stderr}\")\n        return None\n\n    # 2. exhaustive_matcher\n    print(\"   [COLMAP] Running exhaustive_matcher...\")\n    cmd = [\n        COLMAP_EXE, \"exhaustive_matcher\",\n        \"--database_path\", str(db_path),\n        \"--SiftMatching.use_gpu\", \"true\",\n    ]\n    try:\n        subprocess.run(cmd, check=True, capture_output=True, text=True)\n    except subprocess.CalledProcessError as e:\n        print(f\"   [COLMAP ERROR] exhaustive_matcher failed:\\n{e.stderr}\")\n        return None\n\n    # 3. mapper\n    print(\"   [COLMAP] Running mapper...\")\n    cmd = [\n        COLMAP_EXE, \"mapper\",\n        \"--database_path\", str(db_path),\n        \"--image_path\", str(scene_image_dir),\n        \"--output_path\", str(sparse_dir),\n        \"--Mapper.init_min_num_inliers\", \"5\",\n        \"--Mapper.abs_pose_min_num_inliers\", \"5\",\n        \"--Mapper.filter_max_reproj_error\", \"10.0\",\n        \"--Mapper.min_num_matches\", \"5\",\n    ]\n    try:\n        subprocess.run(cmd, check=True, capture_output=True, text=True)\n    except subprocess.CalledProcessError as e:\n        print(\"   [COLMAP ERROR] mapper failed. Scene probably has no consistent matches.\")\n        print(f\"   [COLMAP STDERR]\\n{e.stderr}\")\n        return None\n\n    # COLMAP writes sparse models to sparse/0, sparse/1, ...\n    model_dir = sparse_dir / \"0\"\n    if not model_dir.exists():\n        print(f\"   [COLMAP] Mapper ran but produced no model at {model_dir}\")\n        return None\n\n    print(f\"   [COLMAP] Success! Model created at {model_dir}\")\n    return model_dir\n\n\n# -------------------------------------------------\n# MAIN FUNCTION (pose estimation per scene)\n# -------------------------------------------------\ndef find_poses_for_scene(df: pd.DataFrame) -> pd.DataFrame:\n    \"\"\"\n    For each (dataset, scene):\n    - Copy that scene's images to a temp workspace\n    - Run COLMAP to estimate poses\n    - Extract rotation matrix (3x3) and translation (3,)\n    - Write them back into df as strings \"r11;r12;...;r33\" and \"tx;ty;tz\"\n\n    cluster_-1 scenes get NaN pose.\n\n    If COLMAP fails totally, we generate a random fallback pose so that downstream\n    code still has something to write.\n    \"\"\"\n\n    df_filled = df.copy()\n\n    # ensure pose columns exist and are dtype=object so we can assign strings later\n    for col in ['rotation_matrix', 'translation_vector']:\n        if col not in df_filled.columns:\n            df_filled[col] = None\n        df_filled[col] = df_filled[col].astype(object)\n\n    # group by dataset + scene\n    for (dataset, scene), scene_group in df_filled.groupby(['dataset', 'scene']):\n        print(\"\\n=======================================================\")\n        print(f\"| PROCESSING SCENE: {scene.upper()} (Dataset: {dataset})\")\n        print(\"=======================================================\")\n\n        # special case: cluster_-1 is \"junk / outlier\"\n        if scene == \"cluster_-1\":\n            print(\"   [INFO] Scene is 'cluster_-1'. Skipping COLMAP and writing NaN values.\")\n            dummy_rot_str = \";\".join([\"NaN\"] * 9)\n            dummy_trans_str = \";\".join([\"NaN\"] * 3)\n            for index, _row in scene_group.iterrows():\n                df_filled.loc[index, 'rotation_matrix'] = dummy_rot_str\n                df_filled.loc[index, 'translation_vector'] = dummy_trans_str\n            continue\n\n        # source folder with the real test images\n        dataset_image_dir = Path(ROOT_IMAGE_DIR) / dataset\n        if not dataset_image_dir.exists():\n            print(f\"⚠️ Error: Dataset directory not found, skipping: {dataset_image_dir}\")\n            continue\n\n        # make a fresh workspace for this dataset+scene\n        workspace_path = Path(f\"{WORKSPACE_DIR_BASE}_{dataset}_{scene}\")\n        temp_scene_image_dir = workspace_path / \"temp_scene_images\"\n\n        if workspace_path.exists():\n            shutil.rmtree(workspace_path)\n        os.makedirs(temp_scene_image_dir, exist_ok=True)\n\n        # copy only the images that belong to this scene\n        print(f\"   [INFO] Copying {len(scene_group)} images for scene '{scene}' to temporary folder...\")\n        images_copied = 0\n        for index, row in scene_group.iterrows():\n            image_name = row['image']  # should include extension like .png or .jpg\n            src_path = dataset_image_dir / image_name\n            dst_path = temp_scene_image_dir / image_name\n            if src_path.exists():\n                shutil.copy(str(src_path), str(dst_path))\n                images_copied += 1\n            else:\n                print(f\"   [WARN] Image not found: {src_path}\")\n\n        if images_copied == 0:\n            print(f\"   [ERROR] No images copied for scene '{scene}'. Skipping COLMAP.\")\n            shutil.rmtree(workspace_path, ignore_errors=True)\n            continue\n\n        # run COLMAP pipeline\n        pose_results = {}      # image_name -> (Rij[3x3], t[3])\n        successful_quats = []  # list of [w,x,y,z]\n        successful_trans = []  # list of [tx,ty,tz]\n\n        mean_R = None\n        mean_t = None\n\n        try:\n            model_dir = run_colmap_pipeline(temp_scene_image_dir, workspace_path)\n\n            if model_dir:\n                # try reading poses with pycolmap\n                try:\n                    reconstruction = pycolmap.Reconstruction(str(model_dir))\n                    if reconstruction.num_reg_images() > 0:\n                        for image_id in reconstruction.reg_image_ids():\n                            image = reconstruction.images[image_id]\n\n                            # camera-from-world pose\n                            pose = image.cam_from_world()\n\n                            # pycolmap returns rotation as quaternion (w,x,y,z)\n                            q = pose.rotation.quat\n                            t = pose.translation  # np.array([tx,ty,tz])\n\n                            successful_quats.append(q)\n                            successful_trans.append(t)\n\n                            # convert quaternion -> 3x3 rotation matrix (scipy expects [x,y,z,w])\n                            q_scipy = [q[1], q[2], q[3], q[0]]\n                            rot_matrix = R.from_quat(q_scipy).as_matrix()\n\n                            pose_results[image.name] = (rot_matrix, t)\n\n                        # compute mean pose across registered images\n                        mean_q = average_quaternions(np.array(successful_quats))\n                        mean_t = np.mean(successful_trans, axis=0)\n\n                        mean_q_scipy = [mean_q[1], mean_q[2], mean_q[3], mean_q[0]]\n                        mean_R = R.from_quat(mean_q_scipy).as_matrix()\n\n                        print(f\"   [PyColmap] Successfully read {len(pose_results)} poses.\")\n                        print(f\"   [PyColmap] Mean t: {mean_t}\")\n                        print(f\"   [PyColmap] Mean R:\\n{mean_R}\")\n                    else:\n                        print(\"   [PyColmap] COLMAP ran but registered 0 images.\")\n                except Exception as e:\n                    print(f\"❌ CRITICAL ERROR loading reconstruction with pycolmap: {e}\")\n            else:\n                print(\"   [PyColmap] COLMAP pipeline produced no model. Using fallback.\")\n\n            # still nothing? -> fallback random pose to avoid NaNs\n            if mean_R is None:\n                print(\"   [INFO] No valid COLMAP pose. Generating random fallback pose.\")\n                mean_R = R.random().as_matrix()          # (3,3)\n                mean_t = np.random.rand(3)               # (3,)\n                print(f\"   [INFO] Fallback t: {mean_t}\")\n                print(f\"   [INFO] Fallback R:\\n{mean_R}\")\n\n            # stringify mean pose for default fill\n            mean_R_str = \";\".join(map(str, mean_R.flatten()))\n            mean_t_str = \";\".join(map(str, mean_t.flatten()))\n\n            # write per-row results into df_filled\n            for index, row in scene_group.iterrows():\n                image_name = row['image']\n                if image_name in pose_results:\n                    rot_matrix, t_vec = pose_results[image_name]\n                    rot_str = \";\".join(map(str, rot_matrix.flatten()))\n                    trans_str = \";\".join(map(str, t_vec.flatten()))\n                    df_filled.loc[index, 'rotation_matrix'] = rot_str\n                    df_filled.loc[index, 'translation_vector'] = trans_str\n                else:\n                    # use scene mean (which might be fallback random)\n                    df_filled.loc[index, 'rotation_matrix'] = mean_R_str\n                    df_filled.loc[index, 'translation_vector'] = mean_t_str\n\n        except Exception as e:\n            print(f\"❌ UNHANDLED CRITICAL ERROR during processing scene '{scene}': {e}\")\n\n        finally:\n            # clean workspace\n            shutil.rmtree(workspace_path, ignore_errors=True)\n\n    return df_filled\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:50:33.763487Z","iopub.execute_input":"2025-10-24T09:50:33.763823Z","iopub.status.idle":"2025-10-24T09:50:33.788429Z","shell.execute_reply.started":"2025-10-24T09:50:33.763799Z","shell.execute_reply":"2025-10-24T09:50:33.787648Z"}},"outputs":[],"execution_count":null},{"id":"1ea3ad55-8274-46b0-bea1-185cf354abc2","cell_type":"code","source":"df_result=find_poses_for_scene(submission)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:50:34.421602Z","iopub.execute_input":"2025-10-24T09:50:34.421826Z","iopub.status.idle":"2025-10-24T09:50:34.644346Z","shell.execute_reply.started":"2025-10-24T09:50:34.421811Z","shell.execute_reply":"2025-10-24T09:50:34.643810Z"}},"outputs":[],"execution_count":null},{"id":"fbcaa73d-cc54-4dd2-81ec-298949fec5a2","cell_type":"code","source":"df_result.head(23)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:54:08.517240Z","iopub.execute_input":"2025-10-24T09:54:08.517765Z","iopub.status.idle":"2025-10-24T09:54:08.527982Z","shell.execute_reply.started":"2025-10-24T09:54:08.517727Z","shell.execute_reply":"2025-10-24T09:54:08.527211Z"}},"outputs":[],"execution_count":null},{"id":"261c3b62-cd50-451c-b2e7-649d9d9fada9","cell_type":"code","source":"sub = pd.read_csv(\"/kaggle/working/submission.csv\")\nprint(sub.shape)\nwith pd.option_context('display.max_rows', None, 'display.max_columns', None):  # more options can be specified also\n    print(sub)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:51:04.152831Z","iopub.execute_input":"2025-10-24T09:51:04.153401Z","iopub.status.idle":"2025-10-24T09:51:04.163439Z","shell.execute_reply.started":"2025-10-24T09:51:04.153377Z","shell.execute_reply":"2025-10-24T09:51:04.162878Z"}},"outputs":[],"execution_count":null}]}