{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":126777,"databundleVersionId":15314950,"sourceType":"competition"},{"sourceId":297798014,"sourceType":"kernelVersion"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Jaguar LightGlue Similarity Search for Sub**","metadata":{}},{"cell_type":"code","source":"!pip install -q kornia kornia-rs\n!pip install -q git+https://github.com/cvg/LightGlue.git","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-15T08:44:13.593770Z","iopub.execute_input":"2026-02-15T08:44:13.594426Z","iopub.status.idle":"2026-02-15T08:45:55.651958Z","shell.execute_reply.started":"2026-02-15T08:44:13.594394Z","shell.execute_reply":"2026-02-15T08:45:55.650679Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport numpy as np\nimport cv2\nimport pandas as pd\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\nimport matplotlib.pyplot as plt\n\nfrom lightglue import LightGlue, SuperPoint\nfrom lightglue.utils import rbd\n\nprint(f\"PyTorch: {torch.__version__}\")\nprint(f\"CUDA available: {torch.cuda.is_available()}\")\nif torch.cuda.is_available():\n    print(f\"GPU: {torch.cuda.get_device_name(0)}\")\n    print(f\"GPU Memory: {torch.cuda.get_device_properties(0).total_memory / 1e9:.2f} GB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-15T08:45:55.653470Z","iopub.execute_input":"2026-02-15T08:45:55.653863Z","iopub.status.idle":"2026-02-15T08:46:06.255860Z","shell.execute_reply.started":"2026-02-15T08:45:55.653831Z","shell.execute_reply":"2026-02-15T08:46:06.254744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PAIRS_CSV = '/kaggle/input/jaguar-re-id/test.csv'       \nIMAGE_DIR = '/kaggle/input/notebooks/stpeteishii/jaguar-dataset-background-removal/test'      \nOUTPUT_CSV = 'submission.csv'  \nNORMALIZATION_FACTOR = 100000.0   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-15T08:46:06.257812Z","iopub.execute_input":"2026-02-15T08:46:06.258283Z","iopub.status.idle":"2026-02-15T08:46:06.263334Z","shell.execute_reply.started":"2026-02-15T08:46:06.258258Z","shell.execute_reply":"2026-02-15T08:46:06.262155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = 'cuda' if torch.cuda.is_available() else 'cpu'\nextractor = SuperPoint(max_num_keypoints=2048).eval().to(device)\nmatcher = LightGlue(features='superpoint').eval().to(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-15T08:46:06.264347Z","iopub.execute_input":"2026-02-15T08:46:06.264758Z","iopub.status.idle":"2026-02-15T08:46:07.146657Z","shell.execute_reply.started":"2026-02-15T08:46:06.264734Z","shell.execute_reply":"2026-02-15T08:46:07.145357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load CSV\ndf = pd.read_csv(PAIRS_CSV)\nprint(f\"Total pairs: {len(df)}\")\ndisplay(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-15T08:49:42.637443Z","iopub.execute_input":"2026-02-15T08:49:42.637766Z","iopub.status.idle":"2026-02-15T08:49:42.728347Z","shell.execute_reply.started":"2026-02-15T08:49:42.637741Z","shell.execute_reply":"2026-02-15T08:49:42.727595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nimport torch\nfrom tqdm import tqdm\n\n# --- Step 1: Pre-extract features for all 371 unique images ---\nunique_images = sorted(list(set(df['query_image']) | set(df['gallery_image'])))\nfeatures_cache = {}\n\nprint(f\"Extracting features for {len(unique_images)} images...\")\nwith torch.no_grad():\n    for img_name in tqdm(unique_images):\n        img_path = Path(IMAGE_DIR) / img_name\n        img = cv2.imread(str(img_path), cv2.IMREAD_GRAYSCALE)\n        if img is None: continue\n        \n        # Preprocessing\n        img_tensor = torch.from_numpy(img).float() / 255.0\n        img_tensor = img_tensor.unsqueeze(0).unsqueeze(0).to(device)\n        \n        # Extract features ONCE and store them\n        features_cache[img_name] = extractor.extract(img_tensor)\n\n# --- Step 2: Match pairs using the cached features ---\nsimilarities, num_matches_list, confidences = [], [], []\n\nprint(\"Matching pairs...\")\nwith torch.no_grad():\n    for _, row in tqdm(df.iterrows(), total=len(df)):\n        f0 = features_cache.get(row['query_image'])\n        f1 = features_cache.get(row['gallery_image'])\n        \n        if f0 is None or f1 is None:\n            similarities.append(0.0); num_matches_list.append(0); confidences.append(0.0)\n            continue\n            \n        # Match using pre-computed features (Fast)\n        out = matcher({'image0': f0, 'image1': f1})\n        \n        # Remove batch dim (assuming rbd is your helper function)\n        out = rbd(out)\n        matches = out['matches']\n        scores = out['scores']\n        \n        n_matches = (matches > -1).sum().item()\n        conf = scores.mean().item() if len(scores) > 0 else 0.0\n        \n        num_matches_list.append(n_matches)\n        confidences.append(conf)\n        similarities.append(n_matches / NORMALIZATION_FACTOR)\n\n# Update DataFrame\ndf['similarity'] = similarities\ndf['num_matches'] = num_matches_list\ndf['confidence'] = confidences","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 2, figsize=(15, 10))\n\naxes[0, 0].hist(similarities, bins=50, edgecolor='black')\naxes[0, 0].axvline(np.mean(similarities), color='red', linestyle='--', \n                   label=f'Mean: {np.mean(similarities):.6f}')\naxes[0, 0].set_xlabel('Similarity Score')\naxes[0, 0].set_ylabel('Frequency')\naxes[0, 0].set_title('Similarity Distribution')\naxes[0, 0].legend()\naxes[0, 0].grid(alpha=0.3)\n\naxes[0, 1].hist(num_matches_list, bins=50, edgecolor='black', color='green')\naxes[0, 1].axvline(np.mean(num_matches_list), color='red', linestyle='--',\n                   label=f'Mean: {np.mean(num_matches_list):.1f}')\naxes[0, 1].set_xlabel('Number of Matches')\naxes[0, 1].set_ylabel('Frequency')\naxes[0, 1].set_title('Match Count Distribution')\naxes[0, 1].legend()\naxes[0, 1].grid(alpha=0.3)\n\naxes[1, 0].scatter(num_matches_list, similarities, alpha=0.5)\naxes[1, 0].set_xlabel('Number of Matches')\naxes[1, 0].set_ylabel('Similarity Score')\naxes[1, 0].set_title('Matches vs Similarity')\naxes[1, 0].grid(alpha=0.3)\n\naxes[1, 1].hist(confidences, bins=50, edgecolor='black', color='orange')\naxes[1, 1].axvline(np.mean(confidences), color='red', linestyle='--',\n                   label=f'Mean: {np.mean(confidences):.4f}')\naxes[1, 1].set_xlabel('Match Confidence')\naxes[1, 1].set_ylabel('Frequency')\naxes[1, 1].set_title('Confidence Distribution')\naxes[1, 1].legend()\naxes[1, 1].grid(alpha=0.3)\n\nplt.tight_layout()\nplt.savefig('/content/similarity_analysis.png', dpi=300, bbox_inches='tight')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-15T08:46:07.319551Z","iopub.status.idle":"2026-02-15T08:46:07.319939Z","shell.execute_reply.started":"2026-02-15T08:46:07.319733Z","shell.execute_reply":"2026-02-15T08:46:07.319752Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Submission file\nsubmission_df = df[['row_id', 'similarity']]\nsubmission_df.to_csv(OUTPUT_CSV, index=False)\nprint(f\"✅ Submission file saved: {OUTPUT_CSV}\")\n\n# Save detailed results as well\ndetail_csv = OUTPUT_CSV.replace('.csv', '_detail.csv')\ndf.to_csv(detail_csv, index=False)\nprint(f\"✅ Detailed results saved: {detail_csv}\")\n\n# Verify content\ndisplay(submission_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-15T08:46:07.322278Z","iopub.status.idle":"2026-02-15T08:46:07.322625Z","shell.execute_reply.started":"2026-02-15T08:46:07.322478Z","shell.execute_reply":"2026-02-15T08:46:07.322494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}