{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91498,"databundleVersionId":11655853,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-01T00:23:05.589819Z","iopub.execute_input":"2025-05-01T00:23:05.590064Z","iopub.status.idle":"2025-05-01T00:23:09.366092Z","shell.execute_reply.started":"2025-05-01T00:23:05.590041Z","shell.execute_reply":"2025-05-01T00:23:09.365266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\nImage Matching Challenge 2025 Solution\n\n\"\"\"\n\nimport os\nimport numpy as np\nimport pandas as pd\nimport cv2\nfrom pathlib import Path\nimport matplotlib.pyplot as plt\nfrom sklearn.cluster import DBSCAN\nfrom collections import defaultdict\nfrom tqdm import tqdm\n\n\n# Step 1: Explore and Parse Data\n# =============================\n\ndef explore_competition_data(base_path):\n    \"\"\"\n    Explore the competition data structure and load necessary files\n    \"\"\"\n    print(\"Exploring competition data...\")\n    \n    # Load CSV files\n    sample_submission = pd.read_csv(os.path.join(base_path, 'sample_submission.csv'))\n    train_thresholds = pd.read_csv(os.path.join(base_path, 'train_thresholds.csv'))\n    train_labels = pd.read_csv(os.path.join(base_path, 'train_labels.csv'))\n    \n    print(\"CSV Files:\")\n    print(f\"  Sample Submission: {sample_submission.shape[0]} rows, {sample_submission.shape[1]} columns\")\n    print(f\"  Train Thresholds: {train_thresholds.shape[0]} rows, {train_thresholds.shape[1]} columns\")\n    print(f\"  Train Labels: {train_labels.shape[0]} rows, {train_labels.shape[1]} columns\")\n    \n    # Sample of train labels\n    print(\"\\nSample of train_labels.csv:\")\n    print(train_labels.head())\n    \n    # Get list of datasets from train labels\n    datasets = train_labels['dataset'].unique()\n    print(f\"\\nFound {len(datasets)} datasets: {', '.join(datasets)}\")\n    \n    # Explore train directory structure\n    train_dir = os.path.join(base_path, 'train')\n    train_datasets = os.listdir(train_dir)\n    print(f\"\\nTrain directory contains {len(train_datasets)} datasets\")\n    \n    # Sample a few train datasets\n    for dataset in train_datasets[:3]:\n        dataset_path = os.path.join(train_dir, dataset)\n        if os.path.isdir(dataset_path):\n            images = [f for f in os.listdir(dataset_path) if f.endswith(('.png', '.jpg'))]\n            print(f\"  Dataset '{dataset}': {len(images)} images\")\n            if images:\n                print(f\"    Sample image: {images[0]}\")\n    \n    # Explore test directory structure\n    test_dir = os.path.join(base_path, 'test')\n    test_datasets = os.listdir(test_dir)\n    print(f\"\\nTest directory contains {len(test_datasets)} datasets\")\n    \n    # Sample a few test datasets\n    for dataset in test_datasets[:3]:\n        dataset_path = os.path.join(test_dir, dataset)\n        if os.path.isdir(dataset_path):\n            images = [f for f in os.listdir(dataset_path) if f.endswith(('.png', '.jpg'))]\n            print(f\"  Dataset '{dataset}': {len(images)} images\")\n            if images:\n                print(f\"    Sample image: {images[0]}\")\n    \n    # Analyze scene distribution in train_labels\n    scene_counts = train_labels.groupby(['dataset', 'scene']).size().reset_index(name='count')\n    print(\"\\nScene distribution in training data:\")\n    print(scene_counts.head(10))\n    \n    # Count outliers\n    outliers = train_labels[train_labels['scene'] == 'outliers'].groupby('dataset').size()\n    print(\"\\nOutliers per dataset:\")\n    print(outliers)\n    \n    return sample_submission, train_thresholds, train_labels, datasets\n\n\n# Step 2: Parse and Prepare Data\n# =============================\n\ndef load_training_data(base_path, train_labels):\n    \"\"\"\n    Load training data organized by dataset and scene\n    \"\"\"\n    print(\"Loading training data...\")\n    \n    # Create structure to hold training data\n    training_data = {}\n    \n    # Group train_labels by dataset and scene\n    grouped = train_labels.groupby(['dataset', 'scene'])\n    \n    # Process each group\n    for (dataset, scene), group in grouped:\n        if dataset not in training_data:\n            training_data[dataset] = {}\n        \n        if scene not in training_data[dataset]:\n            training_data[dataset][scene] = []\n        \n        # Add images to the scene\n        for _, row in group.iterrows():\n            image_file = row['image']\n            image_path = os.path.join(base_path, 'train', dataset, image_file)\n            \n            # Extract rotation and translation if not NaN\n            rotation_str = row['rotation_matrix']\n            translation_str = row['translation_vector']\n            \n            rotation = None\n            translation = None\n            \n            if rotation_str != 'nan;nan;nan;nan;nan;nan;nan;nan;nan':\n                rotation_values = rotation_str.split(';')\n                rotation = np.array([float(val) for val in rotation_values]).reshape(3, 3)\n            \n            if translation_str != 'nan;nan;nan':\n                translation_values = translation_str.split(';')\n                translation = np.array([float(val) for val in translation_values])\n            \n            training_data[dataset][scene].append({\n                'file': image_file,\n                'path': image_path,\n                'rotation': rotation,\n                'translation': translation\n            })\n    \n    # Print summary\n    print(\"Training data summary:\")\n    for dataset, scenes in training_data.items():\n        print(f\"  Dataset '{dataset}': {len(scenes)} scenes\")\n        for scene, images in scenes.items():\n            print(f\"    Scene '{scene}': {len(images)} images\")\n    \n    return training_data\n\n\ndef load_test_data(base_path, datasets):\n    \"\"\"\n    Load test data organized by dataset\n    \"\"\"\n    print(\"Loading test data...\")\n    \n    # Create structure to hold test data\n    test_data = {}\n    \n    # Process each dataset\n    for dataset in datasets:\n        test_dir = os.path.join(base_path, 'test', dataset)\n        if not os.path.exists(test_dir):\n            print(f\"  Warning: Test directory for dataset '{dataset}' not found\")\n            continue\n        \n        # List image files\n        image_files = [f for f in os.listdir(test_dir) if f.endswith(('.png', '.jpg'))]\n        \n        # Store image paths\n        test_data[dataset] = [\n            {\n                'file': image_file,\n                'path': os.path.join(test_dir, image_file)\n            }\n            for image_file in image_files\n        ]\n    \n    # Print summary\n    print(\"Test data summary:\")\n    for dataset, images in test_data.items():\n        print(f\"  Dataset '{dataset}': {len(images)} images\")\n    \n    return test_data\n\n\n# Step 3: Feature Extraction\n# =========================\n\ndef extract_features(image_path):\n    \"\"\"\n    Extract SIFT features from an image\n    \"\"\"\n    # Read image\n    img = cv2.imread(str(image_path), cv2.IMREAD_GRAYSCALE)\n    if img is None:\n        print(f\"Could not read image: {image_path}\")\n        return None, None\n    \n    # Initialize SIFT detector\n    sift = cv2.SIFT_create(nfeatures=2000)  # Use more features for better matching\n    \n    # Detect keypoints and compute descriptors\n    keypoints, descriptors = sift.detectAndCompute(img, None)\n    \n    return keypoints, descriptors\n\n\ndef extract_features_for_dataset(images):\n    \"\"\"\n    Extract features for all images in a dataset\n    \"\"\"\n    features = {}\n    \n    for image_info in tqdm(images, desc=\"Extracting features\"):\n        keypoints, descriptors = extract_features(image_info['path'])\n        features[image_info['file']] = (keypoints, descriptors)\n    \n    return features\n\n\n# Step 4: Feature Matching\n# =======================\n\ndef match_features(desc1, desc2, ratio_threshold=0.75):\n    \"\"\"\n    Match features between two images using Lowe's ratio test\n    \"\"\"\n    # Handle None descriptors\n    if desc1 is None or desc2 is None:\n        return []\n    \n    # FLANN parameters\n    FLANN_INDEX_KDTREE = 1\n    index_params = dict(algorithm=FLANN_INDEX_KDTREE, trees=5)\n    search_params = dict(checks=50)\n    \n    # Create FLANN matcher\n    flann = cv2.FlannBasedMatcher(index_params, search_params)\n    \n    # Match descriptors\n    try:\n        matches = flann.knnMatch(desc1, desc2, k=2)\n    except cv2.error:\n        return []\n    \n    # Apply Lowe's ratio test\n    good_matches = []\n    for m, n in matches:\n        if m.distance < ratio_threshold * n.distance:\n            good_matches.append(m)\n    \n    return good_matches\n\n\ndef build_similarity_matrix(features):\n    \"\"\"\n    Build a similarity matrix for all image pairs\n    \"\"\"\n    image_files = list(features.keys())\n    n = len(image_files)\n    \n    # Initialize similarity matrix\n    similarity_matrix = np.zeros((n, n))\n    \n    # Compute pairwise similarities\n    for i in tqdm(range(n), desc=\"Building similarity matrix\"):\n        # Extract descriptors for image i\n        _, desc_i = features[image_files[i]]\n        \n        for j in range(i+1, n):\n            # Extract descriptors for image j\n            _, desc_j = features[image_files[j]]\n            \n            # Match features\n            matches = match_features(desc_i, desc_j)\n            \n            # Number of matches as similarity measure\n            similarity = len(matches)\n            similarity_matrix[i, j] = similarity\n            similarity_matrix[j, i] = similarity  # Symmetric\n    \n    return similarity_matrix, image_files\n\n\n# Step 5: Image Clustering\n# =======================\n\ndef cluster_images(similarity_matrix, image_files, threshold=10, min_samples=2):\n    \"\"\"\n    Cluster images based on similarity matrix\n    \"\"\"\n    # Convert similarity to distance (higher similarity = lower distance)\n    max_sim = np.max(similarity_matrix)\n    if max_sim > 0:\n        distance_matrix = 1 - (similarity_matrix / max_sim)\n    else:\n        distance_matrix = 1 - similarity_matrix\n    \n    # Apply DBSCAN clustering\n    db = DBSCAN(eps=0.5, min_samples=min_samples, metric='precomputed')\n    labels = db.fit_predict(distance_matrix)\n    \n    # Organize images by cluster\n    clusters = defaultdict(list)\n    outliers = []\n    \n    for i, label in enumerate(labels):\n        if label >= 0:  # Not noise\n            clusters[f\"cluster{label+1}\"].append(image_files[i])\n        else:  # Noise points are outliers\n            outliers.append(image_files[i])\n    \n    print(f\"Found {len(clusters)} clusters and {len(outliers)} outliers\")\n    \n    return clusters, outliers\n\n\n# Step 6: Pose Estimation\n# ======================\n\ndef estimate_poses(features, clusters, K=None):\n    \"\"\"\n    Estimate camera poses for images in each cluster\n    \"\"\"\n    # If K is unknown, use a default intrinsic matrix\n    if K is None:\n        K = np.array([\n            [1000, 0, 500],\n            [0, 1000, 500],\n            [0, 0, 1]\n        ])\n    \n    poses = {}\n    \n    # Process each cluster\n    for cluster_label, image_files in clusters.items():\n        if len(image_files) < 2:\n            continue\n        \n        # Use the first image as the reference frame\n        reference_image = image_files[0]\n        kp_ref, desc_ref = features[reference_image]\n        \n        # Set reference pose (identity rotation, zero translation)\n        R_ref = np.eye(3)\n        t_ref = np.zeros((3, 1))\n        poses[reference_image] = {\n            'R': R_ref,\n            't': t_ref\n        }\n        \n        # Estimate poses for other images relative to the reference\n        for image_file in image_files[1:]:\n            kp, desc = features[image_file]\n            \n            # Match features with reference image\n            matches = match_features(desc_ref, desc)\n            \n            if len(matches) < 5:\n                # Not enough matches for reliable pose estimation\n                poses[image_file] = {\n                    'R': np.eye(3),\n                    't': np.zeros((3, 1))\n                }\n                continue\n            \n            # Extract matched points\n            pts_ref = np.float32([kp_ref[m.queryIdx].pt for m in matches])\n            pts = np.float32([kp[m.trainIdx].pt for m in matches])\n            \n            # Estimate essential matrix\n            E, mask = cv2.findEssentialMat(pts_ref, pts, K, method=cv2.RANSAC, prob=0.999, threshold=1.0)\n            \n            # Recover pose (R, t) from essential matrix\n            _, R, t, _ = cv2.recoverPose(E, pts_ref, pts, K, mask=mask)\n            \n            # Store the estimated pose\n            poses[image_file] = {\n                'R': R,\n                't': t\n            }\n    \n    return poses\n\n\n# Step 7: Create Submission\n# ========================\n\ndef create_submission(test_results, output_path='submission.csv'):\n    \"\"\"\n    Create submission file with scene assignments and camera poses\n    \"\"\"\n    # Prepare data for submission\n    rows = []\n    \n    # Process each dataset\n    for dataset, result in test_results.items():\n        clusters = result['clusters']\n        outliers = result['outliers']\n        poses = result['poses']\n        \n        # Add clustered images with poses\n        for cluster_label, image_files in clusters.items():\n            for image_file in image_files:\n                if image_file in poses:\n                    R = poses[image_file]['R']\n                    t = poses[image_file]['t']\n                    \n                    # Format rotation matrix (flatten to row-major)\n                    rot_str = \";\".join([str(x) for x in R.flatten()])\n                    \n                    # Format translation vector\n                    trans_str = \";\".join([str(x) for x in t.flatten()])\n                    \n                    rows.append([\n                        dataset,\n                        cluster_label,\n                        image_file,\n                        rot_str,\n                        trans_str\n                    ])\n                else:\n                    # Image is in cluster but pose could not be estimated\n                    rows.append([\n                        dataset,\n                        cluster_label,\n                        image_file,\n                        \"nan;nan;nan;nan;nan;nan;nan;nan;nan\",\n                        \"nan;nan;nan\"\n                    ])\n        \n        # Add outliers\n        for image_file in outliers:\n            rows.append([\n                dataset,\n                \"outliers\",\n                image_file,\n                \"nan;nan;nan;nan;nan;nan;nan;nan;nan\",\n                \"nan;nan;nan\"\n            ])\n    \n    # Create DataFrame\n    submission_df = pd.DataFrame(\n        rows,\n        columns=['dataset', 'scene', 'image', 'rotation_matrix', 'translation_vector']\n    )\n    \n    # Save to CSV\n    submission_df.to_csv(output_path, index=False)\n    print(f\"Submission saved to {output_path}\")\n    \n    return submission_df\n\n\n# Step 8: Calculate Evaluation Metrics (For Training Data)\n# ======================================================\n\ndef calculate_metrics(predictions, ground_truth):\n    \"\"\"\n    Calculate evaluation metrics based on the competition's criteria\n    \"\"\"\n    # Group predictions and ground truth by dataset\n    pred_by_dataset = predictions.groupby('dataset')\n    gt_by_dataset = ground_truth.groupby('dataset')\n    \n    dataset_scores = []\n    \n    # Process each dataset\n    for dataset in predictions['dataset'].unique():\n        pred_df = pred_by_dataset.get_group(dataset)\n        gt_df = gt_by_dataset.get_group(dataset)\n        \n        # Convert to dictionaries for easier processing\n        pred_dict = {}\n        for _, row in pred_df.iterrows():\n            pred_dict[row['image']] = {\n                'scene': row['scene'],\n                'rotation_matrix': row['rotation_matrix'],\n                'translation_vector': row['translation_vector']\n            }\n        \n        gt_dict = {}\n        for _, row in gt_df.iterrows():\n            gt_dict[row['image']] = {\n                'scene': row['scene'],\n                'rotation_matrix': row['rotation_matrix'],\n                'translation_vector': row['translation_vector']\n            }\n        \n        # Group images by predicted scene\n        pred_scenes = defaultdict(list)\n        for image, data in pred_dict.items():\n            pred_scenes[data['scene']].append(image)\n        \n        # Group images by ground truth scene\n        gt_scenes = defaultdict(list)\n        for image, data in gt_dict.items():\n            gt_scenes[data['scene']].append(image)\n        \n        # Calculate metrics for each ground truth scene\n        scene_metrics = []\n        for gt_scene, gt_images in gt_scenes.items():\n            if gt_scene == 'outliers':\n                continue  # Skip outliers for now\n            \n            # Find the best matching predicted scene\n            best_maa = 0\n            best_cluster_score = 0\n            best_scene = None\n            \n            for pred_scene, pred_images in pred_scenes.items():\n                if pred_scene == 'outliers':\n                    continue\n                \n                # Calculate mAA (recall)\n                common_images = set(gt_images) & set(pred_images)\n                maa = len(common_images) / len(gt_images)\n                \n                # Calculate clustering score (precision)\n                cluster_score = len(common_images) / len(pred_images)\n                \n                # Check if this is the best match\n                if maa > best_maa or (maa == best_maa and cluster_score > best_cluster_score):\n                    best_maa = maa\n                    best_cluster_score = cluster_score\n                    best_scene = pred_scene\n            \n            if best_scene:\n                scene_metrics.append({\n                    'gt_scene': gt_scene,\n                    'pred_scene': best_scene,\n                    'maa': best_maa,\n                    'cluster_score': best_cluster_score,\n                    'f1_score': 2 * best_maa * best_cluster_score / (best_maa + best_cluster_score) if (best_maa + best_cluster_score) > 0 else 0\n                })\n        \n        # Calculate dataset-level metrics\n        if scene_metrics:\n            avg_maa = np.mean([m['maa'] for m in scene_metrics])\n            avg_cluster_score = np.mean([m['cluster_score'] for m in scene_metrics])\n            avg_f1 = np.mean([m['f1_score'] for m in scene_metrics])\n            \n            dataset_scores.append({\n                'dataset': dataset,\n                'avg_maa': avg_maa,\n                'avg_cluster_score': avg_cluster_score,\n                'avg_f1_score': avg_f1\n            })\n            \n            print(f\"Dataset '{dataset}' metrics:\")\n            print(f\"  Average mAA (recall): {avg_maa:.4f}\")\n            print(f\"  Average Clustering Score (precision): {avg_cluster_score:.4f}\")\n            print(f\"  Average F1 Score: {avg_f1:.4f}\")\n    \n    # Calculate overall score\n    if dataset_scores:\n        overall_f1 = np.mean([d['avg_f1_score'] for d in dataset_scores])\n        print(f\"\\nOverall F1 Score: {overall_f1:.4f}\")\n    \n    return dataset_scores\n\n\n# Main Function\n# ============\n\ndef main(base_path, output_path='submission.csv'):\n    \"\"\"\n    Main function to run the entire pipeline\n    \"\"\"\n    # Step 1: Explore competition data\n    sample_submission, train_thresholds, train_labels, datasets = explore_competition_data(base_path)\n    \n    # Step 2: Load training and test data\n    training_data = load_training_data(base_path, train_labels)\n    test_data = load_test_data(base_path, datasets)\n    \n    # Results for each test dataset\n    test_results = {}\n    \n    # Process each dataset\n    for dataset in datasets:\n        print(f\"\\n{'=' * 40}\")\n        print(f\"Processing dataset: {dataset}\")\n        print(f\"{'=' * 40}\")\n        \n        if dataset not in test_data:\n            print(f\"No test data found for dataset '{dataset}', skipping\")\n            continue\n        \n        # Get test images for this dataset\n        test_images = test_data[dataset]\n        \n        # Step 3: Extract features\n        print(\"\\nExtracting features for test images...\")\n        test_features = extract_features_for_dataset(test_images)\n        \n        # Step 4: Build similarity matrix\n        print(\"\\nBuilding similarity matrix...\")\n        similarity_matrix, image_files = build_similarity_matrix(test_features)\n        \n        # Step 5: Cluster images\n        print(\"\\nClustering images...\")\n        clusters, outliers = cluster_images(similarity_matrix, image_files)\n        \n        # Step 6: Estimate poses\n        print(\"\\nEstimating camera poses...\")\n        poses = estimate_poses(test_features, clusters)\n        \n        # Store results\n        test_results[dataset] = {\n            'clusters': clusters,\n            'outliers': outliers,\n            'poses': poses\n        }\n    \n    # Step 7: Create submission\n    print(\"\\nCreating submission file...\")\n    submission_df = create_submission(test_results, output_path)\n    \n    return submission_df\n\n\n# Example usage\nif __name__ == \"__main__\":\n    # Path to competition data\n    base_path = \"/kaggle/input/image-matching-challenge-2025\"\n    \n    # Run the pipeline\n    submission_df = main(base_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T00:23:09.367262Z","iopub.execute_input":"2025-05-01T00:23:09.367589Z","iopub.status.idle":"2025-05-01T00:23:45.861175Z","shell.execute_reply.started":"2025-05-01T00:23:09.367571Z","shell.execute_reply":"2025-05-01T00:23:45.860448Z"}},"outputs":[],"execution_count":null}]}