{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91498,"databundleVersionId":11655853,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#LOADING THE DATASET\n\nimport os #For file and directory operations (ex: checking dataset structure)\n\n# List all files in dataset directory\ndataset_path = \"/kaggle/input/\"\nos.listdir(dataset_path)\n#Just to ensure that the dataset is already there! \n#(Kaggle usually mounts the competition dataset automatically in /kaggle/input/)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:32:35.572106Z","iopub.execute_input":"2025-04-16T18:32:35.572410Z","iopub.status.idle":"2025-04-16T18:32:35.588296Z","shell.execute_reply.started":"2025-04-16T18:32:35.572381Z","shell.execute_reply":"2025-04-16T18:32:35.587026Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#EXPLORING THE DATASET STRUCTURE\n\ndataset_path = \"/kaggle/input/image-matching-challenge-2025\"\n\n# Check files inside (list the available folders)\nos.listdir(dataset_path)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:32:44.231514Z","iopub.execute_input":"2025-04-16T18:32:44.231868Z","iopub.status.idle":"2025-04-16T18:32:44.239570Z","shell.execute_reply.started":"2025-04-16T18:32:44.231845Z","shell.execute_reply":"2025-04-16T18:32:44.238845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Importing necessary libraries \n\nimport numpy as np  # For numerical operations (arrays, math, etc.)\nimport pandas as pd  # For handling CSV files (reading, writing, manipulating tables)\nimport matplotlib.pyplot as plt  # For visualizing images\nfrom PIL import Image  # To open and inspect images","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:33:07.610770Z","iopub.execute_input":"2025-04-16T18:33:07.611063Z","iopub.status.idle":"2025-04-16T18:33:07.956061Z","shell.execute_reply.started":"2025-04-16T18:33:07.611042Z","shell.execute_reply":"2025-04-16T18:33:07.955047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Loading train_labels.csv file\ntrain_path = \"../input/image-matching-challenge-2025/train\" \ntrain_labels = pd.read_csv('/kaggle/input/image-matching-challenge-2025/train_labels.csv')\n\n# Displaying first few rows of it\ntrain_labels.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:33:17.271100Z","iopub.execute_input":"2025-04-16T18:33:17.271544Z","iopub.status.idle":"2025-04-16T18:33:17.332166Z","shell.execute_reply.started":"2025-04-16T18:33:17.271519Z","shell.execute_reply":"2025-04-16T18:33:17.331291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:33:32.503019Z","iopub.execute_input":"2025-04-16T18:33:32.503799Z","iopub.status.idle":"2025-04-16T18:33:32.527830Z","shell.execute_reply.started":"2025-04-16T18:33:32.503763Z","shell.execute_reply":"2025-04-16T18:33:32.526783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_scenes = train_labels[\"scene\"].nunique()\nnum_scenes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:33:47.862886Z","iopub.execute_input":"2025-04-16T18:33:47.863201Z","iopub.status.idle":"2025-04-16T18:33:47.870248Z","shell.execute_reply.started":"2025-04-16T18:33:47.863179Z","shell.execute_reply":"2025-04-16T18:33:47.869263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scene_counts = train_labels[\"scene\"].value_counts()\nscene_counts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:33:57.542853Z","iopub.execute_input":"2025-04-16T18:33:57.543178Z","iopub.status.idle":"2025-04-16T18:33:57.551823Z","shell.execute_reply.started":"2025-04-16T18:33:57.543153Z","shell.execute_reply":"2025-04-16T18:33:57.550665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_datasets = train_labels[\"dataset\"].nunique()\nnum_datasets","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:34:09.231027Z","iopub.execute_input":"2025-04-16T18:34:09.231365Z","iopub.status.idle":"2025-04-16T18:34:09.238684Z","shell.execute_reply.started":"2025-04-16T18:34:09.231339Z","shell.execute_reply":"2025-04-16T18:34:09.237489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_counts = train_labels[\"dataset\"].value_counts()\ndataset_counts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:34:18.655017Z","iopub.execute_input":"2025-04-16T18:34:18.655671Z","iopub.status.idle":"2025-04-16T18:34:18.663478Z","shell.execute_reply.started":"2025-04-16T18:34:18.655639Z","shell.execute_reply":"2025-04-16T18:34:18.662510Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\n\n# Counting images per dataset\ndataset_counts = train_labels[\"dataset\"].value_counts()\n\n# Plotting the bar chart\nplt.figure(figsize=(12, 6))  # Set figure size\nsns.barplot(x=dataset_counts.index, y=dataset_counts.values, palette=\"viridis\")  # Use seaborn for better styling\n\n# Adding labels and title\nplt.xlabel(\"Dataset\", fontsize=14)\nplt.ylabel(\"Number of Images\", fontsize=14)\nplt.title(\"Number of Images per Dataset\", fontsize=16)\nplt.xticks(rotation=45, ha=\"right\")  # Rotate x-axis labels for better readability\nplt.grid(axis=\"y\", linestyle=\"--\", alpha=0.7)  # Add a light grid for clarity\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:34:34.770893Z","iopub.execute_input":"2025-04-16T18:34:34.771239Z","iopub.status.idle":"2025-04-16T18:34:36.179335Z","shell.execute_reply.started":"2025-04-16T18:34:34.771215Z","shell.execute_reply":"2025-04-16T18:34:36.178355Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_thresholds_path = \"../input/image-matching-challenge-2025/train_thresholds.csv\"\ntrain_thresholds = pd.read_csv(train_thresholds_path)\ntrain_thresholds.head() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:34:51.651290Z","iopub.execute_input":"2025-04-16T18:34:51.651750Z","iopub.status.idle":"2025-04-16T18:34:51.666903Z","shell.execute_reply.started":"2025-04-16T18:34:51.651726Z","shell.execute_reply":"2025-04-16T18:34:51.666006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_thresholds.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:35:09.747358Z","iopub.execute_input":"2025-04-16T18:35:09.748138Z","iopub.status.idle":"2025-04-16T18:35:09.761798Z","shell.execute_reply.started":"2025-04-16T18:35:09.748102Z","shell.execute_reply":"2025-04-16T18:35:09.760887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# Step 1: Convert the \"thresholds\" column into a list of lists (split by \";\")\nthreshold_lists = train_thresholds[\"thresholds\"].apply(lambda x: list(map(float, x.split(\";\"))))\n\n# Step 2: Flatten the list (convert list of lists into a single list)\nall_thresholds = np.concatenate(threshold_lists.values)\n\n# Step 3: Plot the histogram\nplt.figure(figsize=(8, 5))\nplt.hist(all_thresholds, bins=30, edgecolor=\"black\")\nplt.xlabel(\"Score Threshold\")\nplt.ylabel(\"Count\")\nplt.title(\"Distribution of Similarity Thresholds\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:35:21.067076Z","iopub.execute_input":"2025-04-16T18:35:21.067676Z","iopub.status.idle":"2025-04-16T18:35:21.394756Z","shell.execute_reply.started":"2025-04-16T18:35:21.067650Z","shell.execute_reply":"2025-04-16T18:35:21.393859Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2  # OpenCV library for image processing\n\n# Pick a scene from the dataset\nscene_name = \"fountain\"  \nscene_images = train_labels[train_labels[\"scene\"] == scene_name][\"image\"].values[:2]  # Picked two images\n\n# Load and display images\nfig, axes = plt.subplots(1, 2, figsize=(10, 5))\n\nfor i, img_name in enumerate(scene_images):\n    img_path = os.path.join(train_path, train_labels[train_labels[\"scene\"] == scene_name][\"dataset\"].values[0], img_name)\n    img = cv2.imread(img_path, cv2.IMREAD_GRAYSCALE)  # Loaded image in grayscale (easier for feature matching)\n    \n    axes[i].imshow(img, cmap=\"gray\")\n    axes[i].set_title(f\"Image: {img_name}\")\n    axes[i].axis(\"off\")\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:35:38.346599Z","iopub.execute_input":"2025-04-16T18:35:38.346984Z","iopub.status.idle":"2025-04-16T18:35:39.838856Z","shell.execute_reply.started":"2025-04-16T18:35:38.346950Z","shell.execute_reply":"2025-04-16T18:35:39.837840Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img1 = cv2.imread(os.path.join(train_path, train_labels[train_labels[\"scene\"] == scene_name][\"dataset\"].values[0], scene_images[0]), cv2.IMREAD_GRAYSCALE)\nimg2 = cv2.imread(os.path.join(train_path, train_labels[train_labels[\"scene\"] == scene_name][\"dataset\"].values[0], scene_images[1]), cv2.IMREAD_GRAYSCALE)\n\n# Initialize ORB detector\norb = cv2.ORB_create()\n\n# Detect keypoints and descriptors\nkp1, des1 = orb.detectAndCompute(img1, None)\nkp2, des2 = orb.detectAndCompute(img2, None)\n\n# Initialize Brute-Force Matcher and match descriptors\nbf = cv2.BFMatcher(cv2.NORM_HAMMING, crossCheck=True)\nmatches = bf.match(des1, des2)\n\n# Sort matches by distance (lower distance = better match)\nmatches = sorted(matches, key=lambda x: x.distance)\n\n# Draw matches\nmatch_img = cv2.drawMatches(img1, kp1, img2, kp2, matches[:50], None, flags=cv2.DrawMatchesFlags_NOT_DRAW_SINGLE_POINTS)\n\nplt.figure(figsize=(12, 6))\nplt.imshow(match_img)\nplt.title(\"Feature Matching using ORB\")\nplt.axis(\"off\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:36:09.770829Z","iopub.execute_input":"2025-04-16T18:36:09.771222Z","iopub.status.idle":"2025-04-16T18:36:11.222269Z","shell.execute_reply.started":"2025-04-16T18:36:09.771191Z","shell.execute_reply":"2025-04-16T18:36:11.221112Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_submission = pd.read_csv('/kaggle/input/image-matching-challenge-2025/sample_submission.csv')\n#sample_submission.head()\n#checking column named to have in dummy submission\n\n# Create dummy values\nsample_submission[\"rotation_matrix\"] = \"1;0;0;0;1;0;0;0;1\"  # Identity matrix as a placeholder\nsample_submission[\"translation_vector\"] = \"0;0;0\"  # Zero translation\n\n# Save the dummy submission file\nsample_submission.to_csv(\"submission.csv\", index=False)\n\nprint(\"Submission file created successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:36:41.771465Z","iopub.execute_input":"2025-04-16T18:36:41.772355Z","iopub.status.idle":"2025-04-16T18:36:41.814975Z","shell.execute_reply.started":"2025-04-16T18:36:41.772325Z","shell.execute_reply":"2025-04-16T18:36:41.814122Z"}},"outputs":[],"execution_count":null}]}