{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91498,"databundleVersionId":11655853,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os \nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\ndataset_path = \"/kaggle/input/\"\nos.listdir(dataset_path)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-14T13:40:55.159367Z","iopub.execute_input":"2025-04-14T13:40:55.159708Z","iopub.status.idle":"2025-04-14T13:40:55.167551Z","shell.execute_reply.started":"2025-04-14T13:40:55.159684Z","shell.execute_reply":"2025-04-14T13:40:55.166201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#EXPLORING THE DATASET STRUCTURE\n\ndataset_path = \"/kaggle/input/image-matching-challenge-2025\"\n\n# Check files inside (list the available folders)\nos.listdir(dataset_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T13:40:55.169837Z","iopub.execute_input":"2025-04-14T13:40:55.170170Z","iopub.status.idle":"2025-04-14T13:40:55.188319Z","shell.execute_reply.started":"2025-04-14T13:40:55.170143Z","shell.execute_reply":"2025-04-14T13:40:55.187303Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"EXPLORATORY DATA ANALYSIS","metadata":{}},{"cell_type":"code","source":"# Importing necessary libraries \n\nimport numpy as np  # For numerical operations (arrays, math, etc.)\nimport pandas as pd  # For handling CSV files (reading, writing, manipulating tables)\nimport matplotlib.pyplot as plt  # For visualizing images\nfrom PIL import Image  # To open and inspect images","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T13:40:55.189231Z","iopub.execute_input":"2025-04-14T13:40:55.189475Z","iopub.status.idle":"2025-04-14T13:40:55.206555Z","shell.execute_reply.started":"2025-04-14T13:40:55.189456Z","shell.execute_reply":"2025-04-14T13:40:55.205604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Loading train_labels.csv file\ntrain_path = \"../input/image-matching-challenge-2025/train\" \ntrain_labels = pd.read_csv('/kaggle/input/image-matching-challenge-2025/train_labels.csv')\n\n# Displaying first few rows of it\ntrain_labels.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T13:40:55.208553Z","iopub.execute_input":"2025-04-14T13:40:55.208907Z","iopub.status.idle":"2025-04-14T13:40:55.242985Z","shell.execute_reply.started":"2025-04-14T13:40:55.208887Z","shell.execute_reply":"2025-04-14T13:40:55.241937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check dataset info (column names, data types, missing values)\ntrain_labels.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T13:40:55.243936Z","iopub.execute_input":"2025-04-14T13:40:55.244146Z","iopub.status.idle":"2025-04-14T13:40:55.255408Z","shell.execute_reply.started":"2025-04-14T13:40:55.244129Z","shell.execute_reply":"2025-04-14T13:40:55.254377Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#SOME KEY QUESTIONS TO ADDRESS:\n\n# Qn1. How many unique scenes are there?\n\n# Ans. Number of unique scenes\nnum_scenes = train_labels[\"scene\"].nunique()\nnum_scenes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T13:40:55.256604Z","iopub.execute_input":"2025-04-14T13:40:55.256885Z","iopub.status.idle":"2025-04-14T13:40:55.283344Z","shell.execute_reply.started":"2025-04-14T13:40:55.256864Z","shell.execute_reply":"2025-04-14T13:40:55.282210Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Qn2. How many images per scene?\n\n# Ans. Count images per scene\nscene_counts = train_labels[\"scene\"].value_counts()\nscene_counts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T13:40:55.284469Z","iopub.execute_input":"2025-04-14T13:40:55.284801Z","iopub.status.idle":"2025-04-14T13:40:55.411597Z","shell.execute_reply.started":"2025-04-14T13:40:55.284773Z","shell.execute_reply":"2025-04-14T13:40:55.410397Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Qn3. How many datasets (different sources of images)?\n\n#Ans. \n# Number of unique datasets\nnum_datasets = train_labels[\"dataset\"].nunique()\nnum_datasets","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T13:40:55.412599Z","iopub.execute_input":"2025-04-14T13:40:55.413230Z","iopub.status.idle":"2025-04-14T13:40:55.428462Z","shell.execute_reply.started":"2025-04-14T13:40:55.413171Z","shell.execute_reply":"2025-04-14T13:40:55.427593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count images per dataset\ndataset_counts = train_labels[\"dataset\"].value_counts()\ndataset_counts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T13:40:55.429464Z","iopub.execute_input":"2025-04-14T13:40:55.429750Z","iopub.status.idle":"2025-04-14T13:40:55.452864Z","shell.execute_reply.started":"2025-04-14T13:40:55.429730Z","shell.execute_reply":"2025-04-14T13:40:55.451835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\n\n# Counting images per dataset\ndataset_counts = train_labels[\"dataset\"].value_counts()\n\n\n# Plotting the bar chart\nplt.figure(figsize=(12, 6))  # Set figure size\nsns.barplot(x=dataset_counts.index, y=dataset_counts.values, palette=\"viridis\")  # Use seaborn for better styling\n\n# Adding labels and title\nplt.xlabel(\"Dataset\", fontsize=14)\nplt.ylabel(\"Number of Images\", fontsize=14)\nplt.title(\"Number of Images per Dataset\", fontsize=16)\nplt.xticks(rotation=45, ha=\"right\")  # Rotate x-axis labels for better readability\nplt.grid(axis=\"y\", linestyle=\"--\", alpha=0.7)  # Add a light grid for clarity\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T13:40:55.455207Z","iopub.execute_input":"2025-04-14T13:40:55.455500Z","iopub.status.idle":"2025-04-14T13:40:55.755828Z","shell.execute_reply.started":"2025-04-14T13:40:55.455479Z","shell.execute_reply":"2025-04-14T13:40:55.754792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_thresholds_path = \"../input/image-matching-challenge-2025/train_thresholds.csv\"\ntrain_thresholds = pd.read_csv(train_thresholds_path)\ntrain_thresholds.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T13:40:55.757296Z","iopub.execute_input":"2025-04-14T13:40:55.757597Z","iopub.status.idle":"2025-04-14T13:40:55.770598Z","shell.execute_reply.started":"2025-04-14T13:40:55.757567Z","shell.execute_reply":"2025-04-14T13:40:55.769626Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Describe thresholds data\ntrain_thresholds.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T13:40:55.771674Z","iopub.execute_input":"2025-04-14T13:40:55.771945Z","iopub.status.idle":"2025-04-14T13:40:55.785948Z","shell.execute_reply.started":"2025-04-14T13:40:55.771925Z","shell.execute_reply":"2025-04-14T13:40:55.784821Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# Step 1: Convert the \"thresholds\" column into a list of lists (split by \";\")\nthreshold_lists = train_thresholds[\"thresholds\"].apply(lambda x: list(map(float, x.split(\";\"))))\n\n# Step 2: Flatten the list (convert list of lists into a single list)\nall_thresholds = np.concatenate(threshold_lists.values)\n\n# Step 3: Plot the histogram\nplt.figure(figsize=(8, 5))\nplt.hist(all_thresholds, bins=30, edgecolor=\"black\")\nplt.xlabel(\"Score Threshold\")\nplt.ylabel(\"Count\")\nplt.title(\"Distribution of Similarity Thresholds\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T13:40:55.787090Z","iopub.execute_input":"2025-04-14T13:40:55.787324Z","iopub.status.idle":"2025-04-14T13:40:56.023992Z","shell.execute_reply.started":"2025-04-14T13:40:55.787305Z","shell.execute_reply":"2025-04-14T13:40:56.022758Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Loading And Visualizing Images","metadata":{}},{"cell_type":"code","source":"import cv2  # OpenCV library for image processing\n\n# Pick a scene from the dataset\nscene_name = \"fountain\"  \nscene_images = train_labels[train_labels[\"scene\"] == scene_name][\"image\"].values[:2]  # Picked two images\n\n# Load and display images\nfig, axes = plt.subplots(1, 2, figsize=(10, 5))\n\nfor i, img_name in enumerate(scene_images):\n    img_path = os.path.join(train_path, train_labels[train_labels[\"scene\"] == scene_name][\"dataset\"].values[0], img_name)\n    img = cv2.imread(img_path, cv2.IMREAD_GRAYSCALE)  # Loaded image in grayscale (easier for feature matching)\n    \n    axes[i].imshow(img, cmap=\"gray\")\n    axes[i].set_title(f\"Image: {img_name}\")\n    axes[i].axis(\"off\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T13:42:12.135767Z","iopub.execute_input":"2025-04-14T13:42:12.136135Z","iopub.status.idle":"2025-04-14T13:42:13.478522Z","shell.execute_reply.started":"2025-04-14T13:42:12.136110Z","shell.execute_reply":"2025-04-14T13:42:13.477356Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"IMAGE MATCHING TECHNIQUES","metadata":{}},{"cell_type":"markdown","source":"Feature Matching using ORB","metadata":{}},{"cell_type":"code","source":"# Load the two images for matching\nimg1 = cv2.imread(os.path.join(train_path, train_labels[train_labels[\"scene\"] == scene_name][\"dataset\"].values[0], scene_images[0]), cv2.IMREAD_GRAYSCALE)\nimg2 = cv2.imread(os.path.join(train_path, train_labels[train_labels[\"scene\"] == scene_name][\"dataset\"].values[0], scene_images[1]), cv2.IMREAD_GRAYSCALE)\n\n# Initialize ORB detector\norb = cv2.ORB_create()\n\n# Detect keypoints and descriptors\nkp1, des1 = orb.detectAndCompute(img1, None)\nkp2, des2 = orb.detectAndCompute(img2, None)\n\n# Initialize Brute-Force Matcher and match descriptors\nbf = cv2.BFMatcher(cv2.NORM_HAMMING, crossCheck=True)\nmatches = bf.match(des1, des2)\n\n# Sort matches by distance (lower distance = better match)\nmatches = sorted(matches, key=lambda x: x.distance)\n\n# Draw matches\nmatch_img = cv2.drawMatches(img1, kp1, img2, kp2, matches[:50], None, flags=cv2.DrawMatchesFlags_NOT_DRAW_SINGLE_POINTS)\n\n# Display the matching result\nplt.figure(figsize=(12, 6))\nplt.imshow(match_img)\nplt.title(\"Feature Matching using ORB\")\nplt.axis(\"off\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T13:43:18.298420Z","iopub.execute_input":"2025-04-14T13:43:18.298856Z","iopub.status.idle":"2025-04-14T13:43:19.670329Z","shell.execute_reply.started":"2025-04-14T13:43:18.298832Z","shell.execute_reply":"2025-04-14T13:43:19.669126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#making a dummy file valid for submission\n\nsample_submission = pd.read_csv('/kaggle/input/image-matching-challenge-2025/sample_submission.csv')\n#sample_submission.head()\n#checking column named to have in dummy submission\n\n# Create dummy values\nsample_submission[\"rotation_matrix\"] = \"1;0;0;0;1;0;0;0;1\"  # Identity matrix as a placeholder\nsample_submission[\"translation_vector\"] = \"0;0;0\"  # Zero translation\n\n# Save the dummy submission file\nsample_submission.to_csv(\"submission.csv\", index=False)\n\nprint(\"Dummy submission file created successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T13:43:55.323321Z","iopub.execute_input":"2025-04-14T13:43:55.323644Z","iopub.status.idle":"2025-04-14T13:43:55.370786Z","shell.execute_reply.started":"2025-04-14T13:43:55.323624Z","shell.execute_reply":"2025-04-14T13:43:55.369740Z"}},"outputs":[],"execution_count":null}]}