{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":71885,"databundleVersionId":8015523,"sourceType":"competition"},{"sourceId":7884485,"sourceType":"datasetVersion","datasetId":4628051},{"sourceId":7884725,"sourceType":"datasetVersion","datasetId":4628331}],"dockerImageVersionId":30674,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install --no-index /kaggle/input/imc2024-packages-lightglue-rerun-kornia/* --no-deps","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom PIL import Image\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\n\nfrom pathlib import Path\n\nimport cv2\nimport plotly.express as px\nimport plotly.graph_objects as go\nimport pycolmap\n\nimport io\nimport re\nimport zipfile\nfrom argparse import ArgumentParser\nfrom pathlib import Path\nfrom typing import Final\n\nimport numpy.typing as npt\nimport requests\n\nimport rerun as rr\nrr.init(\"IMC lizard\")\n\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2024-03-27T21:55:00.370471Z","iopub.execute_input":"2024-03-27T21:55:00.371204Z","iopub.status.idle":"2024-03-27T21:55:04.663858Z","shell.execute_reply.started":"2024-03-27T21:55:00.371170Z","shell.execute_reply":"2024-03-27T21:55:04.662959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define paths to your data\ntrain_data_dir = '/kaggle/input/image-matching-challenge-2024/train'\ntest_data_dir = '/kaggle/input/image-matching-challenge-2024/test'\ntrain_labels_file = '/kaggle/input/image-matching-challenge-2024/train/train_labels.csv'\nsample_submission_file = '/kaggle/input/image-matching-challenge-2024/sample_submission.csv'","metadata":{"execution":{"iopub.status.busy":"2024-03-27T21:55:04.665610Z","iopub.execute_input":"2024-03-27T21:55:04.666946Z","iopub.status.idle":"2024-03-27T21:55:04.671466Z","shell.execute_reply.started":"2024-03-27T21:55:04.666910Z","shell.execute_reply":"2024-03-27T21:55:04.670701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.read_csv(train_labels_file)\ntrain_labels.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-27T21:55:06.580888Z","iopub.execute_input":"2024-03-27T21:55:06.581284Z","iopub.status.idle":"2024-03-27T21:55:06.615102Z","shell.execute_reply.started":"2024-03-27T21:55:06.581256Z","shell.execute_reply":"2024-03-27T21:55:06.614202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We will need these paths later on to 3D visualize the temple object.**","metadata":{}},{"cell_type":"code","source":"!ls /kaggle/input/image-matching-challenge-2024/train/pond/smf\npond_smf = '/kaggle/input/image-matching-challenge-2024/train/pond/smf'\npond_imgs = '/kaggle/input/image-matching-challenge-2024/train/pond/images'","metadata":{"execution":{"iopub.status.busy":"2024-03-27T21:55:08.852261Z","iopub.execute_input":"2024-03-27T21:55:08.852630Z","iopub.status.idle":"2024-03-27T21:55:09.805847Z","shell.execute_reply.started":"2024-03-27T21:55:08.852600Z","shell.execute_reply":"2024-03-27T21:55:09.804704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Dataset and Scene columns basically have the same values:","metadata":{}},{"cell_type":"code","source":"train_labels.groupby(\"dataset\")[\"scene\"].nunique()","metadata":{"execution":{"iopub.status.busy":"2024-03-27T21:55:11.167400Z","iopub.execute_input":"2024-03-27T21:55:11.168391Z","iopub.status.idle":"2024-03-27T21:55:11.179745Z","shell.execute_reply.started":"2024-03-27T21:55:11.168354Z","shell.execute_reply":"2024-03-27T21:55:11.178753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_counts = train_labels[\"dataset\"].value_counts()\ncolors = [\"gold\", \"mediumturquoise\", \"darkorange\", \"lightgreen\", \"red\", \"pink\", \"darkblue\"]\n\nfig = go.Figure(\n    data=[\n        go.Pie(\n            labels=df_counts.index,\n            values=df_counts.values,\n            textfont_size=20,\n            marker=dict(colors=colors, pattern=dict(shape=[\".\", \"x\", \"+\", \"-\", \"|\", \"\\\\\", \"/\"]))\n        )\n    ]\n)\nfig.update_layout(\n    title=\"Pie distribution of dataset images\",\n    legend_title_text='Dataset names:'\n)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-27T21:55:13.176904Z","iopub.execute_input":"2024-03-27T21:55:13.177933Z","iopub.status.idle":"2024-03-27T21:55:13.328418Z","shell.execute_reply.started":"2024-03-27T21:55:13.177890Z","shell.execute_reply":"2024-03-27T21:55:13.327538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_categories = pd.read_csv(\"/kaggle/input/image-matching-challenge-2024/train/categories.csv\")\ntrain_categories.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-27T21:55:15.354810Z","iopub.execute_input":"2024-03-27T21:55:15.355974Z","iopub.status.idle":"2024-03-27T21:55:15.369298Z","shell.execute_reply.started":"2024-03-27T21:55:15.355922Z","shell.execute_reply":"2024-03-27T21:55:15.368413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Let's separate categories, so it will be one category per scene (they are separated with \";\").**","metadata":{}},{"cell_type":"code","source":"train_categories[\"category\"] = train_categories[\"categories\"].str.split(\";\")\ntrain_categories = train_categories.explode(\"category\")\ntrain_categories.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-27T21:55:17.518596Z","iopub.execute_input":"2024-03-27T21:55:17.519419Z","iopub.status.idle":"2024-03-27T21:55:17.534313Z","shell.execute_reply.started":"2024-03-27T21:55:17.519386Z","shell.execute_reply":"2024-03-27T21:55:17.533317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Let's see the distribution of categories in datasets.**","metadata":{}},{"cell_type":"code","source":"category_counts = train_categories[\"category\"].value_counts()\n\nfig = go.Figure(\n    data=[\n        go.Pie(\n            labels=category_counts.index,\n            values=df_counts.values,\n            textfont_size=20,\n            marker=dict(colors=colors, pattern=dict(shape=[\".\", \"x\", \"+\", \"-\", \"|\", \"\\\\\", \"/\"]))\n        )\n    ]\n)\nfig.update_layout(\n    title=\"Distribution of categories in datasets\",\n    legend_title_text='Categories:'\n)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-27T21:55:19.535213Z","iopub.execute_input":"2024-03-27T21:55:19.536105Z","iopub.status.idle":"2024-03-27T21:55:19.551662Z","shell.execute_reply.started":"2024-03-27T21:55:19.536070Z","shell.execute_reply":"2024-03-27T21:55:19.550701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Now let's visualize the relationship between datasets(i.e. scenes) and categories.**","metadata":{}},{"cell_type":"code","source":"fig = px.sunburst(train_categories, path=['scene', 'category'] )\nfig.update_layout(\n    title=\"Scene and category relation\",\n)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-27T21:55:22.350724Z","iopub.execute_input":"2024-03-27T21:55:22.351120Z","iopub.status.idle":"2024-03-27T21:55:22.625931Z","shell.execute_reply.started":"2024-03-27T21:55:22.351089Z","shell.execute_reply":"2024-03-27T21:55:22.625085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DESCRIPTION = \"\"\"\n# Sparse Reconstruction by COLMAP\nThis example was generated from the output of a sparse reconstruction done with COLMAP.\n\n[COLMAP](https://colmap.github.io/index.html) is a general-purpose Structure-from-Motion (SfM) and Multi-View Stereo\n(MVS) pipeline with a graphical and command-line interface.\n\nIn this example a short video clip has been processed offline by the COLMAP pipeline, and we use Rerun to visualize the\nindividual camera frames, estimated camera poses, and resulting point clouds over time.\n\n## How it was made\nThe full source code for this example is available\n[on GitHub](https://github.com/rerun-io/rerun/blob/latest/examples/python/structure_from_motion/main.py).\n\n### Images\nThe images are logged through the [rr.Image archetype](https://www.rerun.io/docs/reference/types/archetypes/image)\nto the [camera/image entity](recording://camera/image).\n\n### Cameras\nThe images stem from pinhole cameras located in the 3D world. To visualize the images in 3D, the pinhole projection has\nto be logged and the camera pose (this is often referred to as the intrinsics and extrinsics of the camera,\nrespectively).\n\nThe [rr.Pinhole archetype](https://www.rerun.io/docs/reference/types/archetypes/pinhole) is logged to\nthe [camera/image entity](recording://camera/image) and defines the intrinsics of the camera. This defines how to go\nfrom the 3D camera frame to the 2D image plane. The extrinsics are logged as an\n[rr.Transform3D archetype](https://www.rerun.io/docs/reference/types/archetypes/transform3d) to the\n[camera entity](recording://camera).\n\n### Reprojection error\nFor each image a [rr.Scalar archetype](https://www.rerun.io/docs/reference/types/archetypes/scalar)\ncontaining the average reprojection error of the keypoints is logged to the\n[plot/avg_reproj_err entity](recording://plot/avg_reproj_err).\n\n### 2D points\nThe 2D image points that are used to triangulate the 3D points are visualized by logging\n[rr.Points3D archetype](https://www.rerun.io/docs/reference/types/archetypes/points2d)\nto the [camera/image/keypoints entity](recording://camera/image/keypoints). Note that these keypoints are a child of the\n[camera/image entity](recording://camera/image), since the points should show in the image plane.\n\n### Colored 3D points\nThe colored 3D points were added to the scene by logging the\n[rr.Points3D archetype](https://www.rerun.io/docs/reference/types/archetypes/points3d)\nto the [points entity](recording://points):\n```python\nrr.log(\"points\", rr.Points3D(points, colors=point_colors), rr.AnyValues(error=point_errors))\n```\n**Note:** we added some [custom per-point errors](recording://points) that you can see when you\nhover over the points in the 3D view.\n\"\"\".strip()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-03-27T21:55:24.684854Z","iopub.execute_input":"2024-03-27T21:55:24.685701Z","iopub.status.idle":"2024-03-27T21:55:24.692777Z","shell.execute_reply.started":"2024-03-27T21:55:24.685668Z","shell.execute_reply":"2024-03-27T21:55:24.691623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FILTER_MIN_VISIBLE = 60\ndef scale_camera(camera, resize: tuple[int, int]) -> tuple[pycolmap.Camera, npt.NDArray[np.float_]]:\n    \"\"\"Scale the camera intrinsics to match the resized image.\"\"\"\n    assert camera.model == \"PINHOLE\"\n    new_width = resize[0]\n    new_height = resize[1]\n    scale_factor = np.array([new_width / camera.width, new_height / camera.height])\n\n    # For PINHOLE camera model, params are: [focal_length_x, focal_length_y, principal_point_x, principal_point_y]\n    new_params = np.append(camera.params[:2] * scale_factor, camera.params[2:] * scale_factor)\n\n    return (pycolmap.Camera(camera.id, camera.model, new_width, new_height, new_params), scale_factor)\n\n\n\ndef read_and_log_sparse_reconstruction(rec_path: Path, img_path: Path, filter_output: bool, resize: tuple[int, int] | None) -> None:\n    print(\"Reading sparse COLMAP reconstruction\")\n    reconstruction = pycolmap.Reconstruction(rec_path)\n    cameras = reconstruction.cameras\n    images = reconstruction.images\n    points3D = reconstruction.points3D\n    print(\"Building visualization by logging to Rerun\")\n\n    if filter_output:\n        # Filter out noisy points\n        points3D = {id: point for id, point in points3D.items() if point.color.any() and len(point.image_ids) > 2}\n\n    rr.log(\"description\", rr.TextDocument(DESCRIPTION, media_type=rr.MediaType.MARKDOWN), timeless=True)\n    rr.log(\"/\", rr.ViewCoordinates.RIGHT_HAND_Y_DOWN, timeless=True)\n    rr.log(\"plot/avg_reproj_err\", rr.SeriesLine(color=[240, 45, 58]), timeless=True)\n\n    # Iterate through images (video frames) logging data related to each frame.\n    ii=0\n    for image in tqdm(sorted(images.values(), key=lambda im: im.name)):  # type: ignore[no-any-return]\n        image_file = img_path / image.name.replace('.jpg', '.png')\n        if not os.path.exists(image_file):\n            continue\n        #print (image_file)\n\n        # COLMAP sets image ids that don't match the original video frame\n        idx_match = re.search(r\"\\d+\", image.name)\n        assert idx_match is not None\n        frame_idx = int(idx_match.group(0))\n\n\n        camera = cameras[image.camera_id]\n        if resize:\n            camera, scale_factor = scale_camera(camera, resize)\n        else:\n            scale_factor = np.array([1.0, 1.0])\n\n        visible_ids = [id_ for id_ in points3D.keys() if image.has_point3D(id_) ]\n        \n\n        if filter_output and len(visible_ids) < FILTER_MIN_VISIBLE:\n            continue\n\n        visible_xyzs = [points3D[idx] for idx in visible_ids]\n        visible_xys = np.array([x.xy for x in image.get_valid_points2D()])\n        if resize:\n            visible_xys *= scale_factor\n\n        rr.set_time_sequence(\"frame\", frame_idx)\n        try:\n            points = [point.xyz for point in visible_xyzs]\n        except Exception as e:\n            print (e)\n            continue\n        point_colors = [point.color for point in visible_xyzs]\n        point_errors = [point.error for point in visible_xyzs]\n\n        rr.log(\"plot/avg_reproj_err\", rr.Scalar(np.mean(point_errors)))\n\n        rr.log(\"points\", rr.Points3D(points, colors=point_colors), rr.AnyValues(error=point_errors))\n\n        # COLMAP's camera transform is \"camera from world\"\n        rr.log(\n            \"camera\", rr.Transform3D(translation=image.cam_from_world.translation,\n                                     rotation=rr.Quaternion(xyzw=image.cam_from_world.rotation.quat), from_parent=True)\n        )\n        rr.log(\"camera\", rr.ViewCoordinates.RDF, timeless=True)  # X=Right, Y=Down, Z=Forward\n\n        # Log camera intrinsics\n        assert str(camera.model) in  [\"CameraModelId.SIMPLE_PINHOLE\", \"CameraModelId.PINHOLE\"]\n        if str(camera.model) == \"CameraModelId.SIMPLE_PINHOLE\":\n            rr.log(\n                \"camera/image\",\n                rr.Pinhole(\n                    resolution=[camera.width, camera.height],\n                    focal_length=[camera.params[0], camera.params[0]],\n                    principal_point=camera.params[1:],\n                ),\n            )\n        else:\n            rr.log(\n                \"camera/image\",\n                rr.Pinhole(\n                    resolution=[camera.width, camera.height],\n                    focal_length=camera.params[:2],\n                    principal_point=camera.params[2:],\n                ),\n            )\n        if resize:\n            bgr = cv2.imread(str(image_file))\n            bgr = cv2.resize(bgr, resize)\n            rgb = cv2.cvtColor(bgr, cv2.COLOR_BGR2RGB)\n            rr.log(\"camera/image\", rr.Image(rgb).compress(jpeg_quality=75))\n        else:\n            rr.log(\"camera/image\", rr.ImageEncoded(path=img_path / image.name.replace('.jpg', '.png')))\n        rr.log(\"camera/image/keypoints\", rr.Points2D(visible_xys, colors=[34, 138, 167]))\n    print (\"Now preparing visualization engine\")\n# Whatever you want to visualize in notebook, you should start the rec = rr.memory_recording()\nrec = rr.memory_recording()\nread_and_log_sparse_reconstruction(Path(pond_smf), Path(pond_imgs), filter_output=False, resize=None)\n# And after it finishes - show it by just calling it  rec\nrec","metadata":{"execution":{"iopub.status.busy":"2024-03-27T21:55:26.749773Z","iopub.execute_input":"2024-03-27T21:55:26.750122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Let's implement the loss function to use it later to train the model.**","metadata":{}},{"cell_type":"code","source":"def calculate_mAA(predictions, ground_truths, thresholds):\n    total_mAA = 0.0\n    num_scenes = len(predictions)\n    \n    for i in range(num_scenes):\n        scene_predictions = predictions[i]\n        scene_ground_truths = ground_truths[i]\n        num_cameras = len(scene_predictions)\n        \n        mAA_sum = 0.0\n        \n        for threshold in thresholds:\n            registered_count = 0\n            for j in range(num_cameras):\n                camera_pred = scene_predictions[j]\n                camera_gt = scene_ground_truths[j]\n                error = np.linalg.norm(camera_pred - camera_gt)\n                if error < threshold:\n                    registered_count += 1\n            mAA_sum += registered_count / num_cameras\n        \n        mAA = mAA_sum / len(thresholds)\n        total_mAA += mAA\n    \n    return total_mAA / num_scenes","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Gonna continue with dataset preparation and model arrchitecture and training. Stay tuned loves**","metadata":{}},{"cell_type":"code","source":"class CustomDataset(Dataset):\n    def __init__(self, root_dir, object_list, transform=None):\n        self.root_dir = root_dir\n        self.object_list = object_list\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.object_list)\n\n    def __getitem__(self, idx):\n        object_name = self.object_list[idx]\n\n        # Load images\n        image_dir = os.path.join(self.root_dir, object_name, 'images')\n        image_files = os.listdir(image_dir)\n        images = [cv2.imread(os.path.join(image_dir, img)) for img in image_files]\n\n        # Load SfM data\n        smf_dir = os.path.join(self.root_dir, object_name, 'smf')\n\n        # Load camera poses\n        camera_poses = self.load_sfm_data(os.path.join(smf_dir, 'cameras'))\n        \n        # Load 3D points\n        points_3d = self.load_sfm_data(os.path.join(smf_dir, 'points3D'))\n\n        sample = {'images': images, 'camera_poses': camera_poses, 'points_3d': points_3d}\n\n        if self.transform:\n            sample = self.transform(sample)\n\n        return sample\n\n    def load_sfm_data(self, file_path):\n        if os.path.exists(file_path):\n            if file_path.endswith('.txt'):\n                return np.loadtxt(file_path)\n            elif file_path.endswith('.bin'):\n                return np.fromfile(file_path, dtype=np.float32)\n\n        return None\n\n# Example usage\nroot_dir = 'path/to/data'\nobject_list = ['pond', 'lizard', 'church', 'multi-temporal-temple-baalshamin', 'dioscuri', 'transp_obj_glass_cup', 'transp_obj_glass_cylinder']\n\n# Define any additional preprocessing or augmentation transforms if needed\ntransform = transforms.Compose([\n    # Add any transforms here\n])\n\n# Create the dataset\ndataset = CustomDataset(root_dir, object_list, transform=transform)","metadata":{},"execution_count":null,"outputs":[]}]}