{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":71885,"databundleVersionId":8143495,"sourceType":"competition"},{"sourceId":7884485,"sourceType":"datasetVersion","datasetId":4628051},{"sourceId":7975259,"sourceType":"datasetVersion","datasetId":4693319},{"sourceId":8026384,"sourceType":"datasetVersion","datasetId":4726252},{"sourceId":8430989,"sourceType":"datasetVersion","datasetId":4910918},{"sourceId":8469545,"sourceType":"datasetVersion","datasetId":4911029},{"sourceId":8529609,"sourceType":"datasetVersion","datasetId":5060829},{"sourceId":8530773,"sourceType":"datasetVersion","datasetId":4910923},{"sourceId":8571286,"sourceType":"datasetVersion","datasetId":5079869}],"dockerImageVersionId":30674,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install --no-index /kaggle/input/imc2024-packages-lightglue-rerun-kornia/* --no-deps\n!pip install --no-index /kaggle/input/dim-dependencies/* --no-deps\n!mkdir -p /root/.cache/torch/hub/checkpoints\n!cp /kaggle/input/imc2024-official-weights/* /root/.cache/torch/hub/checkpoints","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:19.976944Z","iopub.execute_input":"2025-07-08T05:57:19.977167Z","iopub.status.idle":"2025-07-08T05:57:33.398245Z","shell.execute_reply.started":"2025-07-08T05:57:19.977139Z","shell.execute_reply":"2025-07-08T05:57:33.397069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\nimport os\nimport cv2\nimport sys\nimport torch\nimport numpy as np\nimport pandas as pd\nfrom glob import glob\nfrom tqdm import tqdm\nimport multiprocessing\nimport kornia.feature as KF\nfrom torchmetrics import StructuralSimilarityIndexMeasure\nfrom ortools.constraint_solver import pywrapcp, routing_enums_pb2\n\nsys.path.append('/kaggle/input')\nfrom imc24lightglue import ALIKED\nfrom imc24lightglue.utils import read_image, numpy_image_to_torch\n\nINPUT_ROOT = '/kaggle/input/image-matching-challenge-2024'\nOUTPUT_ROOT = '/kaggle/working'\n\nDEBUG = False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:33.400417Z","iopub.execute_input":"2025-07-08T05:57:33.400701Z","iopub.status.idle":"2025-07-08T05:57:44.865905Z","shell.execute_reply.started":"2025-07-08T05:57:33.400677Z","shell.execute_reply":"2025-07-08T05:57:44.864951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if DEBUG:\n    scenes = [\"transp_obj_glass_cylinder\"]\n    categories_df = pd.read_csv(\"/kaggle/input/image-matching-challenge-2024/train/categories.csv\")\nelse:\n    scenes = [x for x in os.listdir(f\"{INPUT_ROOT}/test/\") if os.path.isdir(f\"{INPUT_ROOT}/test/{x}\")]\n    categories_df = pd.read_csv(\"/kaggle/input/image-matching-challenge-2024/test/categories.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:44.86722Z","iopub.execute_input":"2025-07-08T05:57:44.867736Z","iopub.status.idle":"2025-07-08T05:57:44.895366Z","shell.execute_reply.started":"2025-07-08T05:57:44.867703Z","shell.execute_reply":"2025-07-08T05:57:44.894734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_sizes = [1024, 1280, 1600]\n\naliked_extractor = ALIKED(weights=\"/kaggle/input/imc24lightglue/weights/aliked-n16.pth\").cuda().eval()\n\nlightglue_matcher_params = {\n    \"filter_threshold\": 0.2,\n    \"width_confidence\": -1,\n    \"depth_confidence\": -1,\n    \"mp\": True\n}\nlightglue_matcher = KF.LightGlueMatcher(feature_name=\"aliked\", params=lightglue_matcher_params).cuda().eval()\n\nssim_cuda = StructuralSimilarityIndexMeasure(data_range=255.).cuda()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:44.896216Z","iopub.execute_input":"2025-07-08T05:57:44.896451Z","iopub.status.idle":"2025-07-08T05:57:45.684569Z","shell.execute_reply.started":"2025-07-08T05:57:44.896431Z","shell.execute_reply":"2025-07-08T05:57:45.683621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_rmats(n):\n    ux, uy, uz = 0, 0, 1\n\n    thetas = [(360 / n) * i for i in range(n)]\n    Rmats = []\n    for theta in thetas:\n        costheta = np.cos(np.deg2rad(theta)).round(8)\n        sintheta = np.sin(np.deg2rad(theta)).round(8)\n\n        r00 = costheta + ux**2*(1-costheta)\n        r01 = ux * uy * (1-costheta) - uz*sintheta\n        r02 = ux * uz * (1-costheta) + uy*sintheta\n\n        r10 = uy * ux * (1-costheta) + uz*sintheta\n        r11 = costheta + uy**2*(1-costheta)\n        r12 = uy * uz * (1-costheta) - ux*sintheta\n\n        r20 = uz * ux * (1-costheta) - uy*sintheta\n        r21 = uz * uy * (1-costheta) + ux*sintheta\n        r22 = costheta + uz**2*(1-costheta)\n\n        Rmat = np.array([[r00, r01, r02], [r10, r11, r12], [r20, r21, r22]])\n        Rmats.append(Rmat)\n\n    return Rmats\n\n\ndef get_distance_matrix(fnames, flows):\n    distance_matrix = np.zeros((len(fnames), len(fnames)), dtype=np.int32)\n    for idx1, fname1 in enumerate(fnames):\n        for idx2, fname2 in enumerate(fnames):\n            if idx1 == idx2:\n                value = 0\n            else:\n                value = flows[(fname1, fname2)]\n            distance_matrix[idx1, idx2] = value\n    return distance_matrix\n\n\ndef get_distance_matrix_2(fnames, order_idxs_list):\n\n    neighbors_dict = {idx: [] for idx in range(len(fnames))}\n    for order_idxs in order_idxs_list:\n        for idx in range(len(fnames)):\n            neighbor_left = order_idxs[idx-1]\n            neighbor_right = order_idxs[0] if idx == len(fnames) - 1 else order_idxs[idx+1]\n            neighbors_dict[order_idxs[idx]].extend([neighbor_left, neighbor_right])\n\n    flows = {}\n    for i in range(len(fnames)):\n        for j in range(len(fnames)):\n            if i == j:\n                value = 0\n            else:\n                cnt = neighbors_dict[i].count(j)\n                if cnt == 0:\n                    value = 100000000\n                else:\n                    value = 1000000 // cnt\n            flows[(i, j)] = value\n\n    distance_matrix = np.zeros((len(fnames), len(fnames)), dtype=np.int32)\n    for idx1 in range(len(fnames)):\n        for idx2 in range(len(fnames)):\n            distance_matrix[idx1, idx2] = flows[(idx1, idx2)]\n\n    dummy_distance_matrix = np.zeros((len(fnames)+1, len(fnames)+1), dtype=np.int32)\n    dummy_distance_matrix[1:,1:] = distance_matrix\n    return dummy_distance_matrix\n\n\ndef tsp_distance_callback(from_index, to_index):\n    \"\"\"Returns the distance between the two nodes.\"\"\"\n    from_node = manager.IndexToNode(from_index)\n    to_node = manager.IndexToNode(to_index)\n    return distance_matrix[from_node][to_node]\n\n\ndef get_tsp_solution(manager, routing, solution):\n    \"\"\"Prints solution on console.\"\"\"\n    index = routing.Start(0)\n    idxs = []\n    while not routing.IsEnd(index):\n        idxs.append(manager.IndexToNode(index))\n        index = solution.Value(routing.NextVar(index))\n    return np.array(idxs)\n\n\ndef get_img_pairs_all(fnames):\n    index_pairs = []\n    for i in range(len(fnames)):\n        for j in range(i+1, len(fnames)):\n            index_pairs.append((i,j))\n    return index_pairs\n\n\ndef resize(image, image_size):\n    h, w = image.shape[:2]\n    aspect_ratio = h/w\n    smaller_side_size = int(image_size/max(aspect_ratio, 1/aspect_ratio))\n    if aspect_ratio > 1: # H > W\n        new_size = (image_size, smaller_side_size)\n    else: # H <= W\n        new_size = (smaller_side_size, image_size)\n    image = cv2.resize(image, new_size[::-1])\n    return image, new_size\n\n\ndef alg_inference(cache, fname1, fname2, image_size, rot_code_2):\n\n    # Extraction\n    if 'keypoints_alg' not in cache[fname1][image_size][0]:\n        with torch.inference_mode():\n            tensor = cache[fname1][image_size]['tensor_alg'].cuda()\n            pred = aliked_extractor.extract(tensor, resize=None)\n            cache[fname1][image_size][0] = {\n                **cache[fname1][image_size][0],\n                **{\n                    'keypoints_alg': pred['keypoints'],\n                    'descriptors_alg': pred['descriptors']\n                }\n            }\n\n    if 'keypoints_alg' not in cache[fname2][image_size][rot_code_2]:\n        with torch.inference_mode():\n            tensor = torch.rot90(cache[fname2][image_size]['tensor_alg'], rot_code_2, [1, 2]).cuda()\n            pred = aliked_extractor.extract(tensor, resize=None)\n            cache[fname2][image_size][rot_code_2] = {\n                **cache[fname2][image_size][rot_code_2],\n                **{\n                    'keypoints_alg': pred['keypoints'],\n                    'descriptors_alg': pred['descriptors']\n                }\n            }\n\n    # Matching\n    kpts1, kpts2 = cache[fname1][image_size][0]['keypoints_alg'], cache[fname2][image_size][rot_code_2]['keypoints_alg']\n    with torch.inference_mode():\n        _, indices = lightglue_matcher(\n            cache[fname1][image_size][0]['descriptors_alg'][0], \n            cache[fname2][image_size][rot_code_2]['descriptors_alg'][0],\n            KF.laf_from_center_scale_ori(kpts1),\n            KF.laf_from_center_scale_ori(kpts2)\n        )\n        kpts1 = kpts1[0].cpu().numpy()\n        kpts2 = kpts2[0].cpu().numpy()\n        indices = indices.cpu().numpy()\n        \n    mkpts1 = kpts1[indices[..., 0]].astype(np.float32)\n    mkpts2 = kpts2[indices[..., 1]].astype(np.float32)\n\n    try:\n        _, inliers = cv2.findFundamentalMat(mkpts1, mkpts2, cv2.USAC_MAGSAC, ransacReprojThreshold=5, confidence=0.9999, maxIters=50000)\n        inliers = inliers.ravel() > 0\n        mkpts1 = mkpts1[inliers]\n        mkpts2 = mkpts2[inliers]\n    except:\n        pass\n\n    num_matches = len(mkpts1)\n\n    return num_matches\n\n\ndef matching_inference(fname1, fname2, cache=None):\n\n    for fname in [fname1, fname2]:\n        if fname not in cache:\n            img = read_image(fname)\n            h, w = img.shape[:2]\n            cache[fname] = {'image': img, 'h': h, 'w': w} \n            for image_size in image_sizes:\n                if max(h, w) != image_size:\n                    img_r, (h_r, w_r) = resize(img, image_size)\n                else:\n                    img_r = img.copy()\n                    h_r, w_r = img_r.shape[:2]\n                tensor_alg = numpy_image_to_torch(img_r)\n                cache[fname][image_size] = {'tensor_alg': tensor_alg, 'h_r': h_r, 'w_r': w_r, 0: {}, 1: {}, 2: {}, 3: {}}\n\n    # Co-orientation\n    rot_code_2, max_n = 0, 0\n    for rc2 in range(4):\n        n = alg_inference(cache, fname1, fname2, image_sizes[0], rot_code_2=rc2)\n        if n > max_n:\n            rot_code_2 = rc2\n            max_n = n\n\n    num_matches = max_n\n    for image_size in image_sizes[1:]:\n        num_matches += alg_inference(cache, fname1, fname2, image_size, rot_code_2)\n\n    return num_matches\n\n\ndef get_matching_flows(fnames):\n    index_pairs = get_img_pairs_all(fnames=fnames)\n    cache, flows = {}, {}\n    for pair_idx in tqdm(index_pairs, desc=\"Matching\"):\n        idx1, idx2 = pair_idx\n        fname1, fname2 = fnames[idx1], fnames[idx2]\n        num_matches = matching_inference(fname1, fname2, cache)\n        flows[(fname1, fname2)] = flows[(fname2, fname1)] = int((1 / num_matches) * 1e8)\n    return flows\n\n\ndef compute_ssim(im1, im2):\n    with torch.inference_mode():\n        tensor1 = torch.tensor(im1.copy(), dtype=torch.float32)[None][None].cuda()\n        tensor2 = torch.tensor(im2.copy(), dtype=torch.float32)[None][None].cuda()\n        score = ssim_cuda(tensor1, tensor2).cpu().numpy().item()\n    return score\n\n\ndef get_flows(fnames):\n    relative_orientations = {fnames[0]: 0}\n    flow_images = [cv2.imread(fname, cv2.IMREAD_GRAYSCALE) for fname in fnames]\n    flow_images = [resize(image, 2048)[0] for image in flow_images]\n    ssim_flows = {}\n    for i in tqdm(range(len(fnames)), desc=\"Getting flows\"):\n        image_i = flow_images[i]\n        for j in range(i+1, len(fnames)):\n            image_j = flow_images[j]\n            if i == 0:\n                diffs = []\n                try:\n                    diffs.append(compute_ssim(image_i, image_j))\n                except:\n                    diffs.append(0)\n                for rot_code in range(1,4):\n                    try:\n                        diffs.append(compute_ssim(image_i, np.rot90(image_j, rot_code)))\n                    except:\n                        diffs.append(0)\n                min_diff_idx = np.argmax(diffs)\n                relative_orientations[fnames[j]] = min_diff_idx\n            \n            r1, r2 = relative_orientations[fnames[i]], relative_orientations[fnames[j]]\n            im1 = np.rot90(image_i, r1) if r1 else image_i\n            im2 = np.rot90(image_j, r2) if r2 else image_j\n\n            ssim_score = 1 - compute_ssim(im1, im2)\n            ssim_flows[(fnames[i], fnames[j])] = ssim_flows[(fnames[j], fnames[i])] = ssim_score\n\n    return ssim_flows\n\n\ndef arr_to_str(a):\n    return ';'.join([str(x) for x in a.reshape(-1)])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:45.686135Z","iopub.execute_input":"2025-07-08T05:57:45.68651Z","iopub.status.idle":"2025-07-08T05:57:45.718771Z","shell.execute_reply.started":"2025-07-08T05:57:45.686478Z","shell.execute_reply":"2025-07-08T05:57:45.717896Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results_df = pd.DataFrame(columns=['image_path', 'dataset', 'scene', 'rotation_matrix', 'translation_vector'])\nnon_transparent_scenes = []\nfor scene in tqdm(scenes, desc='Running pipeline'):\n\n    torch.cuda.empty_cache()\n    gc.collect()\n    \n    categories = categories_df[categories_df[\"scene\"]==scene][\"categories\"].item()\n    use_crops = (\"symmetr\" not in categories) and (\"transparen\" not in categories)\n    is_transparent = (\"transparen\" in categories)\n\n    print(f\"{scene=} {categories=} {use_crops=} {is_transparent=}\")\n\n    img_dir = f\"{INPUT_ROOT}/{'train' if DEBUG else 'test'}/{scene}/images\"\n    if not os.path.exists(img_dir):\n        continue\n\n    fnames = sorted(glob(f\"{img_dir}/*\"))\n    \n    if is_transparent:\n        Rmats = get_rmats(len(fnames))\n    \n        ssim_flows = get_flows(fnames)\n        matching_flows = get_matching_flows(fnames)\n\n        ssim_min, ssim_max = min(list(ssim_flows.values())), max(list(ssim_flows.values()))\n        matching_min, matching_max = min(list(matching_flows.values())), max(list(matching_flows.values()))\n\n        ssim_flows = {k: (v - ssim_min) / (ssim_max - ssim_min) for k, v in ssim_flows.items()}\n        matching_flows = {k: (v - matching_min) / (matching_max - matching_min) for k, v in matching_flows.items()}\n\n        flows = {k:int((ssim_flows[k] + matching_flows[k]) * 1e6) for k in ssim_flows.keys()}\n        \n        distance_matrix = get_distance_matrix(fnames, flows)\n        \n        ############# N + 1 #############\n        order_idxs_list = []\n        for start_idx in range(len(fnames)):\n            manager = pywrapcp.RoutingIndexManager(len(distance_matrix), 1, start_idx)\n            routing = pywrapcp.RoutingModel(manager)\n            transit_callback_index = routing.RegisterTransitCallback(tsp_distance_callback)\n            routing.SetArcCostEvaluatorOfAllVehicles(transit_callback_index)\n            search_parameters = pywrapcp.DefaultRoutingSearchParameters()\n            search_parameters.first_solution_strategy = (routing_enums_pb2.FirstSolutionStrategy.PATH_CHEAPEST_ARC)\n            solution = routing.SolveWithParameters(search_parameters)\n            order_idxs = get_tsp_solution(manager, routing, solution)\n            order_idxs_list.append(order_idxs)\n\n        distance_matrix = get_distance_matrix_2(fnames, order_idxs_list)\n\n        manager = pywrapcp.RoutingIndexManager(len(distance_matrix), 1, 0)\n        routing = pywrapcp.RoutingModel(manager)\n        transit_callback_index = routing.RegisterTransitCallback(tsp_distance_callback)\n        routing.SetArcCostEvaluatorOfAllVehicles(transit_callback_index)\n        search_parameters = pywrapcp.DefaultRoutingSearchParameters()\n        search_parameters.first_solution_strategy = (routing_enums_pb2.FirstSolutionStrategy.PATH_CHEAPEST_ARC)\n        solution = routing.SolveWithParameters(search_parameters)\n        order_idxs = get_tsp_solution(manager, routing, solution)\n        order_idxs = order_idxs[1:] - 1\n        ############# N + 1 #############\n    else:\n        non_transparent_scenes.append(scene)\n        continue\n\n    # Create submission\n    for fid, fname in enumerate(fnames):\n        image_id = '/'.join(fname.split('/')[-4:])\n        if is_transparent:\n            R = Rmats[np.where(order_idxs==fid)[0].item()]\n            T = np.array([100, 100, 100])\n        elif image_id in results:\n            R = results[image_id]['R']\n            T = results[image_id]['t']\n        else:\n            R = np.eye(3)\n            T = np.zeros(3)\n\n        new_row = pd.DataFrame({'image_path': image_id,\n                                'dataset': scene,\n                                'scene': scene,\n                                'rotation_matrix': arr_to_str(R),\n                                'translation_vector': arr_to_str(T)}, index=[0])\n\n        results_df = pd.concat([results_df, new_row]).reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:45.72018Z","iopub.execute_input":"2025-07-08T05:57:45.720486Z","iopub.status.idle":"2025-07-08T05:57:45.931914Z","shell.execute_reply.started":"2025-07-08T05:57:45.720458Z","shell.execute_reply":"2025-07-08T05:57:45.930975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results_df.to_csv(f\"{OUTPUT_ROOT}/transp_submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:45.935441Z","iopub.execute_input":"2025-07-08T05:57:45.935705Z","iopub.status.idle":"2025-07-08T05:57:45.942138Z","shell.execute_reply.started":"2025-07-08T05:57:45.935684Z","shell.execute_reply":"2025-07-08T05:57:45.941277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:45.943802Z","iopub.execute_input":"2025-07-08T05:57:45.944096Z","iopub.status.idle":"2025-07-08T05:57:45.957631Z","shell.execute_reply.started":"2025-07-08T05:57:45.944076Z","shell.execute_reply":"2025-07-08T05:57:45.95675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\nwith open(\"/kaggle/working/non_transparent_scenes.json\", \"w\") as f:\n    json.dump(non_transparent_scenes, f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:45.958756Z","iopub.execute_input":"2025-07-08T05:57:45.959247Z","iopub.status.idle":"2025-07-08T05:57:45.963835Z","shell.execute_reply.started":"2025-07-08T05:57:45.959227Z","shell.execute_reply":"2025-07-08T05:57:45.962862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"aliked_extractor.cpu()\nlightglue_matcher.cpu()\nssim_cuda.cpu()\n\ndel aliked_extractor, lightglue_matcher, ssim_cuda\ntorch.cuda.empty_cache()\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:45.964973Z","iopub.execute_input":"2025-07-08T05:57:45.96524Z","iopub.status.idle":"2025-07-08T05:57:46.18537Z","shell.execute_reply.started":"2025-07-08T05:57:45.965219Z","shell.execute_reply":"2025-07-08T05:57:46.184438Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"# No Transparent Part","metadata":{}},{"cell_type":"markdown","source":"## Prepare Inference Scripts","metadata":{}},{"cell_type":"code","source":"!cp -r /kaggle/input/imc2024-inference-scripts /kaggle/working/","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:46.186521Z","iopub.execute_input":"2025-07-08T05:57:46.186859Z","iopub.status.idle":"2025-07-08T05:57:48.234298Z","shell.execute_reply.started":"2025-07-08T05:57:46.186831Z","shell.execute_reply":"2025-07-08T05:57:48.233241Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile imc2024-inference-scripts/config.py\n\nimport os\nimport json\n\nclass Config:\n    input_dir_root = \"/kaggle/input\"\n    output_dir = \"/tmp\"\n    check_exist_dir = False\n    \n    input_csv = \"/kaggle/input/image-matching-challenge-2024/sample_submission.csv\"\n    category_csv = \"/kaggle/input/image-matching-challenge-2024/test/categories.csv\"\n    target_datasets = []\n\n    pipeline_json = \"/kaggle/input/configs-imc2024/1280_2048_crop_1280_2048/pipeline.json\"\n    transparent_pipeline_json = \"/kaggle/input/imc2024-pipelines/lg1280+1536+1024+2048_crop-lg1024+1280+1536/transp_pipeline.json\"\n    \n    colmap_mapper_options = {\n        \"min_model_size\": 3, # By default colmap does not generate a reconstruction if less than 10 images are registered. Lower it to 3.\n        \"max_num_models\": 5,\n        #\"num_threads\": 1,\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:48.235651Z","iopub.execute_input":"2025-07-08T05:57:48.235928Z","iopub.status.idle":"2025-07-08T05:57:48.242605Z","shell.execute_reply.started":"2025-07-08T05:57:48.235905Z","shell.execute_reply":"2025-07-08T05:57:48.24179Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile imc2024-inference-scripts/pipeline.py\n\nimport os\nimport time\nimport pandas as pd\nimport numpy as np\nimport gc\nimport kornia as K\n\nfrom tasks.get_pair.get_image_pair_exhaustive import task_get_image_pair_exhaustive\nfrom tasks.get_pair.get_image_pair_DINO import task_get_image_pair_DINO\nfrom tasks.get_pair.get_image_pair_kNN import task_get_image_pair_kNN\nfrom tasks.get_pair.get_transparent_pair import task_get_transparent_pair\nfrom tasks.matching.matching import task_matching\nfrom tasks.matching.rotate_matching_find_best import task_rotate_matching_find_best\nfrom tasks.crop.sfm_mkpc import task_sfm_mkpc\nfrom tasks.crop.pair_mkpc import task_pair_mkpc\nfrom tasks.crop.transparent_crop import task_transparent_crop\nfrom tasks.utils.ransac import task_ransac\nfrom tasks.utils.concat import task_concat\nfrom tasks.utils.rem_less_match_pair import task_rem_less_match_pair\nfrom tasks.utils.count_matching_num import task_count_matching_num\nfrom tasks.utils.extract_inliner_matching_points import task_extract_inliner_matching_points\nfrom tasks.utils.extract_csv_pair import task_extract_csv_pair\nfrom tasks.utils.estimate_rot import task_estimate_rot\nfrom tasks.utils.get_exif import task_get_exif\n\ntask_map = {\n    \"get_image_pair_exhaustive\": task_get_image_pair_exhaustive,\n    \"get_image_pair_DINO\": task_get_image_pair_DINO,\n    \"get_image_pair_kNN\": task_get_image_pair_kNN,\n    \"get_transparent_pair\": task_get_transparent_pair,\n\n    \"matching\": task_matching,\n    \"rotate_matching_find_best\": task_rotate_matching_find_best,\n\n    \"sfm_mkpc\": task_sfm_mkpc,\n    \"pair_mkpc\": task_pair_mkpc,\n    \"transparent_crop\": task_transparent_crop,\n    \n    \"ransac\": task_ransac,\n    \"concat\": task_concat,\n    \"rem_less_match_pair\": task_rem_less_match_pair,\n    \"count_matching_num\": task_count_matching_num,\n    \"extract_inliner_matching_points\": task_extract_inliner_matching_points,\n    \"extract_csv_pair\": task_extract_csv_pair,\n    \"estimate_rot\": task_estimate_rot,\n    \"get_exif\": task_get_exif,\n}\n\nclass Pipeline():\n    def __init__(self, data_dict, work_dir, input_dir_root, pipeline_config, device_id):\n        self.device = K.utils.get_cuda_device_if_available(device_id)\n        print(f\"device: {self.device}\")\n        self.data_dict = data_dict\n        self.work_dir = work_dir\n        self.input_dir_root = input_dir_root\n        self.pipeline_config = pipeline_config\n        self.processing_times = {\n            \"task\": [],\n            \"comment\": [],\n            \"processing_time\": []\n        }\n\n    def exec(self):\n        all_processing_time = 0\n        for p in self.pipeline_config:\n            task = p[\"task\"]\n            comment = p[\"comment\"]\n            p[\"params\"][\"device\"] = self.device\n            p[\"params\"][\"data_dict\"] = self.data_dict\n            p[\"params\"][\"work_dir\"] = self.work_dir\n            p[\"params\"][\"input_dir_root\"] = self.input_dir_root\n            if \"pdb\" in p and p[\"pdb\"]:\n                p[\"params\"][\"pdb\"] = True\n            else:\n                p[\"params\"][\"pdb\"] = False\n            \n            start = time.time()\n            print(f\"===== [{task}] {comment} =====\")\n            task_map[task](p[\"params\"])\n            gc.collect()\n            print(\"====================\")\n            end = time.time()\n\n            self.processing_times[\"task\"].append(task)\n            self.processing_times[\"comment\"].append(comment)\n            self.processing_times[\"processing_time\"].append(end-start)\n            all_processing_time += end-start\n        \n        self.processing_times[\"task\"].append(\"All\")\n        self.processing_times[\"comment\"].append(\"\")\n        self.processing_times[\"processing_time\"].append(all_processing_time)\n        processing_times_df = pd.DataFrame.from_dict(self.processing_times)\n        processing_times_df.to_csv(os.path.join(self.work_dir, \"processing_time.csv\"), index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:48.244135Z","iopub.execute_input":"2025-07-08T05:57:48.244582Z","iopub.status.idle":"2025-07-08T05:57:48.258617Z","shell.execute_reply.started":"2025-07-08T05:57:48.244553Z","shell.execute_reply":"2025-07-08T05:57:48.257819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile imc2024-inference-scripts/reconstruction.py\n\nimport os\nimport numpy as np\nfrom pathlib import Path\nfrom copy import deepcopy\nimport pycolmap\nimport pandas as pd\n\nimport sqlite3\nfrom PIL import Image, ExifTags\nimport h5py\nfrom tqdm import tqdm\nimport warnings\nimport pickle\nimport json\n\n# import sys\n# sys.path.append(\"/mnt/2ndHDD/kaggle/IMC2024/datas/input/colmap-db-import\")\n# from database import *\n# from h5_to_db import *\n\n# def import_into_colmap(\n#     path: Path,\n#     feature_dir: Path,\n#     database_path: str = \"colmap.db\",\n# ) -> None:\n#     \"\"\"Adds keypoints into colmap\"\"\"\n#     db = COLMAPDatabase.connect(database_path)\n#     db.create_tables()\n#     single_camera = False\n#     fname_to_id = add_keypoints(db, feature_dir, path, \"\", \"simple-pinhole\", single_camera)\n#     add_matches(\n#         db,\n#         feature_dir,\n#         fname_to_id,\n#     )\n#     db.commit()\n\ndef arr_to_str(a):\n    \"\"\"Returns ;-separated string representing the input\"\"\"\n    return \";\".join([str(x) for x in a.reshape(-1)])\n\nMAX_IMAGE_ID = 2**31 - 1\n\n\nCREATE_CAMERAS_TABLE = \"\"\"CREATE TABLE IF NOT EXISTS cameras (\n    camera_id INTEGER PRIMARY KEY AUTOINCREMENT NOT NULL,\n    model INTEGER NOT NULL,\n    width INTEGER NOT NULL,\n    height INTEGER NOT NULL,\n    params BLOB,\n    prior_focal_length INTEGER NOT NULL)\"\"\"\n\n\nCREATE_DESCRIPTORS_TABLE = \"\"\"CREATE TABLE IF NOT EXISTS descriptors (\n    image_id INTEGER PRIMARY KEY NOT NULL,\n    rows INTEGER NOT NULL,\n    cols INTEGER NOT NULL,\n    data BLOB,\n    FOREIGN KEY(image_id) REFERENCES images(image_id) ON DELETE CASCADE)\"\"\"\n\n\nCREATE_IMAGES_TABLE = \"\"\"CREATE TABLE IF NOT EXISTS images (\n    image_id INTEGER PRIMARY KEY AUTOINCREMENT NOT NULL,\n    name TEXT NOT NULL UNIQUE,\n    camera_id INTEGER NOT NULL,\n    prior_qw REAL,\n    prior_qx REAL,\n    prior_qy REAL,\n    prior_qz REAL,\n    prior_tx REAL,\n    prior_ty REAL,\n    prior_tz REAL,\n    CONSTRAINT image_id_check CHECK(image_id >= 0 and image_id < {}),\n    FOREIGN KEY(camera_id) REFERENCES cameras(camera_id))\n\"\"\".format(MAX_IMAGE_ID)\n\n\nCREATE_TWO_VIEW_GEOMETRIES_TABLE = \"\"\"\nCREATE TABLE IF NOT EXISTS two_view_geometries (\n    pair_id INTEGER PRIMARY KEY NOT NULL,\n    rows INTEGER NOT NULL,\n    cols INTEGER NOT NULL,\n    data BLOB,\n    config INTEGER NOT NULL,\n    F BLOB,\n    E BLOB,\n    H BLOB)\n\"\"\"\n\n\nCREATE_KEYPOINTS_TABLE = \"\"\"CREATE TABLE IF NOT EXISTS keypoints (\n    image_id INTEGER PRIMARY KEY NOT NULL,\n    rows INTEGER NOT NULL,\n    cols INTEGER NOT NULL,\n    data BLOB,\n    FOREIGN KEY(image_id) REFERENCES images(image_id) ON DELETE CASCADE)\n\"\"\"\n\n\nCREATE_MATCHES_TABLE = \"\"\"CREATE TABLE IF NOT EXISTS matches (\n    pair_id INTEGER PRIMARY KEY NOT NULL,\n    rows INTEGER NOT NULL,\n    cols INTEGER NOT NULL,\n    data BLOB)\"\"\"\n\n\nCREATE_NAME_INDEX = \\\n    \"CREATE UNIQUE INDEX IF NOT EXISTS index_name ON images(name)\"\n\n\nCREATE_ALL = \"; \".join([\n    CREATE_CAMERAS_TABLE,\n    CREATE_IMAGES_TABLE,\n    CREATE_KEYPOINTS_TABLE,\n    CREATE_DESCRIPTORS_TABLE,\n    CREATE_MATCHES_TABLE,\n    CREATE_TWO_VIEW_GEOMETRIES_TABLE,\n    CREATE_NAME_INDEX\n])\n\n\ndef image_ids_to_pair_id(image_id1, image_id2):\n    if image_id1 > image_id2:\n        image_id1, image_id2 = image_id2, image_id1\n    return image_id1 * MAX_IMAGE_ID + image_id2\n\n\ndef array_to_blob(array):\n    return array.tostring()\n\n\nclass COLMAPDatabase(sqlite3.Connection):\n\n    @staticmethod\n    def connect(database_path):\n        return sqlite3.connect(database_path, factory=COLMAPDatabase)\n\n    def __init__(self, *args, **kwargs):\n        super(COLMAPDatabase, self).__init__(*args, **kwargs)\n\n        self.create_tables = lambda: self.executescript(CREATE_ALL)\n        self.create_cameras_table = \\\n            lambda: self.executescript(CREATE_CAMERAS_TABLE)\n        self.create_descriptors_table = \\\n            lambda: self.executescript(CREATE_DESCRIPTORS_TABLE)\n        self.create_images_table = \\\n            lambda: self.executescript(CREATE_IMAGES_TABLE)\n        self.create_two_view_geometries_table = \\\n            lambda: self.executescript(CREATE_TWO_VIEW_GEOMETRIES_TABLE)\n        self.create_keypoints_table = \\\n            lambda: self.executescript(CREATE_KEYPOINTS_TABLE)\n        self.create_matches_table = \\\n            lambda: self.executescript(CREATE_MATCHES_TABLE)\n        self.create_name_index = lambda: self.executescript(CREATE_NAME_INDEX)\n\n    def add_camera(self, model, width, height, params,\n                   prior_focal_length=0, camera_id=None):\n        params = np.asarray(params, np.float64)\n        cursor = self.execute(\n            \"INSERT INTO cameras VALUES (?, ?, ?, ?, ?, ?)\",\n            (camera_id, model, width, height, array_to_blob(params),\n             prior_focal_length))\n        return cursor.lastrowid\n\n    def add_image(self, name, camera_id,\n                  prior_q=np.zeros(4), prior_t=np.zeros(3), image_id=None):\n        cursor = self.execute(\n            \"INSERT INTO images VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)\",\n            (image_id, name, camera_id, prior_q[0], prior_q[1], prior_q[2],\n             prior_q[3], prior_t[0], prior_t[1], prior_t[2]))\n        return cursor.lastrowid\n\n    def add_keypoints(self, image_id, keypoints):\n        assert(len(keypoints.shape) == 2)\n        assert(keypoints.shape[1] in [2, 4, 6])\n\n        keypoints = np.asarray(keypoints, np.float32)\n        self.execute(\n            \"INSERT INTO keypoints VALUES (?, ?, ?, ?)\",\n            (image_id,) + keypoints.shape + (array_to_blob(keypoints),))\n\n    def add_matches(self, image_id1, image_id2, matches):\n        assert(len(matches.shape) == 2)\n        assert(matches.shape[1] == 2)\n        if image_id1 > image_id2:\n            matches = matches[:,::-1]\n        pair_id = image_ids_to_pair_id(image_id1, image_id2)\n        matches = np.asarray(matches, np.uint32)\n        self.execute(\n            \"INSERT INTO matches VALUES (?, ?, ?, ?)\",\n            (pair_id,) + matches.shape + (array_to_blob(matches),))\n\n    def add_two_view_geometry(self, image_id1, image_id2, matches, F=np.eye(3), E=np.eye(3), H=np.eye(3), config=2):\n        assert(len(matches.shape) == 2)\n        assert(matches.shape[1] == 2)\n        if image_id1 > image_id2:\n            matches = matches[:,::-1]\n        pair_id = image_ids_to_pair_id(image_id1, image_id2)\n        matches = np.asarray(matches, np.uint32)\n        F = np.asarray(F, dtype=np.float64)\n        E = np.asarray(E, dtype=np.float64)\n        H = np.asarray(H, dtype=np.float64)\n        self.execute(\n            \"INSERT INTO two_view_geometries VALUES (?, ?, ?, ?, ?, ?, ?, ?)\",\n            (pair_id,) + matches.shape + (array_to_blob(matches), config,\n             array_to_blob(F), array_to_blob(E), array_to_blob(H)))\n\n        \ndef get_focal(height, width, exif):\n    max_size = max(height, width)\n    focal_found, focal = False, None\n    if exif is not None:\n        focal_35mm = None\n        for tag, value in exif.items():\n            focal_35mm = None\n            if ExifTags.TAGS.get(tag, None) == 'FocalLengthIn35mmFilm':\n                focal_35mm = float(value)\n                break\n        if focal_35mm is not None:\n            focal_found = True\n            focal = focal_35mm / 35. * max_size\n            print(f\"Focal found: {focal}\")\n    if focal is None:\n        FOCAL_PRIOR = 1.2\n        focal = FOCAL_PRIOR * max_size\n    return focal_found, focal\n\n\ndef create_camera(db, height, width, exif, camera_model):\n    focal_found, focal = get_focal(height, width, exif)\n    if camera_model == 'simple-pinhole':\n        model = 0 # simple pinhole\n        param_arr = np.array([focal, width / 2, height / 2])\n    if camera_model == 'pinhole':\n        model = 1 # pinhole\n        param_arr = np.array([focal, focal, width / 2, height / 2])\n    elif camera_model == 'simple-radial':\n        model = 2 # simple radial\n        param_arr = np.array([focal, width / 2, height / 2, 0.1])\n    elif camera_model == 'radial':\n        model = 3 # radial\n        param_arr = np.array([focal, width / 2, height / 2, 0., 0.])\n    elif camera_model == 'opencv':\n        model = 4 # opencv\n        param_arr = np.array([focal, focal, width / 2, height / 2, 0., 0., 0., 0.])\n    return db.add_camera(model, width, height, param_arr, prior_focal_length=int(focal_found))\n\n\ndef add_keypoints(db, feature_dir, h_w_exif, camera_model, single_camera=False):\n    keypoint_f = h5py.File(os.path.join(feature_dir, 'keypoints.h5'), 'r')\n    camera_id = None\n    fname_to_id = {}\n    for filename in tqdm(list(keypoint_f.keys())):\n        keypoints = keypoint_f[filename][()]\n        if camera_id is None or not single_camera:\n            height = h_w_exif[filename]['h']\n            width = h_w_exif[filename]['w']\n            exif = h_w_exif[filename]['exif']\n            camera_id = create_camera(db, height, width, exif, camera_model)\n        image_id = db.add_image(filename, camera_id)\n        fname_to_id[filename] = image_id\n        db.add_keypoints(image_id, keypoints)\n    return fname_to_id\n\n\ndef add_matches_and_fms(db, feature_dir, fname_to_id, fms):\n    match_file = h5py.File(os.path.join(feature_dir, 'matches.h5'), 'r')\n    added = set()\n    for key_1 in match_file.keys():\n        group = match_file[key_1]\n        for key_2 in group.keys():\n            id_1 = fname_to_id[key_1]\n            id_2 = fname_to_id[key_2]\n            pair_id = (id_1, id_2)\n            if pair_id in added:\n                warnings.warn(f'Pair {pair_id} ({id_1}, {id_2}) already added!')\n                continue\n            added.add(pair_id)\n            matches = group[key_2][()]\n            db.add_matches(id_1, id_2, matches)\n            db.add_two_view_geometry(id_1, id_2, matches, fms[(key_1, key_2)])\n\n\ndef import_into_colmap(feature_dir, h_w_exif, fms):\n    db = COLMAPDatabase.connect(f\"{feature_dir}/colmap.db\")\n    db.create_tables()\n    fname_to_id = add_keypoints(db, feature_dir, h_w_exif, camera_model='simple-radial', single_camera=False)\n    add_matches_and_fms(db, feature_dir, fname_to_id, fms)\n    db.commit()\n    db.close()\n\n\n\n\nimport math\n_EPS = np.finfo(float).eps * 4.0\n\ndef quaternion_matrix(quaternion):\n    \"\"\"Return homogeneous rotation matrix from quaternion.\"\"\"\n    q = np.array(quaternion, dtype=np.float64, copy=True)\n    n = np.dot(q, q)\n    if n < _EPS:\n        # print(\"special case\")\n        return np.identity(4)\n    q *= math.sqrt(2.0 / n)\n    q = np.outer(q, q)\n    return np.array(\n        [\n            [\n                1.0 - q[2, 2] - q[3, 3],\n                q[1, 2] - q[3, 0],\n                q[1, 3] + q[2, 0],\n                0.0,\n            ],\n            [\n                q[1, 2] + q[3, 0],\n                1.0 - q[1, 1] - q[3, 3],\n                q[2, 3] - q[1, 0],\n                0.0,\n            ],\n            [\n                q[1, 3] - q[2, 0],\n                q[2, 3] + q[1, 0],\n                1.0 - q[1, 1] - q[2, 2],\n                0.0,\n            ],\n            [0.0, 0.0, 0.0, 1.0],\n        ]\n    )\n\n\ndef vector_norm(data, axis=None, out=None):\n    \"\"\"Return length, i.e. Euclidean norm, of ndarray along axis.\"\"\"\n    data = np.array(data, dtype=np.float64, copy=True)\n    if out is None:\n        if data.ndim == 1:\n            return math.sqrt(np.dot(data, data))\n        data *= data\n        out = np.atleast_1d(np.sum(data, axis=axis))\n        np.sqrt(out, out)\n        return out\n    data *= data\n    np.sum(data, axis=axis, out=out)\n    np.sqrt(out, out)\n    return None\n\ndef affine_matrix_from_points(v0, v1, shear=False, scale=True, usesvd=True):\n    \"\"\"Return affine transform matrix to register two point sets.\n    v0 and v1 are shape (ndims, -1) arrays of at least ndims non-homogeneous\n    coordinates, where ndims is the dimensionality of the coordinate space.\n    If shear is False, a similarity transformation matrix is returned.\n    If also scale is False, a rigid/Euclidean traffansformation matrix\n    is returned.\n    By default the algorithm by Hartley and Zissermann [15] is used.\n    If usesvd is True, similarity and Euclidean transformation matrices\n    are calculated by minimizing the weighted sum of squared deviations\n    (RMSD) according to the algorithm by Kabsch [8].\n    Otherwise, and if ndims is 3, the quaternion based algorithm by Horn [9]\n    is used, which is slower when using this Python implementation.\n    The returned matrix performs rotation, translation and uniform scaling\n    (if specified).\"\"\"\n\n    v0 = np.array(v0, dtype=np.float64, copy=True)\n    v1 = np.array(v1, dtype=np.float64, copy=True)\n\n    ndims = v0.shape[0]\n    if ndims < 2 or v0.shape[1] < ndims or v0.shape != v1.shape:\n        raise ValueError(\"input arrays are of wrong shape or type\")\n\n    # move centroids to origin\n    t0 = -np.mean(v0, axis=1)\n    M0 = np.identity(ndims + 1)\n    M0[:ndims, ndims] = t0\n    v0 += t0.reshape(ndims, 1)\n    t1 = -np.mean(v1, axis=1)\n    M1 = np.identity(ndims + 1)\n    M1[:ndims, ndims] = t1\n    v1 += t1.reshape(ndims, 1)\n\n    if shear:\n        # Affine transformation\n        A = np.concatenate((v0, v1), axis=0)\n        u, s, vh = np.linalg.svd(A.T)\n        vh = vh[:ndims].T\n        B = vh[:ndims]\n        C = vh[ndims : 2 * ndims]\n        t = np.dot(C, np.linalg.pinv(B))\n        t = np.concatenate((t, np.zeros((ndims, 1))), axis=1)\n        M = np.vstack((t, ((0.0,) * ndims) + (1.0,)))\n    elif usesvd or ndims != 3:\n        # Rigid transformation via SVD of covariance matrix\n        u, s, vh = np.linalg.svd(np.dot(v1, v0.T))\n        # rotation matrix from SVD orthonormal bases\n        R = np.dot(u, vh)\n        if np.linalg.det(R) < 0.0:\n            # R does not constitute right handed system\n            R -= np.outer(u[:, ndims - 1], vh[ndims - 1, :] * 2.0)\n            s[-1] *= -1.0\n        # homogeneous transformation matrix\n        M = np.identity(ndims + 1)\n        M[:ndims, :ndims] = R\n    else:\n        # Rigid transformation matrix via quaternion\n        # compute symmetric matrix N\n        xx, yy, zz = np.sum(v0 * v1, axis=1)\n        xy, yz, zx = np.sum(v0 * np.roll(v1, -1, axis=0), axis=1)\n        xz, yx, zy = np.sum(v0 * np.roll(v1, -2, axis=0), axis=1)\n        N = [\n            [xx + yy + zz, 0.0, 0.0, 0.0],\n            [yz - zy, xx - yy - zz, 0.0, 0.0],\n            [zx - xz, xy + yx, yy - xx - zz, 0.0],\n            [xy - yx, zx + xz, yz + zy, zz - xx - yy],\n        ]\n        # quaternion: eigenvector corresponding to most positive eigenvalue\n        w, V = np.linalg.eigh(N)\n        q = V[:, np.argmax(w)]\n        # print (vector_norm(q), np.linalg.norm(q))\n        q /= vector_norm(q)  # unit quaternion\n        # homogeneous transformation matrix\n        M = quaternion_matrix(q)\n\n    if scale and not shear:\n        # Affine transformation; scale is ratio of RMS deviations from centroid\n        v0 *= v0\n        v1 *= v1\n        M[:ndims, :ndims] *= math.sqrt(np.sum(v1) / np.sum(v0))\n\n    # move centroids back\n    M = np.dot(np.linalg.inv(M1), np.dot(M, M0))\n    M /= M[ndims, ndims]\n    return M\n\n\n\ndef register_by_Horn(ev_coord, gt_coord, ransac_threshold, inl_cf, strict_cf):\n    \"\"\"Return the best similarity transforms T that registers 3D points pt_ev in <ev_coord> to\n    the corresponding ones pt_gt in <gt_coord> according to a RANSAC-like approach for each\n    threshold value th in <ransac_threshold>.\n\n    Given th, each triplet of 3D correspondences is examined if not already present as strict inlier,\n    a correspondence is a strict inlier if <strict_cf> * err_best < th, where err_best is the registration\n    error for the best model so far.\n    The minimal model given by the triplet is then refined using also its inliers if their total is greater\n    than <inl_cf> * ninl_best, where ninl_best is th number of inliers for the best model so far. Inliers\n    are 3D correspondences (pt_ev, pt_gt) for which the Euclidean distance |pt_gt-T*pt_ev| is less than th.\n    \"\"\"\n\n    # remove invalid cameras, the index is returned\n    idx_cams = np.all(np.isfinite(ev_coord), axis=0)\n    ev_coord = ev_coord[:, idx_cams]\n    gt_coord = gt_coord[:, idx_cams]\n\n    # initialization\n    n = ev_coord.shape[1]\n    r = ransac_threshold.shape[0]\n    ransac_threshold = np.expand_dims(ransac_threshold, axis=0)\n    ransac_threshold2 = ransac_threshold**2\n    ev_coord_1 = np.vstack((ev_coord, np.ones(n)))\n    max_no_inl = np.zeros((1, r))\n    best_inl_err = np.full(r, np.Inf)\n    best_transf_matrix = np.zeros((r, 4, 4))\n    best_err = np.full((n, r), np.Inf)\n    strict_inl = np.full((n, r), False)\n    triplets_used = np.zeros((3, r))\n\n    # run on camera triplets\n    for ii in range(n - 2):\n        for jj in range(ii + 1, n - 1):\n            for kk in range(jj + 1, n):\n                i = [ii, jj, kk]\n                triplets_used_now = np.full((n), False)\n                triplets_used_now[i] = True\n                # if both ii, jj, kk are strict inliers for the best current model just skip\n                if np.all(strict_inl[i]):\n                    continue\n                # get transformation T by Horn on the triplet camera center correspondences\n                transf_matrix = affine_matrix_from_points(\n                    ev_coord[:, i], gt_coord[:, i], usesvd=False\n                )\n                # apply transformation T to test camera centres\n                rotranslated = np.matmul(transf_matrix[:3], ev_coord_1)\n                # compute error and inliers\n                err = np.sum((rotranslated - gt_coord) ** 2, axis=0)\n                inl = np.expand_dims(err, axis=1) < ransac_threshold2\n                no_inl = np.sum(inl, axis=0)\n                # if the number of inliers is close to that of the best model so far, go for refinement\n                to_ref = np.squeeze(\n                    ((no_inl > 2) & (no_inl > max_no_inl * inl_cf)), axis=0\n                )\n                for q in np.argwhere(to_ref):\n                    qq = q[0]\n                    if np.any(\n                        np.all(\n                            (np.expand_dims(inl[:, qq], axis=1) == inl[:, :qq]), axis=0\n                        )\n                    ):\n                        # already done for this set of inliers\n                        continue\n                    # get transformation T by Horn on the inlier camera center correspondences\n                    transf_matrix = affine_matrix_from_points(\n                        ev_coord[:, inl[:, qq]], gt_coord[:, inl[:, qq]]\n                    )\n                    # apply transformation T to test camera centres\n                    rotranslated = np.matmul(transf_matrix[:3], ev_coord_1)\n                    # compute error and inliers\n                    err_ref = np.sum((rotranslated - gt_coord) ** 2, axis=0)\n                    err_ref_sum = np.sum(err_ref, axis=0)\n                    err_ref = np.expand_dims(err_ref, axis=1)\n                    inl_ref = err_ref < ransac_threshold2\n                    no_inl_ref = np.sum(inl_ref, axis=0)\n                    # update the model if better for each threshold\n                    to_update = np.squeeze(\n                        (no_inl_ref > max_no_inl)\n                        | ((no_inl_ref == max_no_inl) & (err_ref_sum < best_inl_err)),\n                        axis=0,\n                    )\n                    if np.any(to_update):\n                        triplets_used[0, to_update] = ii\n                        triplets_used[1, to_update] = jj\n                        triplets_used[2, to_update] = kk\n                        max_no_inl[:, to_update] = no_inl_ref[to_update]\n                        best_err[:, to_update] = np.sqrt(err_ref)\n                        best_inl_err[to_update] = err_ref_sum\n                        strict_inl[:, to_update] = (\n                            best_err[:, to_update]\n                            < strict_cf * ransac_threshold[:, to_update]\n                        )\n                        best_transf_matrix[to_update] = transf_matrix\n    \n    for i in range(r):\n        print(\n            f\"Registered cameras {int(max_no_inl[0, i])}/{n} for threshold {ransac_threshold[0, i]}\"\n        )\n\n    best_model = {\n        \"valid_cams\": idx_cams,\n        \"no_inl\": max_no_inl,\n        \"err\": best_err,\n        \"triplets_used\": triplets_used,\n        \"transf_matrix\": best_transf_matrix,\n    }    \n    return best_model\n\n\ndef reconstruction(data_dict, dataset, scene, base_path, work_dir, colmap_mapper_options):\n    # Import keypoint distances of matches into colmap for RANSAC \n    images_dir = data_dict[dataset][scene][0].parent\n    with open(os.path.join(work_dir, \"h_w_exif.json\"), \"r\") as f:\n        h_w_exif = json.load(f)\n    \n    with open(os.path.join(work_dir, \"fms.pkl\"), \"rb\") as f:\n        fms = pickle.load(f)\n    import_into_colmap(feature_dir=work_dir, h_w_exif=h_w_exif, fms=fms)\n\n    database_path = f\"{work_dir}/colmap.db\"\n    mapper_options = pycolmap.IncrementalPipelineOptions(**colmap_mapper_options)\n    output_path = f\"{work_dir}/colmap_rec_aliked\"\n    os.makedirs(output_path, exist_ok=True)\n\n    db = COLMAPDatabase.connect(database_path)\n    cursor = db.execute(\"SELECT image_id, name from images\")\n    db_data = cursor.fetchall()\n    image_ids = [int(x[0]) for x in db_data]\n    names = [str(x[1]) for x in db_data]\n    db.close()\n\n    df = pd.read_csv(f\"{work_dir}/image_pair.csv\")\n    images_matches = {}\n    for name in names:\n        images_matches[name] = [0, 0]\n    for i, row in df.iterrows():\n        key1 = row[\"key1\"]\n        key2 = row[\"key2\"]\n        match_num = row[\"match_num\"]\n        images_matches[key1][0] += 1\n        images_matches[key1][1] += match_num\n        images_matches[key2][0] += 1\n        images_matches[key2][1] += match_num\n    sorted_images_matches = sorted(images_matches.items(), key=lambda item: (item[1][0], item[1][1]), reverse=True)\n    init_image_name1 = sorted_images_matches[0][0]\n    init_image_id1 = image_ids[names.index(init_image_name1)]\n    mapper_options.init_image_id1 = init_image_id1\n\n    maps = pycolmap.incremental_mapping(database_path=database_path, image_path=images_dir, output_path=output_path, options=mapper_options)\n    print(maps)\n\n    # 2. Look for the best reconstruction: The incremental mapping offered by \n    # pycolmap attempts to reconstruct multiple models, we must pick the best one\n    images_registered  = 0\n    best_idx = None\n    \n    print (\"Looking for the best reconstruction\")\n\n    if isinstance(maps, dict):\n        for idx1, rec in maps.items():\n            print(idx1, rec.summary())\n            try:\n                if len(rec.images) > images_registered:\n                    images_registered = len(rec.images)\n                    best_idx = idx1\n            except Exception:\n                continue\n\n    # Parse the reconstruction object to get the rotation matrix and translation vector\n    # obtained for each image in the reconstruction\n    results = {}\n    camid_im_map = {}\n    if best_idx is not None:\n        for k, im in maps[best_idx].images.items():\n            key = os.path.join(images_dir, im.name)\n            results[key] = {}\n            results[key][\"R\"] = deepcopy(im.cam_from_world.rotation.matrix())\n            results[key][\"t\"] = deepcopy(np.array(im.cam_from_world.translation))\n\n            camid_im_map[im.camera_id] = im.name\n\n    try:\n        if best_idx is not None:\n            for idx1, rec in maps.items():\n                u_cameras = []\n                g_cameras = []\n                if idx1 != best_idx:\n                    for k, im in rec.images.items():\n                        key = os.path.join(images_dir, im.name)\n                        if key in results:\n                            g_R = deepcopy(results[key][\"R\"])\n                            g_t = deepcopy(results[key][\"t\"])\n                            g_C = -g_R.T @ g_t\n\n                            u_R = deepcopy(im.cam_from_world.rotation.matrix())\n                            u_t = deepcopy(np.array(im.cam_from_world.translation))\n                            u_C = -u_R.T @ u_t\n                            g_cameras.append(g_C.reshape(3, 1))\n                            u_cameras.append(u_C.reshape(3, 1))\n                    if len(g_cameras) < 3:\n                        continue\n                    g_cameras = np.array(g_cameras).reshape(3, -1)\n                    u_cameras = np.array(u_cameras).reshape(3, -1)\n                    inl_cf = 0\n                    strict_cf = -1\n                    thresholds = np.array([0.025, 0.05, 0.1, 0.2, 0.5, 1.0])\n                    model = register_by_Horn(\n                        u_cameras, g_cameras,\n                        np.asarray(thresholds), inl_cf, strict_cf\n                    )\n                    T = np.squeeze(model[\"transf_matrix\"][-1])\n                    # print(T)\n                    # print(T[:3].shape)\n                    for k, im in rec.images.items():\n                        key = os.path.join(images_dir, im.name)\n                        if key not in results:\n                            Tcw2 = np.eye(4)\n                            Tcw2[:3, :3] = deepcopy(im.cam_from_world.rotation.matrix())\n                            Tcw2[:3, 3] = deepcopy(np.array(im.cam_from_world.translation))\n                            Tw2c = np.linalg.inv(Tcw2)\n                            Tw1c = np.matmul(T, Tw2c)\n                            Tcw1 = np.linalg.inv(Tw1c)\n                            results[key][\"R\"] = deepcopy(Tcw1[:3, :3])\n                            results[key][\"t\"] = deepcopy(Tcw1[:3, 3])\n    except:\n        pass\n    # with open(os.path.join(work_dir, \"camid_im_map.json\"), \"w\") as f:\n    #     json.dump(camid_im_map, f, indent=2)\n    \n    print(f\"Registered: {dataset} / {scene} -> {len(results)} images\")\n    print(f\"Total: {dataset} / {scene} -> {len(data_dict[dataset][scene])} images\")\n\n    # Create Submission\n    submission = {\n        \"image_path\": [],\n        \"dataset\": [],\n        \"scene\": [],\n        \"rotation_matrix\": [],\n        \"translation_vector\": []\n    }\n    for image in data_dict[dataset][scene]:\n        if str(image) in results:\n            print(image)\n            R = results[str(image)][\"R\"].reshape(-1)\n            T = results[str(image)][\"t\"].reshape(-1)\n        else:\n            R = np.eye(3).reshape(-1)\n            T = np.zeros((3))\n        image_path = str(image.relative_to(base_path))\n        \n        submission[\"image_path\"].append(image_path)\n        submission[\"dataset\"].append(dataset)\n        submission[\"scene\"].append(scene)\n        submission[\"rotation_matrix\"].append(arr_to_str(R))\n        submission[\"translation_vector\"].append(arr_to_str(T))\n    \n    submission_df = pd.DataFrame.from_dict(submission)\n    submission_path = os.path.join(work_dir, \"submission.csv\")\n    print(f\"Save to {submission_path}\")\n    submission_df.to_csv(submission_path, index=False)\n    return submission_path\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:48.259944Z","iopub.execute_input":"2025-07-08T05:57:48.260197Z","iopub.status.idle":"2025-07-08T05:57:48.282382Z","shell.execute_reply.started":"2025-07-08T05:57:48.260167Z","shell.execute_reply":"2025-07-08T05:57:48.281584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile imc2024-inference-scripts/inference.py\nimport multiprocessing\nimport argparse\nimport os\nimport pandas as pd\nimport shutil\nimport json\nimport gc\nimport torch\nfrom pathlib import Path\nfrom config import Config\nfrom pipeline import Pipeline\nfrom reconstruction import reconstruction\n\ndef parse_sample_submission(\n    base_path: str,\n    input_csv: str,\n    target_datasets: list,\n) -> dict[dict[str, list[Path]]]:\n    \"\"\"Construct a dict describing the test data as \n    \n    {\"dataset\": {\"scene\": [<image paths>]}}\n    \"\"\"\n    data_dict = {}\n    with open(input_csv, \"r\") as f:\n        for i, l in enumerate(f):\n            # Skip header\n            if i == 0:\n                print(\"header:\", l)\n\n            if l and i > 0:\n                image_path, dataset, scene, _, _ = l.strip().split(',')\n                if target_datasets is not None and dataset not in target_datasets:\n                    continue\n\n                ### Temp for bug?\n                if not os.path.isfile(os.path.join(base_path,image_path)):\n                    continue\n\n                if dataset not in data_dict:\n                    data_dict[dataset] = {}\n                if scene not in data_dict[dataset]:\n                    data_dict[dataset][scene] = []\n                data_dict[dataset][scene].append(Path(os.path.join(base_path,image_path)))\n\n    for dataset in data_dict:\n        for scene in data_dict[dataset]:\n            print(f\"{dataset} / {scene} -> {len(data_dict[dataset][scene])} images\")\n\n    return data_dict\n\ndef worker_reconstruction(input_queue, submission_path_list):\n    while True:\n        reconstruction_inputs = input_queue.get()\n        if reconstruction_inputs is None:\n            break\n        data_dict, dataset, scene, base_path, work_dir, colmap_mapper_options = reconstruction_inputs\n        submission_path = reconstruction(data_dict, dataset, scene, base_path, work_dir, colmap_mapper_options)\n        submission_path_list.append(submission_path)\n        \ndef run(config):\n    parser = argparse.ArgumentParser()\n    parser.add_argument(\"--device_id\", type=int, default=0)\n    parser.add_argument(\"--target_datasets\", nargs=\"*\", required=False)\n    parser.add_argument(\"--output_dir\", required=False)\n    args = parser.parse_args()\n\n    if args.output_dir:\n        config.output_dir = args.output_dir\n    \n    if args.target_datasets:\n        config.target_datasets = args.target_datasets\n\n    # Check output_dir\n    if config.check_exist_dir:\n        if os.path.isdir(config.output_dir):\n            raise Exception(f\"{config.output_dir} is already exists.\")\n\n    os.makedirs(config.output_dir, exist_ok=True)\n    base_path = os.path.join(config.input_dir_root, \"image-matching-challenge-2024\")\n    feature_dir = os.path.join(config.output_dir, \"feature_outputs\")\n    shutil.copy(config.pipeline_json, config.output_dir)\n    shutil.copy(config.transparent_pipeline_json, config.output_dir)\n\n    # Load category\n    category_df = pd.read_csv(config.category_csv)\n    categories = {}\n    for i, row in category_df.iterrows():\n        categories[row[\"scene\"]] = row[\"categories\"].split(\";\")\n\n    data_dict = parse_sample_submission(base_path, config.input_csv, config.target_datasets)\n    datasets = list(data_dict.keys())\n    \n    manager = multiprocessing.Manager()\n    submission_path_list = manager.list()\n\n    input_queue = multiprocessing.Queue()\n    worker_reconstruction_process = multiprocessing.Process(target=worker_reconstruction, args=(input_queue, submission_path_list))\n    worker_reconstruction_process.start()\n    for dataset in datasets:            \n        for scene in data_dict[dataset]:\n            print(f\"[Scene] {scene}\")\n            work_dir = Path(os.path.join(feature_dir, f\"{dataset}_{scene}\"))\n            work_dir.mkdir(parents=True, exist_ok=True)\n\n            # Switching JSON\n            if \"transparent\" in categories[scene]:\n                json_path = config.transparent_pipeline_json\n            else:\n                json_path = config.pipeline_json\n            print(f\"catefories: {categories[scene]}\")\n            print(f\"json path: {json_path}\")\n            \n            # Exec Pipeline\n            with open(json_path, \"r\") as f:\n                pipeline_config = json.load(f)\n            pipeline = Pipeline(data_dict[dataset][scene], work_dir, config.input_dir_root, pipeline_config, args.device_id)\n            pipeline.exec()\n\n            # Reconstruction & Save CSV\n            print(\"Start Reconstruction\")            \n            torch.cuda.empty_cache()\n            gc.collect()\n            input_queue.put((data_dict, dataset, scene, base_path, work_dir, config.colmap_mapper_options))\n    input_queue.put(None)\n    worker_reconstruction_process.join()\n    # Concat Submission\n    if len(submission_path_list) > 0:\n        submission_df_list = [pd.read_csv(p) for p in submission_path_list]\n        submission_df = pd.concat(submission_df_list).reset_index(drop=True)\n    else:\n        submission_df = pd.DataFrame.from_dict({\n            \"image_path\": [],\n            \"dataset\": [],\n            \"scene\": [],\n            \"rotation_matrix\": [],\n            \"translation_vector\": []\n        })\n    submission_df.to_csv(os.path.join(config.output_dir, \"submission.csv\"), index=False)\n\nif __name__ == '__main__':\n    cfg = Config\n    run(cfg)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:48.283366Z","iopub.execute_input":"2025-07-08T05:57:48.28359Z","iopub.status.idle":"2025-07-08T05:57:48.299171Z","shell.execute_reply.started":"2025-07-08T05:57:48.283565Z","shell.execute_reply":"2025-07-08T05:57:48.298275Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Split scene list","metadata":{}},{"cell_type":"code","source":"root_path = \"/kaggle/input/image-matching-challenge-2024/test\"\ndata_num_list = [sum(len(files) for _, _, files in os.walk(f\"{root_path}/{scene}/images\")) for scene in non_transparent_scenes]\ndata_num_list","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:48.300106Z","iopub.execute_input":"2025-07-08T05:57:48.300427Z","iopub.status.idle":"2025-07-08T05:57:48.393565Z","shell.execute_reply.started":"2025-07-08T05:57:48.300396Z","shell.execute_reply":"2025-07-08T05:57:48.392832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check correct path (if the path is wrong, it should be Notebook exeption)\nif 0 in data_num_list:\n    raise Exception","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:48.394588Z","iopub.execute_input":"2025-07-08T05:57:48.394964Z","iopub.status.idle":"2025-07-08T05:57:48.399166Z","shell.execute_reply.started":"2025-07-08T05:57:48.394929Z","shell.execute_reply":"2025-07-08T05:57:48.398345Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def partition_with_index(lst, lst2):\n    indexed_lst = sorted(enumerate(lst), key=lambda x: x[1], reverse=True)\n    groupA = []\n    groupB = []\n    groupA2 = []\n    groupB2 = []\n    \n    for index, item in indexed_lst:\n        if sum([lst[i] for i in groupA]) <= sum([lst[i] for i in groupB]):\n            groupA.append(index)\n            groupA2.append(lst2[index])\n        else:\n            groupB.append(index)\n            groupB2.append(lst2[index])\n    \n    return groupA2, groupB2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:48.400326Z","iopub.execute_input":"2025-07-08T05:57:48.400565Z","iopub.status.idle":"2025-07-08T05:57:48.412271Z","shell.execute_reply.started":"2025-07-08T05:57:48.400546Z","shell.execute_reply":"2025-07-08T05:57:48.411356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"non_transparent_scenes0, non_transparent_scenes1 = partition_with_index(data_num_list, non_transparent_scenes)\nprint(non_transparent_scenes0)\nprint(non_transparent_scenes1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:48.413363Z","iopub.execute_input":"2025-07-08T05:57:48.413593Z","iopub.status.idle":"2025-07-08T05:57:48.424739Z","shell.execute_reply.started":"2025-07-08T05:57:48.413568Z","shell.execute_reply":"2025-07-08T05:57:48.423956Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Exec Inference","metadata":{}},{"cell_type":"code","source":"torch.cuda.empty_cache()\nimport gc\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:48.425577Z","iopub.execute_input":"2025-07-08T05:57:48.425833Z","iopub.status.idle":"2025-07-08T05:57:48.645743Z","shell.execute_reply.started":"2025-07-08T05:57:48.425801Z","shell.execute_reply":"2025-07-08T05:57:48.644916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def make_command(device_id, target_datasets):\n    cmd = f\"python imc2024-inference-scripts/inference.py --device_id {device_id} --output_dir /tmp/tmp{device_id}\"\n    if len(target_datasets) > 0:\n        cmd += \" --target_datasets\"\n        for scene in target_datasets:\n            cmd += f\" {scene}\"\n    return cmd.split(\" \")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:48.646817Z","iopub.execute_input":"2025-07-08T05:57:48.647139Z","iopub.status.idle":"2025-07-08T05:57:48.659084Z","shell.execute_reply.started":"2025-07-08T05:57:48.647113Z","shell.execute_reply":"2025-07-08T05:57:48.658363Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import subprocess\nproc0 = subprocess.Popen(make_command(0, non_transparent_scenes0))\nproc1 = subprocess.Popen(make_command(1, non_transparent_scenes1))\nproc0.wait()\nproc1.wait()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:57:48.664835Z","iopub.execute_input":"2025-07-08T05:57:48.665109Z","iopub.status.idle":"2025-07-08T06:30:40.46175Z","shell.execute_reply.started":"2025-07-08T05:57:48.665088Z","shell.execute_reply":"2025-07-08T06:30:40.460833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df1 = pd.read_csv(\"/tmp/tmp0/submission.csv\")\ndf2 = pd.read_csv(\"/tmp/tmp1/submission.csv\")\ndf = pd.concat([df1, df2]).reset_index(drop=True)\ndf.to_csv(\"/tmp/submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T06:30:40.462764Z","iopub.execute_input":"2025-07-08T06:30:40.463037Z","iopub.status.idle":"2025-07-08T06:30:40.475359Z","shell.execute_reply.started":"2025-07-08T06:30:40.463016Z","shell.execute_reply":"2025-07-08T06:30:40.474581Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Merge csv","metadata":{}},{"cell_type":"code","source":"df1 = pd.read_csv(f\"{OUTPUT_ROOT}/transp_submission.csv\")\ndf2 = pd.read_csv(\"/tmp/submission.csv\")\n\ndf1 = df1.query(\"scene not in @non_transparent_scenes\").reset_index(drop=True)\ndf = pd.concat([df1, df2])\n\nFINAL_SUB_TMP_PATH = \"/tmp/final_submission.csv\"\ndf.to_csv(FINAL_SUB_TMP_PATH, index=False)\nprint(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T06:30:40.476391Z","iopub.execute_input":"2025-07-08T06:30:40.47664Z","iopub.status.idle":"2025-07-08T06:30:40.516299Z","shell.execute_reply.started":"2025-07-08T06:30:40.476619Z","shell.execute_reply":"2025-07-08T06:30:40.515449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile imc2024-inference-scripts/config.py\n\nimport os\nimport json\n\nclass Config:\n    input_dir_root = \"/kaggle/input\"\n    output_dir = \"/tmp\"\n    check_exist_dir = False\n    \n    input_csv = \"/kaggle/input/image-matching-challenge-2024/sample_submission.csv\"\n    category_csv = \"/kaggle/input/image-matching-challenge-2024/test/categories.csv\"\n    target_datasets = []\n    \n    # matches >= 70\n    pipeline_json = \"/kaggle/input/imc2024-pipelines-v7/exp8_aliked1536+1280_crop-aliked1024_v2_100_70/pipeline.json\"\n    transparent_pipeline_json = \"/kaggle/input/imc2024-pipelines-v7/exp8_aliked1536+1280_crop-aliked1024_v2_100_70/transp_pipeline.json\"\n    \n    colmap_mapper_options = {\n        \"min_model_size\": 3, # By default colmap does not generate a reconstruction if less than 10 images are registered. Lower it to 3.\n        \"max_num_models\": 2,\n        #\"num_threads\": 1,\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T06:30:40.517313Z","iopub.execute_input":"2025-07-08T06:30:40.517544Z","iopub.status.idle":"2025-07-08T06:30:40.522775Z","shell.execute_reply.started":"2025-07-08T06:30:40.517524Z","shell.execute_reply":"2025-07-08T06:30:40.522102Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"thr_null_img = 0.15","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T06:30:40.523891Z","iopub.execute_input":"2025-07-08T06:30:40.524404Z","iopub.status.idle":"2025-07-08T06:30:40.538298Z","shell.execute_reply.started":"2025-07-08T06:30:40.524376Z","shell.execute_reply":"2025-07-08T06:30:40.537669Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# find scenes with low N of reg images\ndf1_ = pd.read_csv(FINAL_SUB_TMP_PATH)\ndf_non_trsp = df1_.query(\"scene in @non_transparent_scenes\").reset_index(drop=True)\nprint(df_non_trsp.shape)\n\nrot_m_null = '1.0;0.0;0.0;0.0;1.0;0.0;0.0;0.0;1.0'\ntr_v_null = '0.0;0.0;0.0'\n\nscene_cnt_dict = df_non_trsp.groupby(['scene']).image_path.count().to_dict()\nprint('scene_cnt_dict', scene_cnt_dict)\n\nmask_null = (df_non_trsp['rotation_matrix'] == rot_m_null) & (df_non_trsp['translation_vector']==tr_v_null)\nscene_cnt_null_dict = df_non_trsp[mask_null].groupby(['scene']).image_path.count().to_dict()\n\npoor_non_trsp_scenes = [k_scene for k_scene, v_cnt in scene_cnt_null_dict.items() if v_cnt/scene_cnt_dict[k_scene] >= thr_null_img]\nprint('poor_non_trsp_scenes', poor_non_trsp_scenes)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T06:30:40.539277Z","iopub.execute_input":"2025-07-08T06:30:40.539812Z","iopub.status.idle":"2025-07-08T06:30:40.566306Z","shell.execute_reply.started":"2025-07-08T06:30:40.539747Z","shell.execute_reply":"2025-07-08T06:30:40.565583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_num_list = [sum(len(files) for _, _, files in os.walk(f\"{root_path}/{scene}/images\")) for scene in poor_non_trsp_scenes]\ndata_num_list","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T06:30:40.567266Z","iopub.execute_input":"2025-07-08T06:30:40.567557Z","iopub.status.idle":"2025-07-08T06:30:40.573387Z","shell.execute_reply.started":"2025-07-08T06:30:40.567521Z","shell.execute_reply":"2025-07-08T06:30:40.572599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"non_transparent_scenes0, non_transparent_scenes1 = partition_with_index(data_num_list, poor_non_trsp_scenes)\nprint(non_transparent_scenes0)\nprint(non_transparent_scenes1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T06:30:40.57449Z","iopub.execute_input":"2025-07-08T06:30:40.575098Z","iopub.status.idle":"2025-07-08T06:30:40.591494Z","shell.execute_reply.started":"2025-07-08T06:30:40.575066Z","shell.execute_reply":"2025-07-08T06:30:40.590666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!rm -rf /tmp/tmp0 /tmp/tmp1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T06:30:40.592338Z","iopub.execute_input":"2025-07-08T06:30:40.592616Z","iopub.status.idle":"2025-07-08T06:30:41.680221Z","shell.execute_reply.started":"2025-07-08T06:30:40.592589Z","shell.execute_reply":"2025-07-08T06:30:41.679056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\ntorch.cuda.empty_cache()\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T06:30:41.681632Z","iopub.execute_input":"2025-07-08T06:30:41.68192Z","iopub.status.idle":"2025-07-08T06:30:41.895553Z","shell.execute_reply.started":"2025-07-08T06:30:41.681896Z","shell.execute_reply":"2025-07-08T06:30:41.894763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import subprocess\nproc0 = subprocess.Popen(make_command(0, non_transparent_scenes0))\nproc1 = subprocess.Popen(make_command(1, non_transparent_scenes1))\nproc0.wait()\nproc1.wait()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T06:30:41.896705Z","iopub.execute_input":"2025-07-08T06:30:41.896989Z","iopub.status.idle":"2025-07-08T06:30:51.081232Z","shell.execute_reply.started":"2025-07-08T06:30:41.896956Z","shell.execute_reply":"2025-07-08T06:30:51.080406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df1_ = pd.read_csv(\"/tmp/tmp0/submission.csv\")\ndf2_ = pd.read_csv(\"/tmp/tmp1/submission.csv\")\ndf_poor = pd.concat([df1_, df2_]).reset_index(drop=True)\ndf_poor.to_csv(\"/tmp/submission_poor.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T06:30:51.082382Z","iopub.execute_input":"2025-07-08T06:30:51.082635Z","iopub.status.idle":"2025-07-08T06:30:51.095621Z","shell.execute_reply.started":"2025-07-08T06:30:51.082613Z","shell.execute_reply":"2025-07-08T06:30:51.094927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scene_poor_cnt_reg_dict, scene_cnt_reg_dict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T06:30:51.118747Z","iopub.execute_input":"2025-07-08T06:30:51.119344Z","iopub.status.idle":"2025-07-08T06:30:51.178425Z","shell.execute_reply.started":"2025-07-08T06:30:51.119312Z","shell.execute_reply":"2025-07-08T06:30:51.177676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if len(replace_rec_scenes) > 0:\n    df1_ = pd.read_csv(f\"{OUTPUT_ROOT}/transp_submission.csv\") # transparent\n    df1_ = df1_.query(\"scene not in @non_transparent_scenes\").reset_index(drop=True)\n    df2_ = pd.read_csv(\"/tmp/submission.csv\") # non transparent\n    \n    df = pd.concat([df1_, df2_])\n    df.to_csv(FINAL_SUB_TMP_PATH, index=False)\n    print(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T06:30:51.179406Z","iopub.execute_input":"2025-07-08T06:30:51.179635Z","iopub.status.idle":"2025-07-08T06:30:51.188964Z","shell.execute_reply.started":"2025-07-08T06:30:51.179616Z","shell.execute_reply":"2025-07-08T06:30:51.188342Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# copy tmp sub to final location (to avoid kaggle early stop)\nFINAL_SUB_PATH = f\"{OUTPUT_ROOT}/submission.csv\"\n!cp {FINAL_SUB_TMP_PATH} {FINAL_SUB_PATH}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T06:49:09.385928Z","iopub.execute_input":"2025-07-08T06:49:09.386629Z","iopub.status.idle":"2025-07-08T06:49:10.416697Z","shell.execute_reply.started":"2025-07-08T06:49:09.386581Z","shell.execute_reply":"2025-07-08T06:49:10.415581Z"}},"outputs":[],"execution_count":null}]}