{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":91249,"databundleVersionId":11294684,"sourceType":"competition"},{"sourceId":11854363,"sourceType":"datasetVersion","datasetId":7448781},{"sourceId":11870207,"sourceType":"datasetVersion","datasetId":7459539},{"sourceId":11901983,"sourceType":"datasetVersion","datasetId":7481775},{"sourceId":11932060,"sourceType":"datasetVersion","datasetId":7501747},{"sourceId":327336,"sourceType":"modelInstanceVersion","modelInstanceId":274744,"modelId":295634}],"dockerImageVersionId":30919,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# !tar xfz /kaggle/input/ultralytics-for-offline-install/archive.tar.gz\n# !pip install --no-index --find-links=./packages -q ultralytics\n# !rm -rf ./packages\n# print(\"package installed ...........................\")\n\n!cp -r /kaggle/input/mhafyolo/pytorch/default/1/MHAF-YOLO-main /kaggle/working/\nprint('PIP INSTALL OK!!!')","metadata":{"trusted":true,"_kg_hide-input":false,"execution":{"iopub.status.busy":"2025-06-03T14:13:23.945633Z","iopub.execute_input":"2025-06-03T14:13:23.945867Z","iopub.status.idle":"2025-06-03T14:13:24.905256Z","shell.execute_reply.started":"2025-06-03T14:13:23.945846Z","shell.execute_reply":"2025-06-03T14:13:24.904279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport torch\nimport cv2\nfrom tqdm.notebook import tqdm\nfrom pathlib import Path\ncurrent_dir = Path.cwd()\nprint(\"this_dir:\", current_dir)\n\ntarget_dir = Path(\"/kaggle/working/MHAF-YOLO-main\") \nos.chdir(target_dir)\n\nfrom ultralytics import YOLOv10\nimport threading\nimport time\nfrom contextlib import nullcontext\nfrom concurrent.futures import ThreadPoolExecutor\nfrom ultralytics.utils.ops import non_max_suppression","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:13:24.906129Z","iopub.execute_input":"2025-06-03T14:13:24.90642Z","iopub.status.idle":"2025-06-03T14:13:40.45704Z","shell.execute_reply.started":"2025-06-03T14:13:24.906396Z","shell.execute_reply":"2025-06-03T14:13:40.456013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set random seed for reproducibility\nnp.random.seed(42)\ntorch.manual_seed(42)\n\n# Define paths\ndata_path = \"/kaggle/input/byu-locating-bacterial-flagellar-motors-2025/\"\ntest_dir = os.path.join(data_path, \"test\")\nsubmission_path = \"/kaggle/working/submission.csv\"\n\n# Model path - adjust if your best model is saved in a different location\nmodel_path = \"/kaggle/input/box20-train/box20__938_531F0.pt\"\n\n# Detection parameters\n#CONFIDENCE_THRESHOLD = 0.65  # Lower threshold to catch more potential motors\nMAX_DETECTIONS_PER_TOMO = 1  # Keep track of top N detections per tomogram\nNMS_IOU_THRESHOLD = 0.2  # Non-maximum suppression threshold for 3D clustering\nCONCENTRATION = 0.5 # ONLY PROCESS 1/20 slices for fast submission\nSIZE = 1024","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:13:40.458221Z","iopub.execute_input":"2025-06-03T14:13:40.458909Z","iopub.status.idle":"2025-06-03T14:13:40.470424Z","shell.execute_reply.started":"2025-06-03T14:13:40.458866Z","shell.execute_reply":"2025-06-03T14:13:40.469261Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# GPU profiling context manager\nclass GPUProfiler:\n    def __init__(self, name):\n        self.name = name\n        self.start_time = None\n        \n    def __enter__(self):\n        if torch.cuda.is_available():\n            torch.cuda.synchronize()\n        self.start_time = time.time()\n        return self\n        \n    def __exit__(self, *args):\n        if torch.cuda.is_available():\n            torch.cuda.synchronize()\n        elapsed = time.time() - self.start_time\n        print(f\"[PROFILE] {self.name}: {elapsed:.3f}s\")\n\n# Check GPU availability and set up optimizations\ndevice = 'cuda:0' if torch.cuda.is_available() else 'cpu'\n# device = ['cuda:0', 'cuda:1']\nBATCH_SIZE = 8  # Default batch size, will be adjusted dynamically if GPU available\n\nif device.startswith('cuda'):\n    # Set CUDA optimization flags\n    torch.backends.cudnn.benchmark = True\n    torch.backends.cudnn.deterministic = False\n    torch.backends.cuda.matmul.allow_tf32 = True  # Allow TF32 on Ampere GPUs\n    torch.backends.cudnn.allow_tf32 = True\n    \n    # Print GPU info\n    gpu_name = torch.cuda.get_device_name(0)\n    gpu_mem = torch.cuda.get_device_properties(0).total_memory / 1e9  # Convert to GB\n    print(f\"Using GPU: {gpu_name} with {gpu_mem:.2f} GB memory\")\n    \n    # Get available GPU memory and set batch size accordingly\n    free_mem = gpu_mem - torch.cuda.memory_allocated(0) / 1e9\n    BATCH_SIZE = max(8, min(32, int(free_mem * 4)))  # 4 images per GB as rough estimate\n    print(f\"Dynamic batch size set to {BATCH_SIZE} based on {free_mem:.2f}GB free memory\")\nelse:\n    print(\"GPU not available, using CPU\")\n    BATCH_SIZE = 8  # Reduce batch size for CPU","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:13:40.471427Z","iopub.execute_input":"2025-06-03T14:13:40.471783Z","iopub.status.idle":"2025-06-03T14:13:40.566766Z","shell.execute_reply.started":"2025-06-03T14:13:40.471717Z","shell.execute_reply":"2025-06-03T14:13:40.565926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def weighted_box_fusion(boxes, iou_threshold=0.4):\n    \"\"\"Applies Weighted Box Fusion to combine overlapping bounding boxes.\"\"\"\n    fused_boxes = []\n    # print(type(boxes))\n    used = [False] * len(boxes)\n\n    for i in range(len(boxes)):\n        if used[i]:\n            continue\n        \n        similar_boxes = [boxes[i]]\n        used[i] = True\n\n        for j in range(i + 1, len(boxes)):\n            if used[j]:\n                continue\n\n            if iou(boxes[i], boxes[j]) > iou_threshold:\n                similar_boxes.append(boxes[j])\n                used[j] = True\n\n        # Compute weighted average for the final box\n        # similar_boxes = np.array(similar_boxes)\n        # similar_boxes = similar_boxes.cpu().numpy()\n        similar_boxes = np.array([t.cpu() for t in similar_boxes])\n        # print(\"similar_boxes: \",similar_boxes)\n        confidences = similar_boxes[:,:, 4]\n        weights = confidences / confidences.sum()\n\n        fused_x1 = np.sum(similar_boxes[:, :, 0] * weights)\n        fused_y1 = np.sum(similar_boxes[:, :, 1] * weights)\n        fused_x2 = np.sum(similar_boxes[:, :, 2] * weights)\n        fused_y2 = np.sum(similar_boxes[:, :, 3] * weights)\n        fused_confidence = np.mean(confidences) #np.max(confidences)  # Take max confidence\n\n        fused_boxes.append((fused_x1, fused_y1, fused_x2, fused_y2, fused_confidence))\n\n    return fused_boxes\n\ndef predict_ensemble_tta(single_model, image_np, device, img_size):\n    \"\"\"\n    For a 640x640 numpy image:\n    - Multiple models (list of models)\n    - Multiple TTA (original, hflip, vflip, rot90)\n    Do the NMS for the last time\n    Return: [K,6] => x1,y1,x2,y2,conf,cls\n    \"\"\"\n    all_boxes = []\n    all_confs = []\n    all_clss = []\n\n    def do_infer(img_tta, invert_func):\n        #for m in models:\n            res = single_model(img_tta, \n                            imgsz=img_size, \n                            # conf=conf_thres,\n                            device=device, \n                            verbose=False)\n            # res = single_model(img_tta,verbose=False)\n            for r in res:\n                # print('res box:', r.boxes)\n                boxes = r.boxes\n                if boxes is None or len(boxes)==0:\n                    # print('predict box is none!')\n                    continue\n                xyxy = boxes.xyxy.cpu().numpy()\n                confs = boxes.conf.cpu().numpy()\n                clss = boxes.cls.cpu().numpy().astype(int)\n                # Inverse transformation\n                xyxy_orig = invert_func(xyxy)\n                all_boxes.append(xyxy_orig)\n                all_confs.append(confs)\n                all_clss.append(clss)\n\n    # Master drawing\n    do_infer(image_np, invert_func=lambda x: x)\n\n    # Horizontal flip\n    org_img_h, org_img_w = image_np.shape[:2]\n    img_hflip = cv2.flip(image_np, 1)\n    def invert_hflip(xyxy):\n        new_ = xyxy.copy()\n        x1 = org_img_w - xyxy[:,2]\n        x2 = org_img_w - xyxy[:,0]\n        new_[:,0] = x1\n        new_[:,2] = x2\n        return new_\n    #do_infer(img_hflip, invert_func=invert_hflip)\n\n    # Vertical flip\n    # img_vflip = cv2.flip(image_np, 0)\n    # def invert_vflip(xyxy):\n    #     new_ = xyxy.copy()\n    #     y1 = img_size - xyxy[:,3]\n    #     y2 = img_size - xyxy[:,1]\n    #     new_[:,1] = y1\n    #     new_[:,3] = y2\n    #     return new_\n    # do_infer(img_vflip, invert_func=invert_vflip)\n\n    # Rotate 90 degrees (clockwise)\n    # img_rot90 = cv2.rotate(image_np, cv2.ROTATE_90_CLOCKWISE)\n    # def invert_rot90(xyxy):\n    #     new_ = xyxy.copy()\n    #     # (x,y)->(y,640-x)\n    #     # Inverse transform (x',y')->(640-y',x')\n    #     x1_old,y1_old = xyxy[:,0], xyxy[:,1]\n    #     x2_old,y2_old = xyxy[:,2], xyxy[:,3]\n    #     X1 = img_size - y2_old\n    #     Y1 = x1_old\n    #     X2 = img_size - y1_old\n    #     Y2 = x2_old\n    #     new_[:,0] = X1\n    #     new_[:,1] = Y1\n    #     new_[:,2] = X2\n    #     new_[:,3] = Y2\n    #     return new_\n    # do_infer(img_rot90, invert_func=invert_rot90)\n\n    if len(all_boxes)==0:\n        # print('all boxes is None!')\n        return None\n\n    boxes_cat = np.concatenate(all_boxes, axis=0)\n    confs_cat = np.concatenate(all_confs, axis=0)\n    clss_cat  = np.concatenate(all_clss, axis=0)\n    cat_data = np.column_stack([boxes_cat, confs_cat, clss_cat])  # shape [N,6]\n    # print('final:',cat_data)\n\n    # Need to add batch dimension => [1,N,6]\n    cat_tensor = torch.from_numpy(cat_data).float().unsqueeze(0).to(device)\n    # NMS \n    # nms_out = non_max_suppression(cat_tensor, iou_thres=0.5, max_det=300)\n    nms_out = weighted_box_fusion(cat_tensor, iou_threshold=0.5)\n    # nms_out => list length =1(a graph), take nms_out[0]\n    if len(nms_out)==0 or nms_out[0] is None or len(nms_out[0])==0:\n        return None\n    # final_nms = nms_out[0].cpu().numpy()  # shape [K,6]\n    final_nms = list(nms_out[0]) # shape [K,6]\n    # print(\"final_nms : \", final_nms)\n    return final_nms","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:13:40.568753Z","iopub.execute_input":"2025-06-03T14:13:40.569045Z","iopub.status.idle":"2025-06-03T14:13:40.584317Z","shell.execute_reply.started":"2025-06-03T14:13:40.569019Z","shell.execute_reply":"2025-06-03T14:13:40.583346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def make_predict(sub_path, model, device, img_size):\n    # print(\"img path:\", sub_path)\n    res_list = []\n    for img in sub_path:\n        # print('images:', img)\n        img_np = cv2.imread(img)\n\n        if img_np is None:\n            res_list.append((0, 0, 0, 0, 0.0))\n            continue\n        \n        res_nms = predict_ensemble_tta(model, img_np, device, img_size = img_size)\n        if res_nms is None:\n            # print('nms is none!')\n            continue\n        else:\n            res_list.append(res_nms)\n\n    return res_list","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:13:40.585524Z","iopub.execute_input":"2025-06-03T14:13:40.585826Z","iopub.status.idle":"2025-06-03T14:13:40.603855Z","shell.execute_reply.started":"2025-06-03T14:13:40.585799Z","shell.execute_reply":"2025-06-03T14:13:40.603085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def normalize_slice(slice_data):\n    \"\"\"\n    Normalize slice data using 2nd and 98th percentiles for better contrast\n    \"\"\"\n    p2 = np.percentile(slice_data, 2)\n    p98 = np.percentile(slice_data, 98)\n    clipped_data = np.clip(slice_data, p2, p98)\n    normalized = 255 * (clipped_data - p2) / (p98 - p2)\n    return np.uint8(normalized)\n\ndef preload_image_batch(file_paths):\n    \"\"\"Preload a batch of images to CPU memory\"\"\"\n    images = []\n    for path in file_paths:\n        img = cv2.imread(path)\n        if img is None:\n            # Try with PIL as fallback\n            img = np.array(Image.open(path))\n        images.append(img)\n    return images\n\ndef process_tomogram(tomo_id, model,CONFIDENCE_THRESHOLD, index=0, total=1,SIZE=SIZE):\n    \"\"\"\n    Process a single tomogram and return the most confident motor detection\n    \"\"\"\n    print(f\"Processing tomogram {tomo_id} ({index}/{total})\")\n    \n    # Get all slice files for this tomogram\n    tomo_dir = os.path.join(test_dir, tomo_id)\n    slice_files = sorted([f for f in os.listdir(tomo_dir) if f.endswith('.jpg')])\n    \n    # Apply CONCENTRATION to reduce the number of slices processed\n    # This will process approximately CONCENTRATION fraction of all slices\n    selected_indices = np.linspace(0, len(slice_files)-1, int(len(slice_files) * CONCENTRATION))\n    selected_indices = np.round(selected_indices).astype(int)\n    slice_files = [slice_files[i] for i in selected_indices]\n    \n    print(f\"Processing {len(slice_files)} out of {len(os.listdir(tomo_dir))} slices based on CONCENTRATION={CONCENTRATION}\")\n    \n    # Create a list to store all detections\n    all_detections = []\n    \n    # Create CUDA streams for parallel processing if using GPU\n    if device.startswith('cuda'):\n        streams = [torch.cuda.Stream() for _ in range(min(8, BATCH_SIZE))]\n    else:\n        streams = [None]\n    \n    # Variables for preloading\n    next_batch_thread = None\n    next_batch_images = None\n    \n    # Process slices in batches\n    for batch_start in range(0, len(slice_files), BATCH_SIZE):\n        # Wait for previous preload thread if it exists\n        if next_batch_thread is not None:\n            next_batch_thread.join()\n            next_batch_images = None\n            \n        batch_end = min(batch_start + BATCH_SIZE, len(slice_files))\n        batch_files = slice_files[batch_start:batch_end]\n        \n        # Start preloading next batch\n        next_batch_start = batch_end\n        next_batch_end = min(next_batch_start + BATCH_SIZE, len(slice_files))\n        next_batch_files = slice_files[next_batch_start:next_batch_end] if next_batch_start < len(slice_files) else []\n        \n        if next_batch_files:\n            next_batch_paths = [os.path.join(tomo_dir, f) for f in next_batch_files]\n            next_batch_thread = threading.Thread(target=preload_image_batch, args=(next_batch_paths,))\n            next_batch_thread.start()\n        else:\n            next_batch_thread = None\n        \n        # Split batch across streams for parallel processing\n        sub_batches = np.array_split(batch_files, len(streams))\n        sub_batch_results = []\n        \n        for i, sub_batch in enumerate(sub_batches):\n            if len(sub_batch) == 0:\n                continue\n                \n            stream = streams[i % len(streams)]\n            with torch.cuda.stream(stream) if stream and device.startswith('cuda') else nullcontext():\n                # Process sub-batch\n                sub_batch_paths = [os.path.join(tomo_dir, slice_file) for slice_file in sub_batch]\n                sub_batch_slice_nums = [int(slice_file.split('_')[1].split('.')[0]) for slice_file in sub_batch]\n                \n                # Run inference with profiling\n                with GPUProfiler(f\"Inference batch {i+1}/{len(sub_batches)}\"):\n                    # sub_results = model(sub_batch_paths, verbose=False)\n                    sub_results = make_predict(sub_batch_paths, model, device, SIZE)\n                    # print(sub_results)\n                \n                # Process each result in this sub-batch\n                # for result in sub_results:\n                    # print('nms_res1:', result)\n                for j,res in enumerate(sub_results):\n                    # print('nms_res2', res)\n                    # x1,y1,x2,y2, confidence, cls_ = res\n                    x1,y1,x2,y2, confidence = res\n                    if confidence >= CONFIDENCE_THRESHOLD:\n                        # Calculate center coordinates\n                        x_center = (x1 + x2) / 2\n                        y_center = (y1 + y2) / 2\n                                \n                        # Store detection with 3D coordinates\n                        all_detections.append({\n                                'z': round(sub_batch_slice_nums[j]),\n                                'y': round(y_center),\n                                'x': round(x_center),\n                                'confidence': float(confidence)\n                            })\n                    # print(\"all_detections:\", all_detections)\n\n                    # if len(result.boxes) > 0:\n                    #     boxes = result.boxes\n                    #     for box_idx, confidence in enumerate(boxes.conf):\n                    #         if confidence >= CONFIDENCE_THRESHOLD:\n                    #             # Get bounding box coordinates\n                    #             x1, y1, x2, y2 = boxes.xyxy[box_idx].cpu().numpy()\n                                \n                    #             # Calculate center coordinates\n                    #             x_center = (x1 + x2) / 2\n                    #             y_center = (y1 + y2) / 2\n                                \n                    #             # Store detection with 3D coordinates\n                    #             all_detections.append({\n                    #                 'z': round(sub_batch_slice_nums[j]),\n                    #                 'y': round(y_center),\n                    #                 'x': round(x_center),\n                    #                 'confidence': float(confidence)\n                    #             })\n        \n        # Synchronize streams\n        if device.startswith('cuda'):\n            torch.cuda.synchronize()\n    \n    # Clean up thread if still running\n    if next_batch_thread is not None:\n        next_batch_thread.join()\n    \n    # 3D Non-Maximum Suppression to merge nearby detections across slices\n    final_detections = perform_3d_nms(all_detections, NMS_IOU_THRESHOLD)\n    \n    # Sort detections by confidence (highest first)\n    final_detections.sort(key=lambda x: x['confidence'], reverse=True)\n    \n    # If there are no detections, return NA values\n    if not final_detections:\n        return {\n            'tomo_id': tomo_id,\n            'Motor axis 0': -1,\n            'Motor axis 1': -1,\n            'Motor axis 2': -1,\n            'cof': 0\n        }\n    \n    # Take the detection with highest confidence\n    best_detection = final_detections[0]\n    \n    # Return result with integer coordinates\n    return {\n        'tomo_id': tomo_id,\n        'Motor axis 0': round(best_detection['z']),\n        'Motor axis 1': round(best_detection['y']),\n        'Motor axis 2': round(best_detection['x']),\n        'cof':best_detection['confidence']\n    }\n\ndef perform_3d_nms(detections, iou_threshold):\n    \"\"\"\n    Perform 3D Non-Maximum Suppression on detections to merge nearby motors\n    \"\"\"\n    if not detections:\n        return []\n    \n    # Sort by confidence (highest first)\n    detections = sorted(detections, key=lambda x: x['confidence'], reverse=True)\n    \n    # List to store final detections after NMS\n    final_detections = []\n    \n    # Define 3D distance function\n    def distance_3d(d1, d2):\n        return np.sqrt((d1['z'] - d2['z'])**2 + \n                       (d1['y'] - d2['y'])**2 + \n                       (d1['x'] - d2['x'])**2)\n    \n    # Maximum distance threshold (based on box size and slice gap)\n    box_size = 24  # Same as annotation box size\n    distance_threshold = box_size * iou_threshold\n    \n    # Process each detection\n    while detections:\n        # Take the detection with highest confidence\n        best_detection = detections.pop(0)\n        final_detections.append(best_detection)\n        \n        # Filter out detections that are too close to the best detection\n        detections = [d for d in detections if distance_3d(d, best_detection) > distance_threshold]\n    \n    return final_detections\n\ndef generate_submission(model_path,submission_path,CONFIDENCE_THRESHOLD):\n    \"\"\"\n    Main function to generate the submission file\n    \"\"\"\n    # Get list of test tomograms\n    test_tomos = sorted([d for d in os.listdir(test_dir) if os.path.isdir(os.path.join(test_dir, d))])\n    total_tomos = len(test_tomos)\n    \n    print(f\"Found {total_tomos} tomograms in test directory\")\n    \n    # Debug image loading for the first tomogram\n    # if test_tomos:\n    #     debug_image_loading(test_tomos[0])\n    \n    # Clear GPU cache before starting\n    if torch.cuda.is_available():\n        torch.cuda.empty_cache()\n    \n    # Initialize model once outside the processing loop\n    print(f\"Loading YOLO model from {model_path}\")\n    model = YOLOv10(model_path)\n    model.to(device)\n    \n    # Additional optimizations for inference\n    if device.startswith('cuda'):\n        # Fuse conv and bn layers for faster inference\n        model.fuse()\n        \n        # Enable model half precision (FP16) if on compatible GPU\n        if torch.cuda.get_device_capability(0)[0] >= 7:  # Volta or newer\n            model.model.half()\n            print(\"Using half precision (FP16) for inference\")\n    \n    # Process tomograms with parallelization\n    results = []\n    motors_found = 0\n    \n    # Using ThreadPoolExecutor with max_workers=1 since each worker uses the GPU already\n    # and we're parallelizing within each tomogram processing\n    with ThreadPoolExecutor(max_workers=4) as executor:\n        future_to_tomo = {}\n        \n        # Submit all tomograms for processing\n        for i, tomo_id in enumerate(test_tomos, 1):\n            future = executor.submit(process_tomogram, tomo_id, model,CONFIDENCE_THRESHOLD, i, total_tomos)\n            future_to_tomo[future] = tomo_id\n        \n        # Process completed futures as they complete\n        for future in future_to_tomo:\n            tomo_id = future_to_tomo[future]\n            if torch.cuda.is_available():\n                    torch.cuda.empty_cache()\n                    \n            result = future.result()\n            results.append(result)\n                \n                # Update motors found count\n            has_motor = not pd.isna(result['Motor axis 0'])\n            if has_motor:\n                motors_found += 1\n                print(f\"Motor found in {tomo_id} at position: \"\n                      f\"z={result['Motor axis 0']}, y={result['Motor axis 1']}, x={result['Motor axis 2']}\")\n            else:\n                print(f\"No motor detected in {tomo_id}\")\n                    \n            print(f\"Current detection rate: {motors_found}/{len(results)} ({motors_found/len(results)*100:.1f}%)\")\n            \n            # try:\n            #     # Clear CUDA cache between tomograms\n            #     if torch.cuda.is_available():\n            #         torch.cuda.empty_cache()\n                    \n            #     result = future.result()\n            #     results.append(result)\n                \n            #     # Update motors found count\n            #     has_motor = not pd.isna(result['Motor axis 0'])\n            #     if has_motor:\n            #         motors_found += 1\n            #         print(f\"Motor found in {tomo_id} at position: \"\n            #               f\"z={result['Motor axis 0']}, y={result['Motor axis 1']}, x={result['Motor axis 2']}\")\n            #     else:\n            #         print(f\"No motor detected in {tomo_id}\")\n                    \n            #     print(f\"Current detection rate: {motors_found}/{len(results)} ({motors_found/len(results)*100:.1f}%)\")\n            \n            # except Exception as e:\n            #     print(f\"Error processing {tomo_id}: {e}\")\n            #     # Create a default entry for failed tomograms\n            #     results.append({\n            #         'tomo_id': tomo_id,\n            #         'Motor axis 0': -1,\n            #         'Motor axis 1': -1,\n            #         'Motor axis 2': -1\n            #     })\n    \n    # Create submission dataframe\n    submission_df = pd.DataFrame(results)\n    \n    # Ensure proper column order\n    submission_df = submission_df[['tomo_id', 'Motor axis 0', 'Motor axis 1', 'Motor axis 2', 'cof']]\n    \n    # Save the submission file\n    submission_df.to_csv(submission_path, index=False)\n    \n    print(f\"\\nSubmission complete!\")\n    print(f\"Motors detected: {motors_found}/{total_tomos} ({motors_found/total_tomos*100:.1f}%)\")\n    print(f\"Submission saved to: {submission_path}\")\n    \n    # Display first few rows of submission\n    print(\"\\nSubmission preview:\")\n    print(submission_df.head())\n    \n    return submission_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:13:40.60478Z","iopub.execute_input":"2025-06-03T14:13:40.605131Z","iopub.status.idle":"2025-06-03T14:13:40.630471Z","shell.execute_reply.started":"2025-06-03T14:13:40.605099Z","shell.execute_reply":"2025-06-03T14:13:40.629488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from functools import reduce # 用于多次merge\n\ndef merge_multiple_dataframes(dataframes_list, final_output_path_merge):\n    \"\"\"\n    融合多个模型的预测结果 (DataFrames列表)。\n    每个 DataFrame 应包含 'tomo_id' 和 'cof' (置信度) 以及坐标列。\n    \"\"\"\n    if not dataframes_list:\n        print(\"错误: 没有提供DataFrame进行融合。\")\n        return pd.DataFrame(columns=['tomo_id', 'Motor axis 0', 'Motor axis 1', 'Motor axis 2'])\n\n    # 为每个DataFrame的列重命名以添加模型特定的后缀 (除了 'tomo_id')\n    renamed_dfs = []\n    for i, df_original in enumerate(dataframes_list):\n        suffix = f'_m{i+1}' # 例如 _m1, _m2, ...\n        df_renamed = df_original.rename(columns=lambda c: c + suffix if c != 'tomo_id' else c)\n        renamed_dfs.append(df_renamed)\n\n    # 使用 functools.reduce 和 pd.merge 进行多次外连接\n    # 初始DataFrame是第一个重命名后的DataFrame\n    if not renamed_dfs: # 如果列表为空 (理论上不会到这里，因为上面有检查)\n        return pd.DataFrame(columns=['tomo_id', 'Motor axis 0', 'Motor axis 1', 'Motor axis 2'])\n        \n    merged_df_all_models = reduce(lambda left, right: pd.merge(left, right, on='tomo_id', how='outer'), renamed_dfs)\n    \n    results_final_merge = []\n    for index, row in merged_df_all_models.iterrows():\n        tomo_id = row['tomo_id']\n        \n        best_cof = -1.0\n        selected_coords = {'z': -1, 'y': -1, 'x': -1} # 默认无检测\n\n        # 遍历每个模型的预测结果\n        for i in range(len(dataframes_list)):\n            model_suffix = f'_m{i+1}'\n            current_cof = row.get(f'cof{model_suffix}', 0.0) # 获取置信度, 缺失则为0\n            \n            # 如果当前模型的置信度更高，则更新最佳选择\n            # 如果置信度相同，默认会保留第一个达到该最高置信度的模型的结果\n            if current_cof > best_cof:\n                best_cof = current_cof\n                selected_coords['z'] = row.get(f'Motor axis 0{model_suffix}', -1)\n                selected_coords['y'] = row.get(f'Motor axis 1{model_suffix}', -1)\n                selected_coords['x'] = row.get(f'Motor axis 2{model_suffix}', -1)\n            # 如果你需要更复杂的相同置信度处理逻辑（例如，偏好特定模型），可以在这里添加\n        \n        # 确保坐标是整数\n        results_final_merge.append({\n            'tomo_id': tomo_id,\n            'Motor axis 0': int(selected_coords['z']),\n            'Motor axis 1': int(selected_coords['y']),\n            'Motor axis 2': int(selected_coords['x']),\n            # 'cof_chosen': best_cof # 可以选择性保留最终选定的置信度进行分析\n        })\n\n    final_submission_df = pd.DataFrame(results_final_merge)\n    \n    # 确保最终列的顺序，并且只有竞赛需要的列\n    final_submission_df = final_submission_df[['tomo_id', 'Motor axis 0', 'Motor axis 1', 'Motor axis 2']]\n    \n    final_submission_df.to_csv(final_output_path_merge, index=False)\n    print(f\"融合了 {len(dataframes_list)} 个模型的提交文件已保存至: {final_output_path_merge}\")\n    return final_submission_df\n\n\ndef prosse_multiple_submissions(csv_paths: list, final_output_path: str):\n    \"\"\"\n    读取多个CSV文件，调用通用融合函数，并返回最终的DataFrame。\n    \"\"\"\n    print(f\"\\n开始融合以下 {len(csv_paths)} 个CSV文件: \")\n    for i, path in enumerate(csv_paths):\n        print(f\"  {i+1}: {path}\")\n\n    loaded_dfs = []\n    try:\n        for i, path in enumerate(csv_paths):\n            df_single_model = pd.read_csv(path)\n            # 确保 'cof' 列存在，并处理NaN\n            if 'cof' not in df_single_model.columns:\n                print(f\"警告: 文件 {path} 中缺少 'cof' 列，将添加默认值0。\")\n                df_single_model['cof'] = 0.0\n            \n            coord_cols_check = ['Motor axis 0', 'Motor axis 1', 'Motor axis 2']\n            for col_name in coord_cols_check:\n                if col_name in df_single_model.columns:\n                    df_single_model[col_name] = df_single_model[col_name].fillna(-1).astype(int)\n                else: # 如果坐标列不存在，添加默认值\n                    print(f\"警告: 文件 {path} 中缺少 '{col_name}' 列，将添加默认值-1。\")\n                    df_single_model[col_name] = -1\n            df_single_model['cof'] = df_single_model['cof'].fillna(0.0)\n            \n            loaded_dfs.append(df_single_model)\n\n        if not loaded_dfs:\n            print(\"错误:未能加载任何CSV文件进行融合。\")\n            return pd.DataFrame(columns=['tomo_id', 'Motor axis 0', 'Motor axis 1', 'Motor axis 2'])\n\n        merged_df = merge_multiple_dataframes(loaded_dfs, final_output_path)\n        print(merged_df.head())\n        return merged_df\n        \n    except FileNotFoundError as e:\n        print(f\"错误: 无法找到一个或多个CSV文件进行融合: {e}\")\n        return pd.DataFrame(columns=['tomo_id', 'Motor axis 0', 'Motor axis 1', 'Motor axis 2'])\n    except Exception as e_merge_main:\n        print(f\"融合过程中发生未知错误: {e_merge_main}\")\n        return pd.DataFrame(columns=['tomo_id', 'Motor axis 0', 'Motor axis 1', 'Motor axis 2'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:13:40.631374Z","iopub.execute_input":"2025-06-03T14:13:40.631667Z","iopub.status.idle":"2025-06-03T14:13:40.653997Z","shell.execute_reply.started":"2025-06-03T14:13:40.631636Z","shell.execute_reply":"2025-06-03T14:13:40.653182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Run the submission pipeline\nif __name__ == \"__main__\":\n    # Time entire process\n    start_time = time.time()\n\n    submission_path1 = \"/kaggle/working/submission1.csv\"\n    model_path1 = \"/kaggle/input/extraslice4-f0/extra_slice4_train_F0_mac14.pt\"\n    CONFIDENCE_THRESHOLD1 = 0.75\n\n    submission_path2 = \"/kaggle/working/submission2.csv\"\n    model_path2 = \"/kaggle/input/mahf-yolo-train/mayolov2f.pt\"\n    CONFIDENCE_THRESHOLD2 = 0.80\n\n    submission_path3 = \"/kaggle/working/submission3.csv\"\n    model_path3 = \"/kaggle/input/box20-train/box20__938_531F0.pt\"\n    CONFIDENCE_THRESHOLD3 = 0.75\n\n    submission_path4 = \"/kaggle/working/submission4.csv\"\n    model_path4 = \"/kaggle/input/slice4-box24-noclean-960/slice4_box24_F0_960.pt\"\n    CONFIDENCE_THRESHOLD4 = 0.75\n    \n    print(f\"\\n--- 开始为模型1生成预测 ({os.path.basename(model_path1)}) ---\")\n    \n    submission1 = generate_submission(\n        model_path=model_path1,\n        submission_path=submission_path1,\n        CONFIDENCE_THRESHOLD=CONFIDENCE_THRESHOLD1\n    )\n\n    print(f\"\\n--- 开始为模型2生成预测 ({os.path.basename(model_path2)}) ---\")\n    submission2 = generate_submission(\n        model_path=model_path2,\n        submission_path=submission_path2,\n        CONFIDENCE_THRESHOLD=CONFIDENCE_THRESHOLD2\n    )\n\n    print(f\"\\n--- 开始为模型3生成预测 ({os.path.basename(model_path3)}) ---\")\n    submission3 = generate_submission(\n        model_path=model_path3,\n        submission_path=submission_path3,\n        CONFIDENCE_THRESHOLD=CONFIDENCE_THRESHOLD3,\n    )\n\n    print(f\"\\n--- 开始为模型4生成预测 ({os.path.basename(model_path4)}) ---\")\n    submission3 = generate_submission(\n        model_path=model_path4,\n        submission_path=submission_path4,\n        CONFIDENCE_THRESHOLD=CONFIDENCE_THRESHOLD4,\n    )\n\n    submission_path = \"/kaggle/working/submission.csv\"\n    # --- 调用新的融合函数处理四个CSV文件 ---\n    print(\"\\n--- 开始融合四个模型的预测结果 ---\")\n    intermediate_csv_paths = [submission_path1, submission_path2, submission_path3, submission_path4]\n    final_submission_df = prosse_multiple_submissions( # 你需要定义这个新的通用融合函数\n        csv_paths=intermediate_csv_paths,\n        final_output_path=submission_path # submission_path 是全局的最终输出路径, e.g., \"/kaggle/working/submission.csv\"\n    )\n    \n    \n    # Generate submission\n    \n    # Print total execution time\n    elapsed = time.time() - start_time\n    print(f\"\\nTotal execution time: {elapsed:.2f} seconds ({elapsed/60:.2f} minutes)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:13:40.654884Z","iopub.execute_input":"2025-06-03T14:13:40.655149Z","iopub.status.idle":"2025-06-03T14:18:12.628016Z","shell.execute_reply.started":"2025-06-03T14:13:40.655121Z","shell.execute_reply":"2025-06-03T14:18:12.627344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# tomo_00e047\t169\t546\t603\n# tomo_01a877\t147\t638\t287\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:18:12.628913Z","iopub.execute_input":"2025-06-03T14:18:12.629221Z","iopub.status.idle":"2025-06-03T14:18:12.632711Z","shell.execute_reply.started":"2025-06-03T14:18:12.629188Z","shell.execute_reply":"2025-06-03T14:18:12.631948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}