{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91249,"databundleVersionId":11294684,"sourceType":"competition"},{"sourceId":11345926,"sourceType":"datasetVersion","datasetId":7099158},{"sourceId":11346218,"sourceType":"datasetVersion","datasetId":7099370},{"sourceId":11358982,"sourceType":"datasetVersion","datasetId":7109145},{"sourceId":11372136,"sourceType":"datasetVersion","datasetId":7119250},{"sourceId":11764439,"sourceType":"datasetVersion","datasetId":7385610},{"sourceId":11921703,"sourceType":"datasetVersion","datasetId":6958946},{"sourceId":11922391,"sourceType":"datasetVersion","datasetId":7495488},{"sourceId":229483235,"sourceType":"kernelVersion"}],"dockerImageVersionId":30919,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1 style=\"color: #6cb4e4;  text-align: center;  padding: 0.25em;  border-top: solid 2.5px #6cb4e4;  border-bottom: solid 2.5px #6cb4e4;  background: -webkit-repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);  background: repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);height:45px;\">\n<b>\nOnly Submission(LoadLocalTrainModel)\n</b></h1> ","metadata":{}},{"cell_type":"markdown","source":"### **ℹ️INFO**\n* First of all, I'm grateful to the host for sharing such a great bass line.\n* This notebook is an inference notebook that performed a unique LocalTrain based on the Train/Inference published by the host.\n    * https://www.kaggle.com/code/andrewjdarley/train-yolo\n    * https://www.kaggle.com/code/andrewjdarley/submission-notebook","metadata":{}},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"## **》》》 [IMPORTANT] Env Params**","metadata":{}},{"cell_type":"code","source":"#model_path=\"/kaggle/input/byu-me-models/yolov8-s-960-Loss-71/best.pt\"\n# model_path=\"/kaggle/input/byu-me-models/yolov8-s-960-Loss-68-fb89-a2/best.pt\"\n# model_path=\"/kaggle/input/byu-me-models/yolov8-s-960-e-133-Loss-65-a2/best.pt\"\n\n# model_path=\"/kaggle/input/byu-models-yolo8lv5-co/content/byu_models_yolo8l/flag_motormixdatacont_s3/weights/best91-90-922-706.pt\"\n# model_path_758=\"/kaggle/input/byu-models-yolo8lv2-co/content/byu_models_yolo8l/flagellar_motorcont/weights/best.pt\"\n# model_path_s2= \"/kaggle/input/byu-training-yolo/runs/detect/flagellar_motors2/weights/best.pt\"\n\n\n# model_path_23_cont30=\"/kaggle/input/byu-models-yolo8lv2-co/content/byu_models_yolo8l/flagellar_motorsextra30cont/weights/best.pt\"\n\nmodel_path_837=\"/kaggle/input/byu-models-yolo8lv66-co/content/modelsbyu/max_recall_e72-p0.90-r0.90-m50.92-m590.63-b1.56-c0.55-d0.69.pt\"\nmodel_path_793=\"/kaggle/input/byu-me-models/yolov8-s-960-Loss-68-a2/best.pt\"\n\n\n\nmodel_path_841=\"/kaggle/input/byu-high-lb-modelss/byu-yolo-modelsx/byu_train8x/byu_train8xs3n20/weights/max_recall_e37-p0.90-r0.90-m50.92-m590.63-b1.10_1.11-c0.27_0.32-d0.63_0.48.pt\"\nmodel_path_822=\"/kaggle/input/byu-me-models/byu_train960d3A500v2/weights/max_recall_e88-p0.74-r0.55-m50.61-m590.37-b1.32-c0.80-d0.53.pt\"\n\nmodels_config = [\n    {\"path\": model_path_841, \"conf\": 0.55, \"start\": 0, \"step\": 2, \"imgsz\": 896,\"augment\":True, \"weight\": 1.0},\n    {\"path\": model_path_822, \"conf\": 0.55, \"start\": 0, \"step\": 2, \"imgsz\": 960,\"augment\":True, \"weight\": 1.0},\n]\nlen(models_config)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T00:30:26.557767Z","iopub.execute_input":"2025-05-22T00:30:26.558056Z","iopub.status.idle":"2025-05-22T00:30:26.571232Z","shell.execute_reply.started":"2025-05-22T00:30:26.558024Z","shell.execute_reply":"2025-05-22T00:30:26.57053Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1 style=\"color: #6cb4e4;  text-align: center;  padding: 0.25em;  border-top: solid 2.5px #6cb4e4;  border-bottom: solid 2.5px #6cb4e4;  background: -webkit-repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);  background: repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);height:45px;\">\n<b>\nInference Pipeline\n</b></h1> ","metadata":{}},{"cell_type":"code","source":"# Constants\nTARGET_SIZE = 960\nTARGET_XY = 960\n\nNUM_WORKERS = 1\nCONFIDENCE_THRESHOLD = 0.60\nMAX_DETECTIONS_PER_TOMO = 3\nBATCH_SIZE = 32\nBOX_SIZE=32\nNMS_IOU=0.1\n\n# Config\ncfg = {\n    'folds': models_config,\n    'con_ths':[0.65,0.60],\n    'nms_box':BOX_SIZE,\n    'nms_iou':NMS_IOU,\n    'batch_size': BATCH_SIZE,\n    'num_workers': 1,\n    'use_amp': True,\n    'preprocess_device': 'cuda',\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T00:30:26.573145Z","iopub.execute_input":"2025-05-22T00:30:26.573466Z","iopub.status.idle":"2025-05-22T00:30:26.587076Z","shell.execute_reply.started":"2025-05-22T00:30:26.573443Z","shell.execute_reply":"2025-05-22T00:30:26.586215Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **》》》 Ultralytics Offline Install**(v8.3.88[2025/03/11 ReleaseVersion])","metadata":{}},{"cell_type":"code","source":"from IPython.display import clear_output\n\n!tar xfvz /kaggle/input/ultralytics-offlineinstall/archive.tar.gz\n!pip install --no-index --find-links=./packages ultralytics\n!rm -rf ./packages\n\nclear_output()","metadata":{"trusted":true,"_kg_hide-input":false,"scrolled":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2025-05-22T00:30:26.58818Z","iopub.execute_input":"2025-05-22T00:30:26.588445Z","iopub.status.idle":"2025-05-22T00:31:31.971827Z","shell.execute_reply.started":"2025-05-22T00:30:26.588425Z","shell.execute_reply":"2025-05-22T00:31:31.970608Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **》》》 Import Libs**","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport time\nimport argparse\nfrom pathlib import Path\n\nfrom tqdm.auto import tqdm\nfrom torch.utils.data import DataLoader, default_collate\nfrom PIL import Image\n\nimport torch\nimport torch.nn as nn\nimport torchvision.transforms.functional as TTF\nimport yaml\n\nimport multiprocessing as mp\nfrom queue import Empty\nfrom ultralytics import YOLO\n\n\nINPUT_PATH = Path('/kaggle/input/byu-locating-bacterial-flagellar-motors-2025')\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T00:31:31.973132Z","iopub.execute_input":"2025-05-22T00:31:31.973555Z","iopub.status.idle":"2025-05-22T00:31:41.597296Z","shell.execute_reply.started":"2025-05-22T00:31:31.973511Z","shell.execute_reply":"2025-05-22T00:31:41.59671Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **》》》 Seed Fix**","metadata":{}},{"cell_type":"code","source":"np.random.seed(42)\ntorch.manual_seed(42)","metadata":{"trusted":true,"_kg_hide-input":false,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2025-05-22T00:31:41.598094Z","iopub.execute_input":"2025-05-22T00:31:41.598506Z","iopub.status.idle":"2025-05-22T00:31:41.609404Z","shell.execute_reply.started":"2025-05-22T00:31:41.598455Z","shell.execute_reply":"2025-05-22T00:31:41.608703Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **》》》 Inference&Submission**","metadata":{}},{"cell_type":"code","source":"import cv2\n\n# Function to get tomograms (train or test)\ndef get_tomos(input_path: Path, data_type: str, *, n=None) -> list[Path]:\n    \"\"\"\n    Args:\n      input_path (Path): Kaggle input directory\n      data_type (str):   'train' or 'test'\n      n (Optional[int]): Use only first n train tomos (not applicable to test)\n    \"\"\"\n    data_path = input_path / data_type\n    tomo_paths = sorted(data_path.glob('*'))\n\n    if (n is not None) and (data_type == 'train'):\n        tomo_paths = tomo_paths[:n]\n    \n    return tomo_paths\n\n# Dataset class to load and preprocess individual slices\nclass Dataset(torch.utils.data.Dataset):\n    \"\"\"\n    Dataset for loading tomogram slices\n    \"\"\"\n    def __init__(self, tomo_path: Path,start=0,step=1):\n        tomo_slices = sorted(tomo_path.glob('*'))\n        self.filenames = sorted([tomo_slices[s] for s in range(start, len(tomo_slices), step)])\n        \n    def __len__(self) -> int:\n        return len(self.filenames)\n\n    def __getitem__(self, i: int) -> dict:\n        filename = self.filenames[i]  # Path to image file\n        filebase = filename.stem\n        assert filebase[:6] == 'slice_'\n        slice_number = int(filebase[6:])  # Extract slice number from filename\n        \n        return {\n            'img': str(filename),\n            'slice_number': slice_number}\n\n    def loader(self, batch_size: int, num_workers: int):\n        \"\"\"DataLoader for the dataset\"\"\"\n        return DataLoader(self, batch_size=batch_size, num_workers=num_workers)\n\ndef accept_confidence_th(pred_conf,conf_model):\n    # if pred_conf < CONFIDENCE_THRESHOLD:\n    #     return False\n    # elif pred_conf>=0.61 and pred_conf<=0.620:\n    #     return False\n    return pred_conf >= conf_model\n\n\n\ndef process_tomogram(tomo_path: Path, models, device: str, cfg):\n    \"\"\"\n    Process a tomogram, extract slices, run multiple models with weighted fusion and NMS\n    \"\"\"\n    batch_size = cfg['batch_size']\n    num_workers = cfg['num_workers']\n    use_amp = cfg['use_amp']\n    preprocess_device = cfg['preprocess_device']\n    assert preprocess_device in ['cuda', 'cpu']\n\n    all_model_detections = []\n    model_weights = []\n\n    for confg, model in models:\n        dataset = Dataset(tomo_path, start=confg['start'], step=confg['step'])\n        dataloader = dataset.loader(batch_size=batch_size, num_workers=num_workers)\n\n        model_detections = []\n\n        for batch in dataloader:\n            img = batch['img']\n            with torch.no_grad():\n                with torch.amp.autocast(device_type='cuda', enabled=use_amp, dtype=torch.float16):\n                    results = model.predict(img, save=False, imgsz=confg['imgsz'], augment=True, verbose=False)\n\n            for j, result in enumerate(results):\n                if len(result.boxes) > 0:\n                    boxes = result.boxes\n                    for box_idx, confidence in enumerate(boxes.conf):\n                        if accept_confidence_th(confidence, confg['conf']):\n                            x1, y1, x2, y2 = boxes.xyxy[box_idx].cpu().numpy()\n                            z = batch['slice_number'][j]\n                            x_center = (x1 + x2) / 2\n                            y_center = (y1 + y2) / 2\n                            model_detections.append({\n                                'z': round(z.item()),\n                                'y': round(y_center),\n                                'x': round(x_center),\n                                'confidence': float(confidence)\n                            })\n\n        # Per-model NMS before fusion\n        pruned = perform_3d_nms(model_detections, iou_threshold=cfg['nms_iou'], box_size=cfg['nms_box'])\n        all_model_detections.append(pruned)\n        model_weights.append(confg.get(\"weight\", 1.0))  \n\n    # Perform weighted fusion NMS across models\n    final_detections = perform_weighted_fusion_3d_nms(\n        model_detections=all_model_detections,\n        model_weights=model_weights,\n        iou_threshold=cfg['nms_iou'],\n        box_size=cfg['nms_box'],\n        top_k=1\n    )\n\n    tomo_id = tomo_path.stem\n    if not final_detections:\n        return {\n            'tomo_id': tomo_id,\n            'Motor axis 0': -1,\n            'Motor axis 1': -1,\n            'Motor axis 2': -1\n        }\n\n    best_detection = final_detections[0]\n    return {\n        'tomo_id': tomo_id,\n        'Motor axis 0': round(best_detection['z']),\n        'Motor axis 1': round(best_detection['y']),\n        'Motor axis 2': round(best_detection['x'])\n    }\n\n\n\ndef perform_weighted_fusion_3d_nms(model_detections, model_weights, iou_threshold=0.6, box_size=32, top_k=1):\n    \"\"\"\n    Fuse detections from multiple models using confidence-weighted scores and apply 3D NMS.\n    \"\"\"\n    assert len(model_detections) == len(model_weights)\n    \n    weighted_detections = []\n    for detections, weight in zip(model_detections, model_weights):\n        for det in detections:\n            weighted_detections.append({\n                'z': det['z'],\n                'y': det['y'],\n                'x': det['x'],\n                'confidence': det['confidence'] * weight\n            })\n\n    # Sort by weighted confidence\n    weighted_detections = sorted(weighted_detections, key=lambda d: d['confidence'], reverse=True)\n    final_detections = []\n\n    def distance_3d(d1, d2):\n        return np.sqrt((d1['z'] - d2['z']) ** 2 +\n                       (d1['y'] - d2['y']) ** 2 +\n                       (d1['x'] - d2['x']) ** 2)\n\n    distance_threshold = box_size * iou_threshold\n\n    while weighted_detections:\n        best = weighted_detections.pop(0)\n        final_detections.append(best)\n        weighted_detections = [d for d in weighted_detections if distance_3d(d, best) > distance_threshold]\n\n    return final_detections[:top_k]\n\n\ndef perform_3d_nms(detections, iou_threshold,box_size):\n    \"\"\"\n    Perform 3D Non-Maximum Suppression on detections to merge nearby motors\n    \"\"\"\n    if not detections:\n        return []\n    \n    # Sort by confidence (highest first)\n    detections = sorted(detections, key=lambda x: x['confidence'], reverse=True)\n    \n    # List to store final detections after NMS\n    final_detections = []\n    \n    # Define 3D distance function\n    def distance_3d(d1, d2):\n        return np.sqrt((d1['z'] - d2['z'])**2 + \n                       (d1['y'] - d2['y'])**2 + \n                       (d1['x'] - d2['x'])**2)\n    \n    # Maximum distance threshold (based on box size and slice gap)\n      # Same as annotation box size\n    distance_threshold = box_size * iou_threshold\n    \n    # Process each detection\n    while detections:\n        # Take the detection with highest confidence\n        best_detection = detections.pop(0)\n        final_detections.append(best_detection)\n        \n        # Filter out detections that are too close to the best detection\n        detections = [d for d in detections if distance_3d(d, best_detection) > distance_threshold]\n    \n    return final_detections","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T00:31:41.622218Z","iopub.execute_input":"2025-05-22T00:31:41.622439Z","iopub.status.idle":"2025-05-22T00:31:41.652078Z","shell.execute_reply.started":"2025-05-22T00:31:41.62242Z","shell.execute_reply":"2025-05-22T00:31:41.651209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_submission(preds: list, th: float, ofilename: str) -> pd.DataFrame: \n    \"\"\"\n    Args:\n      preds (list[dict]): predictions\n      th (float): threshold between no or one moter\n      ofilename (str): submission.csv\n    \"\"\"\n    rows = []\n    count_positive = 0\n    for pred in preds:\n        if pred['Motor axis 0'] < th:\n            zyx = (-1, -1, -1)\n        else:\n            count_positive += 1\n            zyx = (pred['Motor axis 0'],pred['Motor axis 1'],pred['Motor axis 2'])\n\n        row = {'tomo_id': pred['tomo_id'],\n               'Motor axis 0': zyx[0],\n               'Motor axis 1': zyx[1],\n               'Motor axis 2': zyx[2]}\n        rows.append(row)\n\n    submit = pd.DataFrame(rows)\n    submit.to_csv(ofilename, float_format='%.8e', index=False)\n\n    print('Submit %s: %d positives / %d tomo_ids' % (ofilename, count_positive, len(rows)))\n    return submit","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T00:31:41.653112Z","iopub.execute_input":"2025-05-22T00:31:41.653411Z","iopub.status.idle":"2025-05-22T00:31:41.667243Z","shell.execute_reply.started":"2025-05-22T00:31:41.653386Z","shell.execute_reply":"2025-05-22T00:31:41.666538Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_fn(process_id: int,\n               tomo_queue,\n               pred_queue,\n               cfg: dict):\n    \"\"\"\n    Prediction process for each GPU\n    Args:\n      process_id (int): 0 or 1 for 2 GPUs\n      tomo_queue (): tomo_ids\n      pred_queue (): output predictions\n      cfg (dict): config\n    \"\"\"\n    \n    device = torch.device('cuda:%d' % process_id)\n    # model_path = cfg['model_path']\n    folds = cfg['folds']  \n    models=[]\n    for fold in folds:\n        model_path=fold['path']\n        model = YOLO(model_path)  # trained YOLO model\n        model.to(device)\n        model.eval()\n        models.append((fold,model))\n        if process_id == 0:\n            print('Load model', model_path)\n        \n    # Loop over tomograms\n    while not tomo_queue.empty():\n        try:\n            tomo_path = tomo_queue.get(timeout=1)\n            pred=process_tomogram(tomo_path, models, device,cfg)\n            pred_queue.put(pred)\n        except Empty:\n            break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T00:31:41.667978Z","iopub.execute_input":"2025-05-22T00:31:41.668214Z","iopub.status.idle":"2025-05-22T00:31:41.682842Z","shell.execute_reply.started":"2025-05-22T00:31:41.668192Z","shell.execute_reply":"2025-05-22T00:31:41.682114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#\n# Main\n#\n\ntb = time.time()\n\ntomo_paths = get_tomos(INPUT_PATH, 'test')\n\nif len(tomo_paths) == 3:\n    # Some experiment when test is dummy (optional)\n    # tomo_paths = get_tomos(INPUT_PATH, 'train', n=10)\n    pass\n\nprint('Data %d' % len(tomo_paths))\n\nmanager = mp.Manager()\ntomo_queue = manager.Queue()\npred_queue = manager.Queue()\n\nfor tomo_path in tomo_paths:\n    tomo_queue.put(tomo_path)\n    \ntime.sleep(1)\nassert not tomo_queue.empty()\n\n#\n# Launch process\n#\n\nnum_processes = 2\ntb = time.time()\n\nworkers = [mp.Process(target=process_fn,\n                      args=(i, tomo_queue, pred_queue, cfg))\n           for i in range(num_processes)]\n\nfor w in workers:\n    w.start()\n\nfor w in workers:\n    w.join()\n\ndt = time.time() - tb\nprint('%.2f sec for %d tomos' % (dt, len(tomo_paths)))\n\n# queue to list\npreds = []\ntry:\n    while not pred_queue.empty():\n        preds.append(pred_queue.get(timeout=1))\nexcept Empty:\n    pass\n\nassert len(preds) == len(tomo_paths)\n\nth = 0.30\nofilename = 'submission.csv'\nsubmission=create_submission(preds, th, ofilename)\nsubmission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T00:31:41.68381Z","iopub.execute_input":"2025-05-22T00:31:41.684035Z","iopub.status.idle":"2025-05-22T00:33:49.785044Z","shell.execute_reply.started":"2025-05-22T00:31:41.684015Z","shell.execute_reply":"2025-05-22T00:33:49.783587Z"}},"outputs":[],"execution_count":null}]}