{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91249,"databundleVersionId":11294684,"sourceType":"competition"},{"sourceId":11727138,"sourceType":"datasetVersion","datasetId":7361383},{"sourceId":224916709,"sourceType":"kernelVersion"},{"sourceId":240015959,"sourceType":"kernelVersion"},{"sourceId":423980,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":290384,"modelId":311103},{"sourceId":424642,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":290384,"modelId":311103}],"dockerImageVersionId":30919,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# BYU Locating Flagellar Motors\n\n## Submission Generation Notebook\n\nThis is the fourth and final notebook in a series for the BYU Locating Bacterial Flagellar Motors 2025 Kaggle challenge. This notebook creates predictions on test data and generates the competition submission file.\n\n### Notebook Series:\n1. **[Parse Data](https://www.kaggle.com/code/andrewjdarley/parse-data)**: Extracting and preparing 2D slices containing motors to make a YOLO dataset\n2. **[Visualize Data](https://www.kaggle.com/code/andrewjdarley/visualize-data)**: Exploratory data analysis and visualization of annotated motor locations\n3. **[Train YOLO](https://www.kaggle.com/code/andrewjdarley/train-yolo)**: Fine tuning an YOLOv8 object detection model on the prepared dataset\n4. **Submission Notebook (Current)**: Running inference and generating submission files \n\n## Important: Offline Execution\nThis notebook is designed to run in an offline environment. The Ultralytics YOLOv8 package has been installed using the offline installation method from [this reference notebook](https://www.kaggle.com/code/itsuki9180/ultralytics-for-offline-install). This implementation was brilliant. I use my own copy as input that works effectively the same as the original.\n\n## About this Notebook\n\nThis submission notebook implements an optimized inference pipeline that:\n\n1. **Model Loading**: Loads the best trained YOLOv8 weights from the training notebook\n2. **GPU Optimization**: Configures CUDA optimizations, half-precision inference, and memory management\n3. **Parallel Processing**: Uses CUDA streams and batch processing for efficient GPU utilization\n4. **3D Detection**: Processes each slice to locate motors\n5. **Non-Maximum Suppression**: Applies 3D NMS to cluster and merge detections across slices\n6. **Submission Generation**: Creates the final CSV file with predicted motor coordinates\n\nThe code includes advanced optimizations like dynamic batch sizing based on available GPU memory, preloading batches while processing the current batch, and GPU profiling to monitor performance. The CONCENTRATION parameter can be adjusted to trade off between processing speed and detection accuracy. The only reason you'd ever modify CONCENTRATION is just to verify submission capability since full submission takes a few hours.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"!tar xfvz /kaggle/input/ultralytics-for-offline-install/archive.tar.gz\n!pip install --no-index --find-links=./packages ultralytics\n!rm -rf ./packages","metadata":{"trusted":true,"_kg_hide-input":false,"execution":{"iopub.status.busy":"2025-05-27T13:12:22.052567Z","iopub.execute_input":"2025-05-27T13:12:22.052875Z","iopub.status.idle":"2025-05-27T13:13:13.162786Z","shell.execute_reply.started":"2025-05-27T13:12:22.052841Z","shell.execute_reply":"2025-05-27T13:13:13.161573Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"!cd /kaggle/input/install-u-offline/ultralytics\n!pip install ultralytics\n#--no-index --find-links=./ ultralytics","metadata":{"execution":{"iopub.status.busy":"2025-05-08T13:02:29.716007Z","iopub.execute_input":"2025-05-08T13:02:29.716299Z","iopub.status.idle":"2025-05-08T13:04:09.474148Z","shell.execute_reply.started":"2025-05-08T13:02:29.716269Z","shell.execute_reply":"2025-05-08T13:04:09.473370Z"}}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport torch\nimport cv2\nfrom tqdm.notebook import tqdm\nfrom ultralytics import YOLO\nimport threading\nimport time\nfrom contextlib import nullcontext\nfrom concurrent.futures import ThreadPoolExecutor\nfrom queue import Queue\nfrom threading import Lock\n\n# Set random seed for reproducibility\nnp.random.seed(42)\ntorch.manual_seed(42)\n\n# Define paths\ndata_path = \"/kaggle/input/byu-locating-bacterial-flagellar-motors-2025/\"\ntest_dir = os.path.join(data_path, \"test\")\nsubmission_path = \"/kaggle/working/submission.csv\"\n\n# Model path\nmodel_path = \"/kaggle/input/byu-2025/pytorch/default/22/yolov5xud.pt\"\n\n# Detection parameters\nCONFIDENCE_THRESHOLD = 0.2\nMAX_DETECTIONS_PER_TOMO = 1\nNMS_IOU_THRESHOLD = 0.4\nCONCENTRATION = 20\nTARGET_SIZE = 1280  # Model's expected input size divisible by 32\n\n# GPU settings\ndevice = 'cuda' if torch.cuda.is_available() else 'cpu'\nnum_gpus = torch.cuda.device_count() if torch.cuda.is_available() else 0\nprint(f\"Found {num_gpus} GPU(s) available\")\n\n# Dynamic batch sizing\nBATCH_SIZE = 8  # Fixed smaller batch size for varying input sizes\nprint(f\"Using batch size: {BATCH_SIZE}\")\n\nclass ImageLoader:\n    def __init__(self, max_workers=4):\n        self.executor = ThreadPoolExecutor(max_workers=max_workers)\n        self.lock = Lock()\n        \n    def load_and_resize(self, path):\n        \"\"\"Load image and resize to target size maintaining aspect ratio\"\"\"\n        # Load as RGB\n        img = cv2.imread(path)\n        if img is None:\n            img = np.array(Image.open(path).convert('RGB'))\n        else:\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        \n        # Resize maintaining aspect ratio\n        h, w = img.shape[:2]\n        scale = min(TARGET_SIZE / h, TARGET_SIZE / w)\n        new_h, new_w = int(h * scale), int(w * scale)\n        img = cv2.resize(img, (new_w, new_h))\n        \n        # Pad to target size\n        top = (TARGET_SIZE - new_h) // 2\n        bottom = TARGET_SIZE - new_h - top\n        left = (TARGET_SIZE - new_w) // 2\n        right = TARGET_SIZE - new_w - left\n        img = cv2.copyMakeBorder(img, top, bottom, left, right, \n                                cv2.BORDER_CONSTANT, value=(114, 114, 114))\n        \n        # Normalize\n        img = img.astype(np.float32) / 255.0\n        return img, (scale, (left, top))  # Return scaling and padding info\n\ndef process_tomogram(tomo_id, model, index=0, total=1, gpu_id=0):\n    \"\"\"Process tomogram with proper image resizing and scaling\"\"\"\n    torch.cuda.set_device(gpu_id)\n    print(f\"Processing {tomo_id} ({index}/{total}) on GPU {gpu_id}\")\n    \n    tomo_dir = os.path.join(test_dir, tomo_id)\n    slice_files = sorted([f for f in os.listdir(tomo_dir) if f.endswith('.jpg')])\n    \n    # Apply CONCENTRATION\n    num_slices = len(slice_files)\n    step = max(1, int(1 / CONCENTRATION * num_slices))\n    selected_indices = range(0, num_slices, step)\n    slice_files = [slice_files[i] for i in selected_indices if i < num_slices]\n    \n    print(f\"Processing {len(slice_files)}/{num_slices} slices\")\n    \n    all_detections = []\n    loader = ImageLoader()\n    \n    # Process slices one by one due to varying sizes\n    for slice_file in slice_files:\n        slice_path = os.path.join(tomo_dir, slice_file)\n        slice_num = int(slice_file.split('_')[1].split('.')[0])\n        \n        try:\n            # Load and preprocess image\n            img, (scale, (left_pad, top_pad)) = loader.load_and_resize(slice_path)\n            \n            # Convert to tensor\n            img_tensor = torch.from_numpy(img).permute(2, 0, 1).unsqueeze(0).to(device)\n            \n            # Inference\n            with torch.no_grad():\n                results = model(img_tensor)\n            \n            # Process results\n            for result in results:\n                if len(result.boxes) > 0:\n                    boxes = result.boxes\n                    for box_idx, confidence in enumerate(boxes.conf):\n                        if confidence >= CONFIDENCE_THRESHOLD:\n                            # Adjust coordinates back to original image space\n                            x1, y1, x2, y2 = boxes.xyxy[box_idx].cpu().numpy()\n                            \n                            # Remove padding and scale back\n                            x1 = (x1 - left_pad) / scale\n                            y1 = (y1 - top_pad) / scale\n                            x2 = (x2 - left_pad) / scale\n                            y2 = (y2 - top_pad) / scale\n                            \n                            # Calculate center\n                            x_center = (x1 + x2) / 2\n                            y_center = (y1 + y2) / 2\n                            \n                            all_detections.append({\n                                'tomo_id': tomo_id,\n                                'z': slice_num,\n                                'y': round(y_center),\n                                'x': round(x_center),\n                                'confidence': float(confidence)\n                            })\n        except Exception as e:\n            print(f\"Error processing {slice_file}: {e}\")\n            continue\n    \n    # If no detections, return default values\n    if not all_detections:\n        return {\n            'tomo_id': tomo_id,\n            'Motor axis 0': -1,\n            'Motor axis 1': -1,\n            'Motor axis 2': -1\n        }\n    \n    # Find best detection\n    best_detection = max(all_detections, key=lambda x: x['confidence'])\n    return {\n        'tomo_id': tomo_id,\n        'Motor axis 0': best_detection['z'],\n        'Motor axis 1': best_detection['y'],\n        'Motor axis 2': best_detection['x']\n    }\n\ndef initialize_models():\n    \"\"\"Initialize models with proper settings\"\"\"\n    models = []\n    if num_gpus > 0:\n        for i in range(num_gpus):\n            print(f\"Initializing model on GPU {i}\")\n            torch.cuda.set_device(i)\n            model = YOLO(model_path)\n            model.to(f'cuda:{i}')\n            model.fuse()\n            model.model.eval()\n            \n            # Warmup\n            dummy_input = torch.randn(1, 3, TARGET_SIZE, TARGET_SIZE, \n                                    device=f'cuda:{i}', dtype=torch.float32)\n            with torch.no_grad():\n                _ = model(dummy_input)\n            \n            models.append(model)\n            torch.cuda.empty_cache()\n    else:\n        print(\"Initializing model on CPU\")\n        model = YOLO(model_path)\n        model.to('cpu')\n        model.model.eval()\n        models.append(model)\n    \n    return models\n\ndef generate_submission():\n    \"\"\"Generate submission with error handling\"\"\"\n    test_tomos = sorted([d for d in os.listdir(test_dir) if os.path.isdir(os.path.join(test_dir, d))])\n    total_tomos = len(test_tomos)\n    print(f\"Found {total_tomos} tomograms\")\n    \n    models = initialize_models()\n    results = []\n    \n    with ThreadPoolExecutor(max_workers=num_gpus if num_gpus > 0 else 1) as executor:\n        futures = []\n        for idx, tomo_id in enumerate(test_tomos, 1):\n            gpu_id = (idx-1) % num_gpus if num_gpus > 0 else 0\n            futures.append(executor.submit(\n                process_tomogram, tomo_id, models[gpu_id], idx, total_tomos, gpu_id\n            ))\n        \n        for future in futures:\n            try:\n                results.append(future.result())\n            except Exception as e:\n                print(f\"Error processing tomogram: {e}\")\n                results.append({\n                    'tomo_id': tomo_id,\n                    'Motor axis 0': -1,\n                    'Motor axis 1': -1,\n                    'Motor axis 2': -1\n                })\n    \n    # Ensure consistent output format\n    submission_data = []\n    for res in results:\n        submission_data.append({\n            'tomo_id': res['tomo_id'],\n            'Motor axis 0': res.get('Motor axis 0', -1),\n            'Motor axis 1': res.get('Motor axis 1', -1),\n            'Motor axis 2': res.get('Motor axis 2', -1)\n        })\n    \n    submission_df = pd.DataFrame(submission_data)\n    submission_df.to_csv(submission_path, index=False)\n    print(f\"Submission saved to {submission_path}\")\n    return submission_df\n\nif __name__ == \"__main__\":\n    start_time = time.time()\n    submission = generate_submission()\n    elapsed = time.time() - start_time\n    print(f\"Total time: {elapsed:.2f}s ({elapsed/60:.2f}min)\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T13:13:52.332801Z","iopub.execute_input":"2025-05-27T13:13:52.333185Z","iopub.status.idle":"2025-05-27T13:14:06.823288Z","shell.execute_reply.started":"2025-05-27T13:13:52.333156Z","shell.execute_reply":"2025-05-27T13:14:06.822508Z"}},"outputs":[],"execution_count":null}]}