{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":91249,"databundleVersionId":11294684,"sourceType":"competition"},{"sourceId":323401,"sourceType":"modelInstanceVersion","modelInstanceId":272449,"modelId":293428},{"sourceId":326838,"sourceType":"modelInstanceVersion","modelInstanceId":274391,"modelId":295285}],"dockerImageVersionId":30919,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from ultralytics import YOLO","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T02:27:52.580412Z","iopub.execute_input":"2025-04-08T02:27:52.580773Z","iopub.status.idle":"2025-04-08T02:27:58.592806Z","shell.execute_reply.started":"2025-04-08T02:27:52.580725Z","shell.execute_reply":"2025-04-08T02:27:58.592155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport torch\nimport cv2\nfrom tqdm.notebook import tqdm\nfrom ultralytics import YOLO\nimport threading\nimport time\nfrom contextlib import nullcontext\nfrom concurrent.futures import ThreadPoolExecutor\nimport torch.nn as nn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T02:27:58.593908Z","iopub.execute_input":"2025-04-08T02:27:58.594369Z","iopub.status.idle":"2025-04-08T02:27:59.082563Z","shell.execute_reply.started":"2025-04-08T02:27:58.594337Z","shell.execute_reply":"2025-04-08T02:27:59.081907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Model path - adjust if your best model is saved in a different location\nmodel_path = \"/kaggle/input/y18/pytorch/default/1/best.pt\"\n\n# Define paths\ndata_path = \"/kaggle/input/byu-locating-bacterial-flagellar-motors-2025/\"\ntest_dir = os.path.join(data_path, \"test\")\nsubmission_path = \"/kaggle/working/submission.csv\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T02:27:59.084214Z","iopub.execute_input":"2025-04-08T02:27:59.084534Z","iopub.status.idle":"2025-04-08T02:27:59.087992Z","shell.execute_reply.started":"2025-04-08T02:27:59.084514Z","shell.execute_reply":"2025-04-08T02:27:59.087264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Detection parameters\nCONFIDENCE_THRESHOLD = 0.59  # Lower threshold to catch more potential motors\nMAX_DETECTIONS_PER_TOMO = 3  # Keep track of top N detections per tomogram\nNMS_IOU_THRESHOLD = 0.1  # Non-maximum suppression threshold for 3D clustering\nCONCENTRATION = 4 # ONLY PROCESS 1/10 slices for fast submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T02:27:59.089264Z","iopub.execute_input":"2025-04-08T02:27:59.089457Z","iopub.status.idle":"2025-04-08T02:27:59.105214Z","shell.execute_reply.started":"2025-04-08T02:27:59.089441Z","shell.execute_reply":"2025-04-08T02:27:59.104577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class GPUProfiler:\n    def __init__(self, name):\n        self.name = name\n        self.start_time = None\n        \n    def __enter__(self):\n        if torch.cuda.is_available():\n            torch.cuda.synchronize()\n        self.start_time = time.time()\n        return self\n        \n    def __exit__(self, *args):\n        if torch.cuda.is_available():\n            torch.cuda.synchronize()\n        elapsed = time.time() - self.start_time\n        print(f\"[PROFILE] {self.name}: {elapsed:.3f}s\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T02:27:59.105931Z","iopub.execute_input":"2025-04-08T02:27:59.106166Z","iopub.status.idle":"2025-04-08T02:27:59.119045Z","shell.execute_reply.started":"2025-04-08T02:27:59.106147Z","shell.execute_reply":"2025-04-08T02:27:59.118401Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check GPU availability and set up optimizations\ndevice = 'cuda:0' if torch.cuda.is_available() else 'cpu'\nBATCH_SIZE = 16  # Default batch size, will be adjusted dynamically if GPU available\nbox_size = 24  # Same as annotation box size","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T02:27:59.119784Z","iopub.execute_input":"2025-04-08T02:27:59.120082Z","iopub.status.idle":"2025-04-08T02:27:59.196027Z","shell.execute_reply.started":"2025-04-08T02:27:59.120039Z","shell.execute_reply":"2025-04-08T02:27:59.195327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if device.startswith('cuda'):\n    # Set CUDA optimization flags\n    torch.backends.cudnn.benchmark = True\n    torch.backends.cudnn.deterministic = False\n    torch.backends.cuda.matmul.allow_tf32 = True  # Allow TF32 on Ampere GPUs\n    torch.backends.cudnn.allow_tf32 = True\n    \n    # Print GPU info\n    gpu_name = torch.cuda.get_device_name(0)\n    gpu_mem = torch.cuda.get_device_properties(0).total_memory / 1e9  # Convert to GB\n    print(f\"Using GPU: {gpu_name} with {gpu_mem:.2f} GB memory\")\n    \n    # Get available GPU memory and set batch size accordingly\n    free_mem = gpu_mem - torch.cuda.memory_allocated(0) / 1e9\n    BATCH_SIZE = max(8, min(32, int(free_mem * 4)))  # 4 images per GB as rough estimate\n    print(f\"Dynamic batch size set to {BATCH_SIZE} based on {free_mem:.2f}GB free memory\")\nelse:\n    print(\"GPU not available, using CPU\")\n    BATCH_SIZE = 4  # Reduce batch size for CPU","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T02:27:59.196813Z","iopub.execute_input":"2025-04-08T02:27:59.197088Z","iopub.status.idle":"2025-04-08T02:27:59.234329Z","shell.execute_reply.started":"2025-04-08T02:27:59.197046Z","shell.execute_reply":"2025-04-08T02:27:59.233484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch.nn as nn\nfrom torch.ao.quantization import fuse_modules","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T02:27:59.236595Z","iopub.execute_input":"2025-04-08T02:27:59.236822Z","iopub.status.idle":"2025-04-08T02:27:59.240128Z","shell.execute_reply.started":"2025-04-08T02:27:59.236803Z","shell.execute_reply":"2025-04-08T02:27:59.239275Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class DnCNN(nn.Module):\n    def __init__(self, channels=1, num_of_layers=17):  # <-- 반드시 17\n        super(DnCNN, self).__init__()\n        kernel_size = 3\n        padding = 1\n        features = 64\n        layers = []\n\n        layers.append(nn.Conv2d(channels, features, kernel_size, padding=padding, bias=False))\n        layers.append(nn.ReLU(inplace=True))\n\n        for _ in range(num_of_layers - 2):  # 15번 반복\n            layers.append(nn.Conv2d(features, features, kernel_size, padding=padding, bias=False))\n            layers.append(nn.BatchNorm2d(features))\n            layers.append(nn.ReLU(inplace=True))\n\n        layers.append(nn.Conv2d(features, channels, kernel_size, padding=padding, bias=False))  # 마지막 Conv\n\n        self.dncnn = nn.Sequential(*layers)\n\n    def forward(self, x):\n        out = self.dncnn(x)\n        return x - out\n\ndef fuse_dncnn(model):\n    fused_list = []\n    # DnCNN 구조는 [Conv, BN, ReLU] 반복이므로 3개 단위로 묶어줌\n    for i in range(2, len(model.dncnn) - 1, 3):\n        # Conv2d, BatchNorm2d, ReLU 묶기\n        fused_list.append([str(i), str(i + 1), str(i + 2)])\n    \n    # 첫 Conv + ReLU는 따로 처리 (BN 없음)\n    fuse_modules(model.dncnn, [['0', '1']], inplace=True)\n    \n    # 나머지 Conv+BN+ReLU 묶기\n    fuse_modules(model.dncnn, fused_list, inplace=True)\n    \n    return model\n\nfrom collections import OrderedDict\n\n# 모델 정의 및 로딩\nmodel = DnCNN(num_of_layers=17)\nstate_dict = torch.load('/kaggle/input/cnn2/pytorch/default/1/net.pth', map_location='cuda')\n\n# \"module.\" 접두사 제거\nnew_state_dict = OrderedDict((k.replace('module.', ''), v) for k, v in state_dict.items())\n\n# 모델에 state_dict 로드\nmodel.load_state_dict(new_state_dict)\nmodel.eval().cuda()\n\n# ✅ DnCNN 디노이즈 함수\ndef denoise_dncnn(img_tensor_list):\n    batch = []\n    for t in img_tensor_list:\n        if t.max() > 1:  # 0~255면 정규화\n            t = t.float() / 255.0\n        if t.ndim == 3 and t.shape[0] != 1:  # RGB이면 흑백으로 변환 필요\n            t = t.mean(dim=0, keepdim=True)  # 임시로 평균 처리\n        elif t.ndim == 2:  # (H, W)면 채널 차원 추가\n            t = t.unsqueeze(0)\n        batch.append(t)\n\n    batch_tensor = torch.stack(batch, dim=0).cuda()  # (N, 1, H, W)\n    \n    img_tensor = batch_tensor.cuda()\n\n    with torch.no_grad():\n        out = model(img_tensor)\n\n    result = out.squeeze().clamp(0, 1).cpu().numpy() \n\n    # 명시적으로 GPU 메모리 해제\n    del img_tensor, out\n    torch.cuda.empty_cache()\n\n    return result.astype(np.uint8)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T02:27:59.241150Z","iopub.execute_input":"2025-04-08T02:27:59.241429Z","iopub.status.idle":"2025-04-08T02:27:59.509285Z","shell.execute_reply.started":"2025-04-08T02:27:59.241409Z","shell.execute_reply":"2025-04-08T02:27:59.508571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preload_image_batch(file_paths):\n    \"\"\"Preload a batch of images to CPU memory, apply DnCNN, and return results\"\"\"\n    images = []\n    for path in file_paths:\n        img = cv2.imread(path, cv2.IMREAD_GRAYSCALE)  # numpy, shape: (H, W)\n        if img is None:\n            try:\n                img = np.array(Image.open(path).convert(\"L\"))  # fallback (PIL)\n            except:\n                continue  # skip if still fails\n        try:\n            images.append(img)\n            \n        except Exception as e:\n            print(f\"❌ 에러 발생: {path} → {e}\")\n            continue\n    return images","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T02:27:59.510029Z","iopub.execute_input":"2025-04-08T02:27:59.510277Z","iopub.status.idle":"2025-04-08T02:27:59.514889Z","shell.execute_reply.started":"2025-04-08T02:27:59.510257Z","shell.execute_reply":"2025-04-08T02:27:59.514140Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_tomogram(tomo_id, model, index=0, total=1):\n    \"\"\"\n    Process a single tomogram and return the most confident motor detection\n    \"\"\"\n    print(f\"Processing tomogram {tomo_id} ({index}/{total})\")\n    \n    # Get all slice files for this tomogram\n    tomo_dir = os.path.join(test_dir, tomo_id)\n    slice_files = sorted([f for f in os.listdir(tomo_dir) if f.endswith('.jpg')])\n    \n    # Apply CONCENTRATION to reduce the number of slices processed\n    # This will process approximately CONCENTRATION fraction of all slices\n    selected_indices = np.linspace(0, len(slice_files)-1, int(len(slice_files) * CONCENTRATION))\n    selected_indices = np.round(selected_indices).astype(int)\n    slice_files = [slice_files[i] for i in selected_indices]\n    \n    print(f\"Processing {len(slice_files)} out of {len(os.listdir(tomo_dir))} slices based on CONCENTRATION={CONCENTRATION}\")\n    \n    # Create a list to store all detections\n    all_detections = []\n    \n    # Create CUDA streams for parallel processing if using GPU\n    if device.startswith('cuda'):\n        streams = [torch.cuda.Stream() for _ in range(min(4, BATCH_SIZE))]\n    else:\n        streams = [None]\n    \n    # Variables for preloading\n    next_batch_thread = None\n    next_batch_images = None\n    \n    # Process slices in batches\n    for batch_start in range(0, len(slice_files), BATCH_SIZE):\n        # Wait for previous preload thread if it exists\n        if next_batch_thread is not None:\n            next_batch_thread.join()\n            next_batch_images = None\n            \n        batch_end = min(batch_start + BATCH_SIZE, len(slice_files))\n        batch_files = slice_files[batch_start:batch_end]\n        \n        # Start preloading next batch\n        next_batch_start = batch_end\n        next_batch_end = min(next_batch_start + BATCH_SIZE, len(slice_files))\n        next_batch_files = slice_files[next_batch_start:next_batch_end] if next_batch_start < len(slice_files) else []\n        if next_batch_files:\n            next_batch_paths = [os.path.join(tomo_dir, f) for f in next_batch_files]\n            next_batch_thread = threading.Thread(target=preload_image_batch, args=(next_batch_paths,))\n            next_batch_thread.start()\n        else:\n            next_batch_thread = None\n        \n        # Split batch across streams for parallel processing\n        sub_batches = np.array_split(batch_files, len(streams))\n        sub_batch_results = []\n        \n        for i, sub_batch in enumerate(sub_batches):\n            if len(sub_batch) == 0:\n                continue\n                \n            stream = streams[i % len(streams)]\n            with torch.cuda.stream(stream) if stream and device.startswith('cuda') else nullcontext():\n                # Process sub-batch\n                sub_batch_paths = [os.path.join(tomo_dir, slice_file) for slice_file in sub_batch]\n                sub_batch_slice_nums = [int(slice_file.split('_')[1].split('.')[0]) for slice_file in sub_batch]\n\n                # Run inference with profiling\n                with GPUProfiler(f\"Inference batch {i+1}/{len(sub_batches)}\"):\n                    sub_results = model(sub_batch_paths, verbose=False)\n                \n                # Process each result in this sub-batch\n                for j, result in enumerate(sub_results):\n                    if len(result.boxes) > 0:\n                        boxes = result.boxes\n                        for box_idx, confidence in enumerate(boxes.conf):\n                            if confidence >= CONFIDENCE_THRESHOLD:\n                                # Get bounding box coordinates\n                                x1, y1, x2, y2 = boxes.xyxy[box_idx].cpu().numpy()\n                                \n                                # Calculate center coordinates\n                                x_center = (x1 + x2) / 2\n                                y_center = (y1 + y2) / 2\n                                \n                                # Store detection with 3D coordinates\n                                all_detections.append({\n                                    'z': round(sub_batch_slice_nums[j]),\n                                    'y': round(y_center),\n                                    'x': round(x_center),\n                                    'confidence': float(confidence)\n                                })\n        \n        # Synchronize streams\n        if device.startswith('cuda'):\n            torch.cuda.synchronize()\n    \n    # Clean up thread if still running\n    if next_batch_thread is not None:\n        next_batch_thread.join()\n    \n    # 3D Non-Maximum Suppression to merge nearby detections across slices\n    final_detections = perform_3d_nms(all_detections, NMS_IOU_THRESHOLD)\n    \n    # Sort detections by confidence (highest first)\n    final_detections.sort(key=lambda x: x['confidence'], reverse=True)\n    \n    # If there are no detections, return NA values\n    if not final_detections:\n        return {\n            'tomo_id': tomo_id,\n            'Motor axis 0': -1,\n            'Motor axis 1': -1,\n            'Motor axis 2': -1\n        }\n    \n    # Take the detection with highest confidence\n    best_detection = final_detections[0]\n    \n    # Return result with integer coordinates\n    return {\n        'tomo_id': tomo_id,\n        'Motor axis 0': round(best_detection['z']),\n        'Motor axis 1': round(best_detection['y']),\n        'Motor axis 2': round(best_detection['x'])\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T02:27:59.515593Z","iopub.execute_input":"2025-04-08T02:27:59.515792Z","iopub.status.idle":"2025-04-08T02:27:59.535754Z","shell.execute_reply.started":"2025-04-08T02:27:59.515775Z","shell.execute_reply":"2025-04-08T02:27:59.534931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define 3D distance function\ndef distance_3d(d1, d2):\n    return np.sqrt((d1['z'] - d2['z'])**2 + \n                       (d1['y'] - d2['y'])**2 + \n                       (d1['x'] - d2['x'])**2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T02:27:59.536606Z","iopub.execute_input":"2025-04-08T02:27:59.536884Z","iopub.status.idle":"2025-04-08T02:27:59.553408Z","shell.execute_reply.started":"2025-04-08T02:27:59.536857Z","shell.execute_reply":"2025-04-08T02:27:59.552664Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def perform_3d_nms(detections, iou_threshold):\n    \"\"\"\n    Perform 3D Non-Maximum Suppression on detections to merge nearby motors\n    \"\"\"\n    if not detections:\n        return []\n    \n    # Sort by confidence (highest first)\n    detections = sorted(detections, key=lambda x: x['confidence'], reverse=True)\n    \n    # List to store final detections after NMS\n    final_detections = []\n    \n    # Maximum distance threshold (based on box size and slice gap)\n    distance_threshold = box_size * iou_threshold\n    \n    # Process each detection\n    while detections:\n        # Take the detection with highest confidence\n        best_detection = detections.pop(0)\n        final_detections.append(best_detection)\n        \n        # Filter out detections that are too close to the best detection\n        detections = [d for d in detections if distance_3d(d, best_detection) > distance_threshold]\n    \n    return final_detections","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T02:27:59.554026Z","iopub.execute_input":"2025-04-08T02:27:59.554310Z","iopub.status.idle":"2025-04-08T02:27:59.571865Z","shell.execute_reply.started":"2025-04-08T02:27:59.554290Z","shell.execute_reply":"2025-04-08T02:27:59.571143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def generate_submission():\n    \"\"\"\n    Main function to generate the submission file\n    \"\"\"\n    # Get list of test tomograms\n    test_tomos = sorted([d for d in os.listdir(test_dir) if os.path.isdir(os.path.join(test_dir, d))])\n    total_tomos = len(test_tomos)\n    \n    print(f\"Found {total_tomos} tomograms in test directory\")\n    \n    # Clear GPU cache before starting\n    if torch.cuda.is_available():\n        torch.cuda.empty_cache()\n    \n    # Initialize model once outside the processing loop\n    print(f\"Loading YOLO model from {model_path}\")\n    model = YOLO(model_path)\n    model.to(device)\n    \n    # Additional optimizations for inference\n    if device.startswith('cuda'):\n        # Fuse conv and bn layers for faster inference\n        model.fuse()\n        \n        # Enable model half precision (FP16) if on compatible GPU\n        if torch.cuda.get_device_capability(0)[0] >= 7:  # Volta or newer\n            model.model.half()\n            print(\"Using half precision (FP16) for inference\")\n    \n    # Process tomograms with parallelization\n    results = []\n    motors_found = 0\n    \n    # Using ThreadPoolExecutor with max_workers=1 since each worker uses the GPU already\n    # and we're parallelizing within each tomogram processing\n    with ThreadPoolExecutor(max_workers=1) as executor:\n        future_to_tomo = {}\n        \n        # Submit all tomograms for processing\n        for i, tomo_id in enumerate(test_tomos, 1):\n            future = executor.submit(process_tomogram, tomo_id, model, i, total_tomos)\n            future_to_tomo[future] = tomo_id\n        \n        # Process completed futures as they complete\n        for future in future_to_tomo:\n            tomo_id = future_to_tomo[future]\n            try:\n                # Clear CUDA cache between tomograms\n                if torch.cuda.is_available():\n                    torch.cuda.empty_cache()\n                    \n                result = future.result()\n                results.append(result)\n                \n                # Update motors found count\n                has_motor = not pd.isna(result['Motor axis 0'])\n                if has_motor:\n                    motors_found += 1\n                    print(f\"Motor found in {tomo_id} at position: \"\n                          f\"z={result['Motor axis 0']}, y={result['Motor axis 1']}, x={result['Motor axis 2']}\")\n                else:\n                    print(f\"No motor detected in {tomo_id}\")\n                    \n                print(f\"Current detection rate: {motors_found}/{len(results)} ({motors_found/len(results)*100:.1f}%)\")\n            \n            except Exception as e:\n                print(f\"Error processing {tomo_id}: {e}\")\n                # Create a default entry for failed tomograms\n                results.append({\n                    'tomo_id': tomo_id,\n                    'Motor axis 0': -1,\n                    'Motor axis 1': -1,\n                    'Motor axis 2': -1\n                })\n    \n    # Create submission dataframe\n    submission_df = pd.DataFrame(results)\n    \n    # Ensure proper column order\n    submission_df = submission_df[['tomo_id', 'Motor axis 0', 'Motor axis 1', 'Motor axis 2']]\n    \n    # Save the submission file\n    submission_df.to_csv(submission_path, index=False)\n    \n    print(f\"\\nSubmission complete!\")\n    print(f\"Motors detected: {motors_found}/{total_tomos} ({motors_found/total_tomos*100:.1f}%)\")\n    print(f\"Submission saved to: {submission_path}\")\n    \n    # Display first few rows of submission\n    print(\"\\nSubmission preview:\")\n    print(submission_df.head())\n    \n    return submission_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T02:27:59.572593Z","iopub.execute_input":"2025-04-08T02:27:59.572793Z","iopub.status.idle":"2025-04-08T02:27:59.587675Z","shell.execute_reply.started":"2025-04-08T02:27:59.572776Z","shell.execute_reply":"2025-04-08T02:27:59.586947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Time entire process\nstart_time = time.time()\n    \n# Generate submission\nsubmission = generate_submission()\n    \n# Print total execution time\nelapsed = time.time() - start_time\nprint(f\"\\nTotal execution time: {elapsed:.2f} seconds ({elapsed/60:.2f} minutes)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T02:27:59.588359Z","iopub.execute_input":"2025-04-08T02:27:59.588624Z","iopub.status.idle":"2025-04-08T02:32:15.550956Z","shell.execute_reply.started":"2025-04-08T02:27:59.588592Z","shell.execute_reply":"2025-04-08T02:32:15.550238Z"}},"outputs":[],"execution_count":null}]}