{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"sourceType":"competition"},{"sourceId":10402120,"sourceType":"datasetVersion","datasetId":6445526}],"dockerImageVersionId":30823,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        os.path.join(dirname, filename)\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-08T12:52:00.874218Z","iopub.execute_input":"2025-01-08T12:52:00.874530Z","iopub.status.idle":"2025-01-08T12:52:07.275945Z","shell.execute_reply.started":"2025-01-08T12:52:00.874504Z","shell.execute_reply":"2025-01-08T12:52:07.275294Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 引入套件","metadata":{}},{"cell_type":"code","source":"import torch\nfrom torch.utils.data import DataLoader\nfrom torch.utils.data import Dataset\nimport torchvision\nfrom torchvision.models.detection import fasterrcnn_resnet50_fpn\nfrom torchvision.datasets import ImageFolder\nfrom torchvision import transforms\nimport torchvision.transforms as T\nfrom torchvision.models.detection.faster_rcnn import FastRCNNPredictor\nimport pydicom\nimport math\nimport cv2 as cv\nimport tensorflow as tf\nfrom tqdm import tqdm\n\nUSE_PRETRAINED_MODEL = True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T12:52:07.277073Z","iopub.execute_input":"2025-01-08T12:52:07.277303Z","iopub.status.idle":"2025-01-08T12:52:07.281452Z","shell.execute_reply.started":"2025-01-08T12:52:07.277283Z","shell.execute_reply":"2025-01-08T12:52:07.280599Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 載入FastRCNN模型","metadata":{}},{"cell_type":"code","source":"if not USE_PRETRAINED_MODEL:\n    # Load the pre-trained Faster R-CNN model with a ResNet-50 backbone\n    model = fasterrcnn_resnet50_fpn(weights=True)\n    \n    # Number of classes (your dataset classes + 1 for background)\n    num_classes = 2  # For example, 2 classes + background\n    \n    # Get the number of input features for the classifier\n    in_features = model.roi_heads.box_predictor.cls_score.in_features\n    \n    # Replace the head of the model with a new one (for the number of classes in your dataset)\n    model.roi_heads.box_predictor = FastRCNNPredictor(in_features, num_classes)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T12:52:07.283311Z","iopub.execute_input":"2025-01-08T12:52:07.283588Z","iopub.status.idle":"2025-01-08T12:52:07.286946Z","shell.execute_reply.started":"2025-01-08T12:52:07.283558Z","shell.execute_reply":"2025-01-08T12:52:07.286187Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Formatting Data","metadata":{}},{"cell_type":"code","source":"input_size = 244\n\ndef format_image(img, box):\n    height, width = img.shape \n    max_size = max(height, width)\n    r = max_size / input_size\n    new_width = int(width / r)\n    new_height = int(height / r)\n    new_size = (new_width, new_height)\n    resized = cv.resize(img, new_size, interpolation= cv.INTER_LINEAR)\n    new_image = np.zeros((input_size, input_size), dtype=np.uint8)\n    new_image[0:new_height, 0:new_width] = resized\n\n    x, y, w, h = (box[0], box[1], box[2], box[3]) if box[0] else (0.0,0.0,0.0,0.0)\n    new_box = [int((x)/ r), int((y)/ r), int(w/ r), int(h/ r)] if box[0] else [0.0,0.0,0.0,0.0]\n\n    return new_image, new_box","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T12:52:07.288334Z","iopub.execute_input":"2025-01-08T12:52:07.288566Z","iopub.status.idle":"2025-01-08T12:52:07.293843Z","shell.execute_reply.started":"2025-01-08T12:52:07.288548Z","shell.execute_reply":"2025-01-08T12:52:07.293097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nfrom torchvision import transforms as T\n\n# 定義轉換器\ntransform = T.Compose([\n    T.ToTensor(),                     # 確保影像轉換為 PyTorch Tensor\n    T.Normalize(mean=[0.5], std=[0.5]) # 正規化\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T12:52:07.294702Z","iopub.execute_input":"2025-01-08T12:52:07.294978Z","iopub.status.idle":"2025-01-08T12:52:07.298680Z","shell.execute_reply.started":"2025-01-08T12:52:07.294947Z","shell.execute_reply":"2025-01-08T12:52:07.297943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define transformations (e.g., resizing, normalization)\ntransform = T.Compose([\n    T.ToTensor(),\n])\n# Custom Dataset class or using an existing one\nclass PneumoniaDataset(Dataset):\n    def __init__(self, dataframe, transforms=None):\n        dataframe = dataframe.reset_index(drop=True)\n        # Initialize dataset paths and annotations here\n        self.transforms = transforms\n        # Your dataset logic (image paths, annotations, etc.)\n        self.dataframe = dataframe\n\n    def __getitem__(self, idx):\n        row = self.dataframe.iloc[idx]\n        img_path = row['patientId']\n        img_path = \"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/\"+img_path+\".dcm\"\n        # Load DICOM image\n        temp_img = pydicom.dcmread(img_path).pixel_array\n        # 確認標註框是否有效\n        temp_box = [row['x'], row['y'], row['width'], row['height']] if not math.isnan(row['x']) else [0.0, 0.0, 0.0, 0.0]\n\n        # 格式化影像與標註框\n        img, box = format_image(temp_img, temp_box)\n        \n        # 修正邊界框的格式\n        xmin, ymin, width, height = box\n        # print('width = ', width)\n        # print('height = ', height)\n        xmax = xmin + width\n        ymax = ymin + height\n\n        # 處理有效框或空框\n        if width <= 0.0 or height <= 0.0:\n            # print('空框處理')\n            boxes = torch.zeros((0, 4), dtype=torch.float32)  # 空框\n            labels = torch.zeros((0,), dtype=torch.int64)  # 空標籤\n        else:\n            if xmin > xmax or ymin > ymax:\n                print('wrong coordination')\n        \n            # 構建有效框\n            boxes = torch.tensor([[xmin, ymin, xmax, ymax]], dtype=torch.float32)\n            labels = torch.tensor([row['Target']], dtype=torch.int64)\n\n        # 影像正規化\n        # 影像正規化並轉換為 float32\n        img = img.astype(np.float32) / 255.  # 明確指定 float32\n\n       # 建立目標字典\n        target = {}\n        target[\"boxes\"] = boxes.clone().detach().float()  # 確保格式正確\n        target[\"labels\"] = labels.clone().detach().long() # labels 保持為 int64\n        # Apply transforms\n        if self.transforms is not None:\n            img = self.transforms(img)\n        return img, target\n    def __len__(self):\n        # Return the length of your dataset\n        return len(self.dataframe)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T12:52:07.299484Z","iopub.execute_input":"2025-01-08T12:52:07.299756Z","iopub.status.idle":"2025-01-08T12:52:07.307452Z","shell.execute_reply.started":"2025-01-08T12:52:07.299727Z","shell.execute_reply":"2025-01-08T12:52:07.306742Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 載入資料集","metadata":{}},{"cell_type":"code","source":"data_labels = pd.read_csv('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv')\ndataframe = data_labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T12:52:07.308384Z","iopub.execute_input":"2025-01-08T12:52:07.308657Z","iopub.status.idle":"2025-01-08T12:52:07.336944Z","shell.execute_reply.started":"2025-01-08T12:52:07.308629Z","shell.execute_reply":"2025-01-08T12:52:07.336340Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 檢查 DataFrame 長度和索引\nprint(f\"DataFrame 長度: {len(dataframe)}\")\nprint(f\"DataFrame 索引: {dataframe.index}\")\n\n# 確認是否存在索引不連續問題\nprint(f\"索引是否連續: {dataframe.index.is_monotonic_increasing}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T12:52:07.337741Z","iopub.execute_input":"2025-01-08T12:52:07.337936Z","iopub.status.idle":"2025-01-08T12:52:07.342268Z","shell.execute_reply.started":"2025-01-08T12:52:07.337919Z","shell.execute_reply":"2025-01-08T12:52:07.341497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_labels[:1001]['Target'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T12:52:07.344413Z","iopub.execute_input":"2025-01-08T12:52:07.344608Z","iopub.status.idle":"2025-01-08T12:52:07.350199Z","shell.execute_reply.started":"2025-01-08T12:52:07.344590Z","shell.execute_reply":"2025-01-08T12:52:07.349399Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 取前6000個資料作為訓練集\ntrain_dataset = PneumoniaDataset(data_labels[:6001],transform)\n\n# 取第6001~6200個資料作為驗證集\nvalid_dataset = PneumoniaDataset(data_labels[6001:6801],transform)\n\n# 建立DataLoader\ntrain_loader = DataLoader(train_dataset, batch_size=4, shuffle=True, \n                                   collate_fn=lambda x: tuple(zip(*x)))\nvalid_loader = DataLoader(valid_dataset, batch_size=4, shuffle=False, \n                                    collate_fn=lambda x: tuple(zip(*x)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T12:52:07.351353Z","iopub.execute_input":"2025-01-08T12:52:07.351553Z","iopub.status.idle":"2025-01-08T12:52:07.356687Z","shell.execute_reply.started":"2025-01-08T12:52:07.351535Z","shell.execute_reply":"2025-01-08T12:52:07.355871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 查看訓練集資料內容\nfor i, (img, target) in enumerate(train_dataset):\n    print(f\"Index {i}:\")\n    print(\"  Image shape:\", img.shape)\n    print(\"  Target:\", target)\n    \n    # 示範只看前 5 筆\n    if i == 4:\n        break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T12:52:07.357580Z","iopub.execute_input":"2025-01-08T12:52:07.357800Z","iopub.status.idle":"2025-01-08T12:52:07.401113Z","shell.execute_reply.started":"2025-01-08T12:52:07.357781Z","shell.execute_reply":"2025-01-08T12:52:07.400411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if not USE_PRETRAINED_MODEL:\n    # Move model to GPU if available\n    if torch.cuda.is_availabel():\n        device = torch.device('cuda')\n        print(\"using cude as device.\")\n    else:\n        torch.device('cpu')\n        print(\"using cpu as device.\")\n    \n    model.to(device)\n    \n    # Set up the optimizer\n    params = [p for p in model.parameters() if p.requires_grad]\n    optimizer = torch.optim.SGD(params, lr=0.005, momentum=0.9, \n                                                       weight_decay=0.0005)\n    # Learning rate scheduler\n    lr_scheduler = torch.optim.lr_scheduler.StepLR(optimizer, step_size=3, \n                                                                   gamma=0.1)\n    # Train the model\n    num_epochs = 3\n    for epoch in range(num_epochs):\n        model.train()\n        train_loss = 0.0\n    \n       # Training loop\n        for images, targets in tqdm(train_loader):\n            images = list(image.to(device) for image in images)\n            targets = [{k: v.to(device) for k, v in t.items()} for t in targets]\n    \n            # Zero the gradients\n            optimizer.zero_grad()\n    \n            # Forward pass\n            loss_dict = model(images, targets)\n            losses = sum(loss for loss in loss_dict.values())\n    \n            # Backward pass\n            losses.backward()\n            optimizer.step()\n            train_loss += losses.item()\n    \n        # Update the learning rate\n        lr_scheduler.step()\n        print(f'Epoch: {epoch + 1}, Loss: {train_loss / len(train_loader)}')\n    print(\"Training complete!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T12:52:07.401815Z","iopub.execute_input":"2025-01-08T12:52:07.402065Z","iopub.status.idle":"2025-01-08T12:52:07.408336Z","shell.execute_reply.started":"2025-01-08T12:52:07.402018Z","shell.execute_reply":"2025-01-08T12:52:07.407544Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if not USE_PRETRAINED_MODEL:\n    # 儲存完整模型\n    torch.save(model, '/kaggle/working/fasterrcnn_model.pth')\n    print(\"模型訓練完成並儲存至: /kaggle/working/fasterrcnn_model.pth\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T12:52:07.409182Z","iopub.execute_input":"2025-01-08T12:52:07.409472Z","iopub.status.idle":"2025-01-08T12:52:07.412851Z","shell.execute_reply.started":"2025-01-08T12:52:07.409443Z","shell.execute_reply":"2025-01-08T12:52:07.412096Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 將訓練好的模型直接拿來用","metadata":{}},{"cell_type":"code","source":"# 1. 先建立跟訓練時相同結構的模型\nmodel = fasterrcnn_resnet50_fpn(weights=False)  # 訓練好的模型就不需要再去下載預訓練權重，可以使用 weights=False\n\n# 2. 根據先前訓練時的 class 數量，替換掉 roi_heads.box_predictor\nnum_classes = 2  # 要與當初訓練時設定的 class 數量相同\nin_features = model.roi_heads.box_predictor.cls_score.in_features\nmodel.roi_heads.box_predictor = FastRCNNPredictor(in_features, num_classes)\n\n# 3. 載入模型權重\ndevice = torch.device('cuda') if torch.cuda.is_available() else torch.device('cpu')\nmodel = torch.load(\"/kaggle/input/trained-model/fasterrcnn_model.pth\", map_location=device)\n# 4. 將模型放到指定裝置上 (GPU or CPU)\nmodel.to(device)\n\nprint(\"模型已成功載入！\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T12:52:07.413706Z","iopub.execute_input":"2025-01-08T12:52:07.413978Z","iopub.status.idle":"2025-01-08T12:52:08.140363Z","shell.execute_reply.started":"2025-01-08T12:52:07.413950Z","shell.execute_reply":"2025-01-08T12:52:08.139464Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CONF = 0.65\n\ndef post_process_predictions(predictions, targets, conf_thresh):\n    \"\"\"\n    將模型輸出的預測結果 (predictions)，根據 conf_thresh 做過濾並只取最高信心度的框。\n    \n    參數:\n        predictions (list of dict): 模型對一個 batch 的推論結果，每個元素是一張影像的預測字典，\n                                    其結構常見為 {'boxes': ..., 'labels': ..., 'scores': ...}\n        targets (list of dict):     該 batch 各影像對應的標註 (ground truth)，用來對照。\n        conf_thresh (float):        信心度閾值，預設 0.7\n    \n    回傳:\n        results (list of dict):     後處理後的結果列表，每個元素包含:\n            {\n                'best_box': tensor([...]) 或 None,\n                'best_label': int 或 None,\n                'best_score': float 或 None,\n                'all_filtered_boxes': tensor([...]),\n                'all_filtered_labels': tensor([...]),\n                'all_filtered_scores': tensor([...]),\n                'no_high_conf': bool,  # 如果沒有任何框達到閾值，這裡為 True\n            }\n    \"\"\"\n    results = []\n\n    for i in range(len(predictions)):\n        pred_boxes = predictions[i]['boxes']\n        pred_labels = predictions[i]['labels']\n        pred_scores = predictions[i]['scores']\n\n        # 先依照 conf_thresh 過濾\n        high_conf_mask = pred_scores > conf_thresh\n        filtered_boxes = pred_boxes[high_conf_mask]\n        filtered_labels = pred_labels[high_conf_mask]\n        filtered_scores = pred_scores[high_conf_mask]\n\n        # 建立一個字典存放後處理結果\n        result_dict = {\n            'best_box': None,\n            'best_label': None,\n            'best_score': None,\n            'all_filtered_boxes': filtered_boxes,\n            'all_filtered_labels': filtered_labels,\n            'all_filtered_scores': filtered_scores,\n            'no_high_conf': False\n        }\n\n        if len(filtered_scores) == 0:\n            # 沒有任何預測超過閾值\n            result_dict['no_high_conf'] = True\n        else:\n            # 在篩選過的框中，找最高信心度\n            max_score_idx = torch.argmax(filtered_scores)\n            result_dict['best_box'] = filtered_boxes[max_score_idx]\n            result_dict['best_label'] = filtered_labels[max_score_idx].item()\n            result_dict['best_score'] = filtered_scores[max_score_idx].item()\n\n        results.append(result_dict)\n\n    return results\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T12:52:08.141516Z","iopub.execute_input":"2025-01-08T12:52:08.141837Z","iopub.status.idle":"2025-01-08T12:52:08.147615Z","shell.execute_reply.started":"2025-01-08T12:52:08.141804Z","shell.execute_reply":"2025-01-08T12:52:08.146772Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 輸出驗證集預測結果","metadata":{}},{"cell_type":"code","source":"model.eval()\n\nwith torch.no_grad():\n    batch_num = 1\n    for images, targets in valid_loader:\n        if(batch_num == 3):\n            break\n        print('===========================================')\n        print(f\"Validating batch {batch_num}\")\n        batch_num += 1\n        images = list(img.to(device) for img in images)\n        predictions = model(images)  # 一次對整個 batch 做推論\n        \n        # 呼叫我們的後處理函式\n        processed_results = post_process_predictions(predictions, targets, CONF)\n        \n        # 在這裡，可以決定如何輸出或使用 processed_results\n        # 例如，逐張影像查看結果:\n        \n        for i, result in enumerate(processed_results):\n            print(f\"image{i}: actual_labels:\", targets[i]['labels'])\n            if result['no_high_conf']:\n                print(f\"  No box over {CONF} confidence for image {i}.\")\n                print('----------------------------------------------')\n                continue\n            print(f\"  Highest confidence box :\")\n            print(\"    Box:\", result['best_box'])\n            print(\"    Label:\", result['best_label'])\n            print(\"    Score:\", result['best_score'])\n            print('----------------------------------------------')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T12:52:08.148268Z","iopub.execute_input":"2025-01-08T12:52:08.148455Z","iopub.status.idle":"2025-01-08T12:52:08.603673Z","shell.execute_reply.started":"2025-01-08T12:52:08.148438Z","shell.execute_reply":"2025-01-08T12:52:08.602970Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 評估模型並輸出Confusion Matrix","metadata":{}},{"cell_type":"code","source":"def evaluate_model(valid_loader, model, device):\n    \"\"\"\n    使用簡化的『有無預測框』來判斷 TP, FP, TN, FN，\n    並計算 Accuracy 與 F1-Score。\n    \n    參數:\n      valid_loader: 驗證資料的 DataLoader\n      model:        已訓練好的 PyTorch 目標偵測模型 (e.g. fasterrcnn_resnet50_fpn)\n      device:       'cpu' or 'cuda'\n      post_process_fn: 您的後處理函式 (例如 post_process_predictions)\n      conf_thresh:  閾值（預設 0.7）\n      \n    回傳:\n      None (直接在函式裡打印結果)\n    \"\"\"\n    model.eval()\n    \n    # 建立混淆矩陣 (Confusion Matrix) 的四個計數器\n    TP = 0\n    FP = 0\n    TN = 0\n    FN = 0\n\n    with torch.no_grad():\n        print(f'Total batch number: {len(valid_loader)}')\n        num = 0\n        for images, targets in valid_loader:\n            images = [img.to(device) for img in images]\n            predictions = model(images)  # 模型對一個 batch 的推論\n            \n            # 後處理 -> 得到每張影像最終篩選/最高分框資訊\n            processed_results = post_process_predictions(predictions, targets, CONF)\n\n            # 針對此 batch 的每張影像，更新 TP/FP/TN/FN\n            for i, result in enumerate(processed_results):\n                # 1. 實際標註（是否有標籤）\n                #    假設 labels 不為空 -> 視為「實際正 (actual positive)」\n                #    若 labels 為空 -> 「實際負 (actual negative)」\n                actual_positive = (len(targets[i]['labels']) > 0)\n                \n                # 2. 模型預測（是否有框 > conf_thresh）\n                #    如果後處理後的 'no_high_conf' 為 False，代表有至少一個框超過閾值\n                predicted_positive = (not result['no_high_conf'])\n                \n                # 3. 根據 actual_positive / predicted_positive 分類成 TP, FP, TN, FN\n                if actual_positive and predicted_positive:\n                    TP += 1\n                elif not actual_positive and predicted_positive:\n                    FP += 1\n                elif actual_positive and not predicted_positive:\n                    FN += 1\n                else:\n                    TN += 1\n            num += 1\n            print(f'batch {num} done')\n\n    # 計算指標\n    total = TP + FP + TN + FN\n    accuracy = (TP + TN) / total if total > 0 else 0.0\n\n    # 避免分母為 0\n    precision = TP / (TP + FP + 1e-8)\n    recall    = TP / (TP + FN + 1e-8)\n    f1        = 2 * precision * recall / (precision + recall + 1e-8)\n\n    print(\"Confusion Matrix (簡化二元判斷，只判有無畫框):\")\n    print(f\"  TP = {TP}, FP = {FP}, TN = {TN}, FN = {FN}\")\n    print(f\"Accuracy:  {accuracy:.4f}\")\n    print(f\"Precision: {precision:.4f}\")\n    print(f\"Recall:    {recall:.4f}\")\n    print(f\"F1-Score:  {f1:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T12:52:08.604574Z","iopub.execute_input":"2025-01-08T12:52:08.604864Z","iopub.status.idle":"2025-01-08T12:52:08.612073Z","shell.execute_reply.started":"2025-01-08T12:52:08.604831Z","shell.execute_reply":"2025-01-08T12:52:08.611270Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"evaluate_model(valid_loader, model, device)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T12:52:08.612702Z","iopub.execute_input":"2025-01-08T12:52:08.612888Z","iopub.status.idle":"2025-01-08T12:52:49.811794Z","shell.execute_reply.started":"2025-01-08T12:52:08.612871Z","shell.execute_reply":"2025-01-08T12:52:49.810896Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# IoU計算","metadata":{}},{"cell_type":"code","source":"def compute_iou(box1, box2):\n    \"\"\"\n    計算兩個框的 IoU (Intersection over Union)。\n    box1, box2 皆是 [xmin, ymin, xmax, ymax] 的格式 (可以是 Tensor 或 ndarray)。\n    回傳一個介於 [0, 1] 的浮點數。\n    \"\"\"\n    # 先確保是 numpy array (若傳入的是 Tensor 則轉成 numpy)\n    if torch.is_tensor(box1):\n        box1 = box1.detach().cpu().numpy()\n    if torch.is_tensor(box2):\n        box2 = box2.detach().cpu().numpy()\n    \n    x1_min, y1_min, x1_max, y1_max = box1\n    x2_min, y2_min, x2_max, y2_max = box2\n\n    # 交集部分\n    inter_xmin = max(x1_min, x2_min)\n    inter_ymin = max(y1_min, y2_min)\n    inter_xmax = min(x1_max, x2_max)\n    inter_ymax = min(y1_max, y2_max)\n\n    inter_width = max(0, inter_xmax - inter_xmin)\n    inter_height = max(0, inter_ymax - inter_ymin)\n    inter_area = inter_width * inter_height\n\n    # 各自面積\n    area1 = (x1_max - x1_min) * (y1_max - y1_min)\n    area2 = (x2_max - x2_min) * (y2_max - y2_min)\n\n    # 聯集面積 = area1 + area2 - inter_area\n    union_area = area1 + area2 - inter_area\n    \n    if union_area == 0:\n        return 0.0\n    \n    iou = inter_area / union_area\n    return iou\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T12:52:49.812608Z","iopub.execute_input":"2025-01-08T12:52:49.812831Z","iopub.status.idle":"2025-01-08T12:52:49.818076Z","shell.execute_reply.started":"2025-01-08T12:52:49.812805Z","shell.execute_reply":"2025-01-08T12:52:49.817272Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 繪製單一框的方法","metadata":{}},{"cell_type":"code","source":"import torch\nimport numpy as np\nimport matplotlib.patches as patches\n\ndef draw_image_and_boxes_on_ax(ax, img_tensor, target, predicted_box=None):\n    \"\"\"\n    在給定的 Matplotlib Axes 上，繪製單張影像 + 真實標註(紅色) + 預測框(綠色)，\n    並在圖上顯示 IoU (若 predicted_box 不為 None)。\n\n    參數:\n      ax:             Matplotlib Axes 物件 (例如由 plt.subplots() 建立)。\n      img_tensor:     形狀 [C,H,W] (或 [H,W]) 的 PyTorch Tensor (灰階或 RGB)。\n      target:         {'boxes': Tensor, 'labels': Tensor}，是真實標註。\n      predicted_box:  shape = [4] (xmin, ymin, xmax, ymax)，若無可填 None。\n    \"\"\"\n    # ---------- 1) Tensor -> NumPy，調整通道順序 ----------\n    img_np = img_tensor.detach().cpu().numpy()\n    if len(img_np.shape) == 3 and img_np.shape[0] in [1, 3]:\n        # [C, H, W] -> [H, W, C]\n        img_np = np.transpose(img_np, (1, 2, 0))  \n\n    # 判斷灰階\n    is_grayscale = False\n    if img_np.ndim == 3 and img_np.shape[2] == 1:\n        is_grayscale = True\n        img_np = img_np[..., 0]  # 變成 [H, W]\n\n    # ---------- 2) 顯示影像 ----------\n    if is_grayscale:\n        ax.imshow(img_np, cmap='gray')\n    else:\n        ax.imshow(img_np)\n    ax.axis('off')\n\n    # ---------- 3) 繪製真實框 (紅色) ----------\n    gt_boxes = target[\"boxes\"]\n    for box_tensor in gt_boxes:\n        box = box_tensor.cpu().numpy().astype(int)\n        xmin, ymin, xmax, ymax = box\n        rect_w = xmax - xmin\n        rect_h = ymax - ymin\n        rect = patches.Rectangle((xmin, ymin), rect_w, rect_h,\n                                 linewidth=2, edgecolor='red', facecolor='none')\n        ax.add_patch(rect)\n\n    # ---------- 4) 繪製預測框 (綠色)，並計算 IoU ----------\n    if predicted_box is not None:\n        box_pred = predicted_box.detach().cpu().numpy().astype(int)\n        xmin_p, ymin_p, xmax_p, ymax_p = box_pred\n        rect_w = xmax_p - xmin_p\n        rect_h = ymax_p - ymin_p\n        rect = patches.Rectangle((xmin_p, ymin_p), rect_w, rect_h,\n                                 linewidth=2, edgecolor='green', facecolor='none')\n        ax.add_patch(rect)\n\n        # 針對每個真實框，都計算 IoU 並顯示\n        for box_tensor in gt_boxes:\n            iou_val = compute_iou(box_tensor, predicted_box)\n            box_gt = box_tensor.cpu().numpy().astype(int)\n            xmin_g, ymin_g, xmax_g, ymax_g = box_gt\n\n            # 可以把 IoU 數字畫在真實框上方一點點\n            text_x = xmin_g\n            text_y = max(ymin_g - 5, 0)  # 避免座標<0\n\n            ax.text(text_x, text_y, f\"IoU={iou_val:.2f}\", \n                    color='yellow', fontsize=10,\n                    bbox=dict(facecolor='blue', alpha=0.3))\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T12:52:49.818888Z","iopub.execute_input":"2025-01-08T12:52:49.819144Z","iopub.status.idle":"2025-01-08T12:52:49.827014Z","shell.execute_reply.started":"2025-01-08T12:52:49.819123Z","shell.execute_reply":"2025-01-08T12:52:49.826225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nmodel.eval()\nall_ious =[]\nwith torch.no_grad():\n    for batch_idx, (images, targets) in enumerate(valid_loader):\n        images = [img.to(device) for img in images]\n        predictions = model(images)\n        processed_results = post_process_predictions(predictions, targets, CONF)\n\n        # 每個 batch 顯示 ncols = batch_size 張圖 (一行)\n        batch_size = len(images)\n        fig, axs = plt.subplots(nrows=1, ncols=batch_size, figsize=(5*batch_size, 5))\n\n        # 若只有 1 張圖，axs 不是 list，需要手動包成 list\n        if batch_size == 1:\n            axs = [axs]\n\n        for i, result in enumerate(processed_results):\n            # 繪圖邏輯\n            predicted_box = None\n            if not result['no_high_conf']:\n                predicted_box = result['best_box']  # shape=[4,]\n\n            # 在 axs[i] 上繪圖\n            draw_image_and_boxes_on_ax(\n                ax=axs[i],\n                img_tensor=images[i],\n                target=targets[i],\n                predicted_box=predicted_box\n            )\n\n            # IoU邏輯\n            gt_boxes = targets[i][\"boxes\"]  # shape: [N, 4]\n            has_gt = (len(gt_boxes) > 0)         # True 表示 GT 有框\n            has_pred = (not result['no_high_conf'])  # True 表示預測有框\n            \n            # -- Case 1: 兩者都沒有框 -> skip\n            if (not has_gt) and (not has_pred):\n                # 什麼都不做，不加入 IoU\n                continue\n\n            # -- Case 2: 兩者都有框 -> 計算 IoU\n            if has_gt and has_pred:\n                predicted_box = result['best_box']  # shape=[4,]\n                # 計算對所有 gt_box 的 IoU，取最大值\n                max_iou_for_this_image = 0.0\n                for gt_box in gt_boxes:\n                    iou_val = compute_iou(gt_box, predicted_box)\n                    max_iou_for_this_image = max(max_iou_for_this_image, iou_val)\n                all_ious.append(max_iou_for_this_image)\n            \n            else:\n                # -- Case 3: 僅一方有框 -> IoU=0\n                #   (要嘛 GT 有框但模型沒框，\n                #    或   GT 沒框但模型有框)\n                all_ious.append(0.0)\n\n        plt.suptitle(f\"Batch {batch_idx} (showing {batch_size} images)\", fontsize=16)\n        plt.tight_layout()\n        plt.show()\n\n\n# 最後計算平均 IoU\nif len(all_ious) > 0:\n    mean_iou = sum(all_ious) / len(all_ious)\nelse:\n    mean_iou = 0.0\n\nprint(f'length of all_ious : {len(all_ious)}')\nprint(f\"Average IoU over validation set = {mean_iou:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T12:52:49.827885Z","iopub.execute_input":"2025-01-08T12:52:49.828162Z","iopub.status.idle":"2025-01-08T12:55:10.096311Z","shell.execute_reply.started":"2025-01-08T12:52:49.828141Z","shell.execute_reply":"2025-01-08T12:55:10.095419Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"CONF = 0.65, Ave IoU = 0.1911","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}}]}