{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":75176,"databundleVersionId":8252256,"sourceType":"competition"}],"dockerImageVersionId":30762,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nfrom pathlib import Path\nfrom sklearn.utils import resample\nfrom sklearn.model_selection import train_test_split\nimport torch\nfrom torch.utils.data import DataLoader, Dataset\nimport torchvision.transforms as transforms\nfrom PIL import Image\nimport numpy as np\nimport os\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import accuracy_score, f1_score, precision_score, recall_score, classification_report\nfrom collections import defaultdict\nimport torch.optim as optim\nfrom torch.utils.tensorboard import SummaryWriter\nimport torchvision\nfrom torchvision.models.detection import FasterRCNN\nfrom torchvision.models.detection.rpn import AnchorGenerator\nimport torch.nn as nn\nimport gc","metadata":{"execution":{"iopub.status.busy":"2024-08-27T07:33:46.715035Z","iopub.execute_input":"2024-08-27T07:33:46.715327Z","iopub.status.idle":"2024-08-27T07:34:05.290552Z","shell.execute_reply.started":"2024-08-27T07:33:46.715294Z","shell.execute_reply":"2024-08-27T07:34:05.289711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Đường dẫn đến thư mục chứa dữ liệu gốc\nROOT = Path(\"/kaggle/input/amia-public-challenge-2024\")\n","metadata":{"execution":{"iopub.status.busy":"2024-08-27T07:34:05.292228Z","iopub.execute_input":"2024-08-27T07:34:05.292806Z","iopub.status.idle":"2024-08-27T07:34:05.297251Z","shell.execute_reply.started":"2024-08-27T07:34:05.292771Z","shell.execute_reply":"2024-08-27T07:34:05.296217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Đọc file train.csv\ntrain_df = pd.read_csv(ROOT / \"train.csv\")\n\n# Tạo cột mới 'new_class_id' cho bài toán 3 lớp\ndef map_class(row):\n    if row['class_id'] == 14:\n        row['new_class_id'] = 0  # Normal\n        row['x_min'] = 0.0\n        row['y_min'] = 0.0\n        row['x_max'] = 1.0\n        row['y_max'] = 1.0\n    elif row['class_id'] == 0:\n        row['new_class_id'] = 1  # Aortic enlargement\n    else:\n        row['new_class_id'] = 2  # Other abnormality\n    return row\n\n# Áp dụng hàm map_class để gắn nhãn và các giá trị bounding box cho lớp 'No finding'\ntrain_df = train_df.apply(map_class, axis=1)\n\n# Kiểm tra lại DataFrame đã được cập nhật\nprint(train_df.head())\n\n# Tách dữ liệu theo nhãn mới\ndf_class_0 = train_df[train_df['new_class_id'] == 0]  # Normal\ndf_class_1 = train_df[train_df['new_class_id'] == 1]  # Aortic enlargement\ndf_class_2 = train_df[train_df['new_class_id'] == 2]  # Other abnormality\n\n# Giảm số lượng mẫu của lớp 0 (Normal) để bằng với lớp 1\ndf_class_0_downsampled = resample(df_class_0, \n                                  replace=False,  # không sao chép\n                                  n_samples=len(df_class_1),  # số lượng mẫu bằng lớp 1\n                                  random_state=42)\n\n# Trộn các mẫu trong lớp 2 để có đủ các bệnh và số lượng bằng lớp 1\ngrouped_class_2 = df_class_2.groupby('class_id')\nsamples_per_class_2 = len(df_class_1) // len(grouped_class_2)\ndf_class_2_balanced = grouped_class_2.apply(lambda x: x.sample(samples_per_class_2, replace=True, random_state=42))\ndf_class_2_balanced = df_class_2_balanced.reset_index(drop=True, level=1)\ndf_class_2_balanced = df_class_2_balanced.reset_index(drop=True)\n\n# Kết hợp lại dữ liệu đã cân bằng\nbalanced_df = pd.concat([df_class_0_downsampled, df_class_1, df_class_2_balanced])\n\n# Chia dữ liệu thành tập huấn luyện (70%), tập tạm kiểm tra (30%)\ntrain_data, temp_data = train_test_split(balanced_df, test_size=0.3, random_state=42, stratify=balanced_df['new_class_id'])\n\n# Chia tập tạm kiểm tra thành tập validation (10%) và tập kiểm tra (20%)\nvalid_data, test_data = train_test_split(temp_data, test_size=2/3, random_state=42, stratify=temp_data['new_class_id'])","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-08-27T07:34:05.298800Z","iopub.execute_input":"2024-08-27T07:34:05.299153Z","iopub.status.idle":"2024-08-27T07:34:34.819128Z","shell.execute_reply.started":"2024-08-27T07:34:05.299112Z","shell.execute_reply":"2024-08-27T07:34:34.818120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Ghi dữ liệu vào các file CSV\nWORK_DIR = Path(\"/kaggle/working\")\ntrain_data.to_csv(WORK_DIR / \"train_balanced.csv\", index=False)\nvalid_data.to_csv(WORK_DIR / \"valid_balanced.csv\", index=False)\ntest_data.to_csv(WORK_DIR / \"test_balanced.csv\", index=False)\nprint(\"Files saved successfully in /kaggle/working/\")","metadata":{"execution":{"iopub.status.busy":"2024-08-27T07:34:34.821493Z","iopub.execute_input":"2024-08-27T07:34:34.821805Z","iopub.status.idle":"2024-08-27T07:34:34.950945Z","shell.execute_reply.started":"2024-08-27T07:34:34.821773Z","shell.execute_reply":"2024-08-27T07:34:34.950017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ChestXrayDataset(Dataset):\n    def __init__(self, dataframe, root_dir, transform=None):\n        self.dataframe = dataframe\n        self.root_dir = root_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.dataframe)\n\n    def __getitem__(self, idx):\n        # Lấy thông tin hình ảnh\n        img_name = os.path.join(self.root_dir, self.dataframe.iloc[idx]['image_id'] + '.png')\n        image = Image.open(img_name).convert(\"RGB\")\n        \n        if self.transform:\n            image = self.transform(image)\n        \n        # Xử lý thông tin bounding box và nhãn\n        new_class_id = self.dataframe.iloc[idx]['new_class_id']\n        \n        if new_class_id == 0:  # \"No finding\"\n            # Sử dụng bounding box giả định cho lớp \"No finding\"\n            boxes = torch.tensor([[0.0, 0.0, 1.0, 1.0]], dtype=torch.float32)\n            labels = torch.tensor([0], dtype=torch.int64)  # Lớp 0 là lớp giả định cho \"No finding\"\n        else:\n            # Sử dụng thông tin bounding box từ dataframe\n            box_values = self.dataframe.iloc[idx][['x_min', 'y_min', 'x_max', 'y_max']].values\n            box_values = box_values.astype(float)  # Đảm bảo tất cả các giá trị là float\n            \n            # Kiểm tra bounding box hợp lệ\n            if box_values[2] <= box_values[0] or box_values[3] <= box_values[1]:\n                # Nếu không hợp lệ, sử dụng giá trị mặc định\n                boxes = torch.tensor([[0.0, 0.0, 1.0, 1.0]], dtype=torch.float32)\n            else:\n                boxes = torch.as_tensor(box_values, dtype=torch.float32).unsqueeze(0)\n            \n            labels = torch.tensor([new_class_id], dtype=torch.int64)\n        \n        # Chuẩn bị thông tin cho mô hình phát hiện đối tượng\n        target = {}\n        target[\"boxes\"] = boxes\n        target[\"labels\"] = labels\n        \n        return image, target","metadata":{"execution":{"iopub.status.busy":"2024-08-27T07:34:34.952218Z","iopub.execute_input":"2024-08-27T07:34:34.952543Z","iopub.status.idle":"2024-08-27T07:34:34.964381Z","shell.execute_reply.started":"2024-08-27T07:34:34.952510Z","shell.execute_reply":"2024-08-27T07:34:34.963445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def collate_fn(batch):\n    images, targets = zip(*batch)\n    return list(images), list(targets)","metadata":{"execution":{"iopub.status.busy":"2024-08-27T07:34:34.965854Z","iopub.execute_input":"2024-08-27T07:34:34.966751Z","iopub.status.idle":"2024-08-27T07:34:34.976609Z","shell.execute_reply.started":"2024-08-27T07:34:34.966716Z","shell.execute_reply":"2024-08-27T07:34:34.975697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Đường dẫn tới thư mục chứa hình ảnh\nTRAIN_DIR = ROOT / \"train/train\"\n\n# Các transform cho dữ liệu huấn luyện và validation\ntransform = transforms.Compose([\n    transforms.Resize((224, 224)),  # Giảm kích thước ảnh để tiết kiệm bộ nhớ\n    transforms.ToTensor(),\n    transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225])\n])\n\n# Tạo dataset và dataloader\ntrain_dataset = ChestXrayDataset(dataframe=train_data, root_dir=TRAIN_DIR, transform=transform)\nvalid_dataset = ChestXrayDataset(dataframe=valid_data, root_dir=TRAIN_DIR, transform=transform)\ntest_dataset = ChestXrayDataset(dataframe=test_data, root_dir=TRAIN_DIR, transform=transform)\n\ntrain_loader = DataLoader(train_dataset, batch_size=4, shuffle=True, collate_fn=collate_fn)\nvalid_loader = DataLoader(valid_dataset, batch_size=4, shuffle=False, collate_fn=collate_fn)\ntest_loader = DataLoader(test_dataset, batch_size=4, shuffle=False, collate_fn=collate_fn)\n","metadata":{"execution":{"iopub.status.busy":"2024-08-27T07:34:34.977691Z","iopub.execute_input":"2024-08-27T07:34:34.977998Z","iopub.status.idle":"2024-08-27T07:34:34.990105Z","shell.execute_reply.started":"2024-08-27T07:34:34.977966Z","shell.execute_reply":"2024-08-27T07:34:34.989182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Kiểm tra một số mẫu từ dataset\nfor i in range(15):\n    image, target = train_dataset[i]\n    print(f\"Sample {i}: \", target)","metadata":{"execution":{"iopub.status.busy":"2024-08-27T07:34:34.991297Z","iopub.execute_input":"2024-08-27T07:34:34.991633Z","iopub.status.idle":"2024-08-27T07:34:35.716919Z","shell.execute_reply.started":"2024-08-27T07:34:34.991599Z","shell.execute_reply":"2024-08-27T07:34:35.715980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sử dụng TensorBoard\nwriter = SummaryWriter(log_dir='/kaggle/working/tensorboard_logs')\n","metadata":{"execution":{"iopub.status.busy":"2024-08-27T07:34:35.718444Z","iopub.execute_input":"2024-08-27T07:34:35.718833Z","iopub.status.idle":"2024-08-27T07:34:35.726705Z","shell.execute_reply.started":"2024-08-27T07:34:35.718791Z","shell.execute_reply":"2024-08-27T07:34:35.725861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torchvision.models.detection as detection\n\n# Load Faster R-CNN model pre-trained on COCO\nmodel = torchvision.models.detection.fasterrcnn_resnet50_fpn(weights=detection.FasterRCNN_ResNet50_FPN_Weights.COCO_V1)\n\n# Thay đổi số lượng lớp đầu ra thành 3 (Normal, Aortic enlargement, Other abnormality)\nnum_classes = 3  # 3 classes: Normal, Aortic enlargement, Other abnormality\nin_features = model.roi_heads.box_predictor.cls_score.in_features\nmodel.roi_heads.box_predictor = nn.Linear(in_features, num_classes)\n\n# Thiết lập thiết bị huấn luyện\ndevice = torch.device('cuda') if torch.cuda.is_available() else torch.device('cpu')\nmodel.to(device)","metadata":{"execution":{"iopub.status.busy":"2024-08-27T07:34:35.730039Z","iopub.execute_input":"2024-08-27T07:34:35.730667Z","iopub.status.idle":"2024-08-27T07:34:40.056646Z","shell.execute_reply.started":"2024-08-27T07:34:35.730629Z","shell.execute_reply":"2024-08-27T07:34:40.055758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Định nghĩa các hàm lưu và tải checkpoint\ndef save_checkpoint(model, optimizer, scheduler, epoch, best_score, loss, file_path):\n    if isinstance(model, torch.nn.DataParallel):\n        model = model.module\n    \n    torch.save({\n        'epoch': epoch,\n        'model_state_dict': model.state_dict(),\n        'optimizer_state_dict': optimizer.state_dict(),\n        'scheduler_state_dict': scheduler.state_dict() if scheduler else None,\n        'loss': loss,\n        'best_score': best_score\n    }, file_path)\n    print(f\"Checkpoint saved at {file_path}\")\n\ndef load_checkpoint(file_path, model, optimizer, scheduler):\n    checkpoint = torch.load(file_path)\n    model.load_state_dict(checkpoint['model_state_dict'])\n    optimizer.load_state_dict(checkpoint['optimizer_state_dict'])\n    if scheduler is not None and 'scheduler_state_dict' in checkpoint:\n        scheduler.load_state_dict(checkpoint['scheduler_state_dict'])\n    start_epoch = checkpoint['epoch'] + 1\n    best_score = checkpoint['best_score']\n    loss = checkpoint['loss']\n    print(f\"Loaded checkpoint from {file_path} (epoch {start_epoch-1}, best_score {best_score})\")\n    return start_epoch, best_score","metadata":{"execution":{"iopub.status.busy":"2024-08-27T07:34:40.057908Z","iopub.execute_input":"2024-08-27T07:34:40.058175Z","iopub.status.idle":"2024-08-27T07:34:40.067156Z","shell.execute_reply.started":"2024-08-27T07:34:40.058145Z","shell.execute_reply":"2024-08-27T07:34:40.066178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Hàm tính IoU\ndef calculate_iou(box1, box2):\n    x1_min, y1_min, x1_max, y1_max = box1\n    x2_min, y2_min, x2_max, y2_max = box2\n\n    # Calculate the (x, y)-coordinates of the intersection rectangle\n    x_inter_min = max(x1_min, x2_min)\n    y_inter_min = max(y1_min, y2_min)\n    x_inter_max = min(x1_max, x2_max)\n    y_inter_max = min(y1_max, y2_max)\n\n    # Compute the area of intersection rectangle\n    inter_area = max(0, x_inter_max - x_inter_min) * max(0, y_inter_max - y_inter_min)\n\n    # Compute the area of both the prediction and ground-truth rectangles\n    box1_area = (x1_max - x1_min) * (y1_max - y1_min)\n    box2_area = (x2_max - x2_min) * (y2_max - y2_min)\n\n    # Compute the IoU\n    iou = inter_area / float(box1_area + box2_area - inter_area)\n    \n    return iou","metadata":{"execution":{"iopub.status.busy":"2024-08-27T07:34:40.068381Z","iopub.execute_input":"2024-08-27T07:34:40.068774Z","iopub.status.idle":"2024-08-27T07:34:40.089302Z","shell.execute_reply.started":"2024-08-27T07:34:40.068730Z","shell.execute_reply":"2024-08-27T07:34:40.088336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cấu hình tối ưu hóa\noptimizer = optim.Adam(model.parameters(), lr=0.001)\nscheduler = torch.optim.lr_scheduler.StepLR(optimizer, step_size=3, gamma=0.1)\nnum_epochs = 10","metadata":{"execution":{"iopub.status.busy":"2024-08-27T07:34:40.090344Z","iopub.execute_input":"2024-08-27T07:34:40.090657Z","iopub.status.idle":"2024-08-27T07:34:40.099341Z","shell.execute_reply.started":"2024-08-27T07:34:40.090624Z","shell.execute_reply":"2024-08-27T07:34:40.098517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lists to store losses and metrics\ntrain_losses = []\nval_losses = []\ntrain_accuracies = []\nval_accuracies = []\nious = []\nf1_scores = []\n\nbest_f1 = 0.0  # Track the best F1-score","metadata":{"execution":{"iopub.status.busy":"2024-08-27T07:34:40.100249Z","iopub.execute_input":"2024-08-27T07:34:40.100575Z","iopub.status.idle":"2024-08-27T07:34:40.110663Z","shell.execute_reply.started":"2024-08-27T07:34:40.100537Z","shell.execute_reply":"2024-08-27T07:34:40.109737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Hàm huấn luyện cập nhật\ndef train_one_epoch(model, optimizer, data_loader, device, epoch):\n    model.train()\n    total_loss = 0\n    correct_predictions = 0\n    total = 0\n    all_pred_boxes = []\n    all_true_boxes = []\n    \n    for images, targets in data_loader:\n        images = list(image.to(device) for image in images)\n        targets = [{k: v.to(device) for k, v in t.items()} for t in targets]\n        \n        optimizer.zero_grad()\n\n        try:\n            # Kiểm tra định dạng của targets trước khi truyền vào mô hình\n            if not all(['boxes' in t and 'labels' in t for t in targets]):\n                raise ValueError(\"Missing 'boxes' or 'labels' in targets\")\n\n            # Lấy đầu ra từ mô hình\n            loss_dict = model(images, targets)\n            losses = sum(loss for loss in loss_dict.values())\n            \n            losses.backward()\n            optimizer.step()\n            \n            total_loss += losses.item()\n        \n            # Tính toán các chỉ số\n            with torch.no_grad():\n                outputs = model(images)\n                for i, output in enumerate(outputs):\n                    pred_labels = output['labels'].cpu().numpy()\n                    true_labels = targets[i]['labels'].cpu().numpy()\n                \n                    all_pred_boxes.extend(output['boxes'].cpu().numpy())\n                    all_true_boxes.extend(targets[i]['boxes'].cpu().numpy())\n                \n                    correct_predictions += (pred_labels == true_labels).sum()\n                    total += len(true_labels)\n        except Exception as e:\n            print(f\"Error processing batch: {e}\")\n            continue\n\n    epoch_loss = total_loss / len(data_loader)\n    epoch_acc = correct_predictions / total if total > 0 else 0\n    \n    # Tính toán IoU cho từng cặp predicted và true bounding boxes\n    iou_scores = []\n    for pred_box, true_box in zip(all_pred_boxes, all_true_boxes):\n        iou = calculate_iou(pred_box, true_box)\n        iou_scores.append(iou)\n    \n    avg_iou = np.mean(iou_scores) if len(iou_scores) > 0 else 0\n\n    print(f\"Epoch {epoch}, Loss: {epoch_loss}, Accuracy: {epoch_acc}, Avg IoU: {avg_iou}\")\n    \n    # Xóa bộ nhớ cache của GPU\n    torch.cuda.empty_cache()\n    gc.collect()\n    \n    return epoch_loss, epoch_acc, avg_iou","metadata":{"execution":{"iopub.status.busy":"2024-08-27T07:34:40.111834Z","iopub.execute_input":"2024-08-27T07:34:40.112140Z","iopub.status.idle":"2024-08-27T07:34:40.125674Z","shell.execute_reply.started":"2024-08-27T07:34:40.112108Z","shell.execute_reply":"2024-08-27T07:34:40.124654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_one_epoch(model, optimizer, data_loader, device, epoch):\n    model.train()\n    total_loss = 0\n    correct_predictions = 0\n    total = 0\n    all_pred_boxes = []\n    all_true_boxes = []\n    \n    for images, targets in data_loader:\n        images = list(image.to(device) for image in images)\n        targets = [{k: v.to(device) for k, v in t.items()} for t in targets]\n        \n        optimizer.zero_grad()\n\n        try:\n            # Kiểm tra định dạng của targets trước khi truyền vào mô hình\n            if not all(['boxes' in t and 'labels' in t for t in targets]):\n                raise ValueError(\"Missing 'boxes' or 'labels' in targets\")\n\n            # Thay đổi cách bạn lấy đầu ra từ mô hình\n            loss_dict = model(images, targets)\n            losses = sum(loss for loss in loss_dict.values())\n            \n            losses.backward()\n            optimizer.step()\n            \n            total_loss += losses.item()\n        \n            # Calculate metrics\n            with torch.no_grad():\n                outputs = model(images)\n                for i, output in enumerate(outputs):\n                    pred_labels = output['labels'].cpu().numpy()\n                    true_labels = targets[i]['labels'].cpu().numpy()\n                \n                    all_pred_boxes.extend(output['boxes'].cpu().numpy())\n                    all_true_boxes.extend(targets[i]['boxes'].cpu().numpy())\n                \n                    correct_predictions += (pred_labels == true_labels).sum()\n                    total += len(true_labels)\n        except Exception as e:\n            print(f\"Error processing batch: {e}\")\n            continue\n\n    epoch_loss = total_loss / len(data_loader)\n    epoch_acc = correct_predictions / total if total > 0 else 0\n    \n    # Calculate IoU for each pair of predicted and true bounding boxes\n    iou_scores = []\n    for pred_box, true_box in zip(all_pred_boxes, all_true_boxes):\n        iou = calculate_iou(pred_box, true_box)\n        iou_scores.append(iou)\n    \n    avg_iou = np.mean(iou_scores) if len(iou_scores) > 0 else 0\n\n    print(f\"Epoch {epoch}, Loss: {epoch_loss}, Accuracy: {epoch_acc}, Avg IoU: {avg_iou}\")\n    \n    # Xóa bộ nhớ cache của GPU\n    torch.cuda.empty_cache()\n    gc.collect()\n    \n    return epoch_loss, epoch_acc, avg_iou","metadata":{"execution":{"iopub.status.busy":"2024-08-27T07:34:40.126869Z","iopub.execute_input":"2024-08-27T07:34:40.127159Z","iopub.status.idle":"2024-08-27T07:34:40.140326Z","shell.execute_reply.started":"2024-08-27T07:34:40.127129Z","shell.execute_reply":"2024-08-27T07:34:40.139395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Vòng lặp huấn luyện\nstart_epoch = 0\nfor epoch in range(start_epoch, num_epochs):\n    try:\n        train_loss, train_acc, train_iou = train_one_epoch(model, optimizer, train_loader, device, epoch)\n        train_losses.append(train_loss)\n        train_accuracies.append(train_acc)\n        ious.append(train_iou)\n\n        # Validation step\n        model.eval()\n        val_loss = 0\n        val_correct = 0\n        val_total = 0\n        all_y_true = []\n        all_y_pred = []\n        with torch.no_grad():\n            for images, targets in valid_loader:\n                images = list(image.to(device) for image in images)\n                targets = [{k: v.to(device) for k, v in t.items()} for t in targets]\n\n                # Kiểm tra loss của mô hình\n                loss_dict = model(images, targets)\n                losses = sum(loss for loss in loss_dict.values())\n                val_loss += losses.item()\n\n                # Dự đoán từ mô hình\n                outputs = model(images)\n                for i, output in enumerate(outputs):\n                    pred_labels = output['labels'].cpu().numpy()\n                    true_labels = targets[i]['labels'].cpu().numpy()\n\n                    all_y_pred.extend(pred_labels)\n                    all_y_true.extend(true_labels)\n\n                    val_correct += (pred_labels == true_labels).sum()\n                    val_total += len(true_labels)\n\n        epoch_val_loss = val_loss / len(valid_loader)\n        epoch_val_acc = val_correct / val_total\n        val_losses.append(epoch_val_loss)\n        val_accuracies.append(epoch_val_acc)\n\n        # Tính toán các chỉ số phân loại\n        classification_report_dict = classification_report(all_y_true, all_y_pred, target_names=['Normal', 'Aortic enlargement', 'Other abnormality'], output_dict=True)\n        val_f1 = classification_report_dict['Aortic enlargement']['f1-score']\n        val_precision = classification_report_dict['Aortic enlargement']['precision']\n        val_recall = classification_report_dict['Aortic enlargement']['recall']\n\n        print(f\"Validation Loss: {epoch_val_loss}, Validation Accuracy: {epoch_val_acc}, F1-Score (Aortic enlargement): {val_f1}\")\n\n        # Lưu mô hình tốt nhất\n        if val_f1 > best_f1:\n            best_f1 = val_f1\n            save_checkpoint(model, optimizer, scheduler, epoch, best_f1, epoch_val_loss, f'/kaggle/working/best_model.pth.tar')\n\n        # Log vào TensorBoard\n        writer.add_scalar('Loss/train', train_loss, epoch)\n        writer.add_scalar('Loss/val', epoch_val_loss, epoch)\n        writer.add_scalar('Accuracy/train', train_acc, epoch)\n        writer.add_scalar('Accuracy/val', epoch_val_acc, epoch)\n        writer.add_scalar('F1-Score/val', val_f1, epoch)\n        writer.add_scalar('Precision/val', val_precision, epoch)\n        writer.add_scalar('Recall/val', val_recall, epoch)\n\n        # Step the scheduler\n        scheduler.step()\n\n        # Xóa bộ nhớ cache của GPU sau mỗi epoch\n        torch.cuda.empty_cache()\n        gc.collect()\n    except Exception as e:\n        print(f\"Error in epoch {epoch}: {e}\")\n        continue","metadata":{"execution":{"iopub.status.busy":"2024-08-27T07:34:40.141569Z","iopub.execute_input":"2024-08-27T07:34:40.141851Z","iopub.status.idle":"2024-08-27T07:36:30.455852Z","shell.execute_reply.started":"2024-08-27T07:34:40.141818Z","shell.execute_reply":"2024-08-27T07:36:30.454275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Đóng TensorBoard\nwriter.close()","metadata":{"execution":{"iopub.status.busy":"2024-08-27T07:36:30.456815Z","iopub.status.idle":"2024-08-27T07:36:30.457303Z","shell.execute_reply.started":"2024-08-27T07:36:30.457051Z","shell.execute_reply":"2024-08-27T07:36:30.457076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Hiển thị một số hình ảnh kiểm tra và kiểm thử mô hình\ndef display_sample_images(data_loader, model, device):\n    model.eval()\n    images, targets = next(iter(data_loader))\n    images = list(img.to(device) for img in images)\n\n    with torch.no_grad():\n        outputs = model(images)\n\n    fig, ax = plt.subplots(1, len(images), figsize=(20, 10))\n    for i in range(len(images)):\n        img = images[i].permute(1, 2, 0).cpu().numpy()\n        ax[i].imshow(img)\n        boxes = outputs[i]['boxes'].cpu().numpy()\n        for box in boxes:\n            xmin, ymin, xmax, ymax = box\n            rect = plt.Rectangle((xmin, ymin), xmax - xmin, ymax - ymin, fill=False, color='red')\n            ax[i].add_patch(rect)\n        ax[i].set_title(f\"Predicted Labels: {outputs[i]['labels'].cpu().numpy()}\")\n\n    plt.show()\n\n# Display test images with predictions\ndisplay_sample_images(test_loader, model, device)","metadata":{"execution":{"iopub.status.busy":"2024-08-27T07:36:30.459406Z","iopub.status.idle":"2024-08-27T07:36:30.459922Z","shell.execute_reply.started":"2024-08-27T07:36:30.459673Z","shell.execute_reply":"2024-08-27T07:36:30.459699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Kiểm thử mô hình\ndef test_model(model, data_loader, device):\n    model.eval()\n    y_true = []\n    y_pred = []\n    iou_scores = []\n    \n    with torch.no_grad():\n        for images, targets in data_loader:\n            images = list(image.to(device) for image in images)\n            outputs = model(images)\n            \n            for i, output in enumerate(outputs):\n                pred_labels = output['labels'].cpu().numpy()\n                true_labels = targets[i]['labels'].cpu().numpy()\n                \n                y_pred.extend(pred_labels)\n                y_true.extend(true_labels)\n                \n                pred_boxes = output['boxes'].cpu().numpy()\n                true_boxes = targets[i]['boxes'].cpu().numpy()\n                \n                # Calculate IoU for each pair of predicted and true bounding boxes\n                for pred_box, true_box in zip(pred_boxes, true_boxes):\n                    iou = calculate_iou(pred_box, true_box)\n                    iou_scores.append(iou)\n    \n    avg_iou = np.mean(iou_scores)\n    print(f\"Test Accuracy: {accuracy_score(y_true, y_pred)}\")\n    print(f\"Avg IoU on Test Set: {avg_iou}\")\n    print(classification_report(y_true, y_pred, target_names=['Normal', 'Aortic enlargement', 'Other abnormality']))\n\n# Kiểm thử trên tập test\ntest_model(model, test_loader, device)","metadata":{"execution":{"iopub.status.busy":"2024-08-27T07:36:30.461229Z","iopub.status.idle":"2024-08-27T07:36:30.461723Z","shell.execute_reply.started":"2024-08-27T07:36:30.461450Z","shell.execute_reply":"2024-08-27T07:36:30.461493Z"},"trusted":true},"execution_count":null,"outputs":[]}]}