{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":75176,"databundleVersionId":8252256,"sourceType":"competition"}],"dockerImageVersionId":30762,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!git clone https://github.com/ultralytics/yolov5  # Clone YOLOv5 repository\n%cd yolov5\n%pip install -qr requirements.txt  # Install dependencies","metadata":{"execution":{"iopub.status.busy":"2024-08-27T13:28:33.568737Z","iopub.execute_input":"2024-08-27T13:28:33.569050Z","iopub.status.idle":"2024-08-27T13:28:52.824688Z","shell.execute_reply.started":"2024-08-27T13:28:33.569016Z","shell.execute_reply":"2024-08-27T13:28:52.823519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"import pandas as pd\nimport os\nfrom pathlib import Path\nfrom sklearn.utils import resample\nfrom sklearn.model_selection import train_test_split\nimport yaml\nfrom shutil import copyfile\n\n# Đường dẫn đến thư mục chứa dữ liệu gốc\nROOT = Path(\"/kaggle/input/amia-public-challenge-2024\")\n\n# Đọc file train.csv\ntrain_df = pd.read_csv(ROOT / \"train.csv\")\n\n# Tạo cột mới 'new_class_id' cho bài toán 3 lớp\ndef map_class(row):\n    if row['class_id'] == 14:\n        row['new_class_id'] = 0  # Normal\n        row['x_min'] = 0.0\n        row['y_min'] = 0.0\n        row['x_max'] = 1.0\n        row['y_max'] = 1.0\n    elif row['class_id'] == 0:\n        row['new_class_id'] = 1  # Aortic enlargement\n    else:\n        row['new_class_id'] = 2  # Other abnormality\n    return row\n\n# Áp dụng hàm map_class để gắn nhãn và các giá trị bounding box cho lớp 'No finding'\ntrain_df = train_df.apply(map_class, axis=1)\n\n# Tách dữ liệu theo nhãn mới\ndf_class_0 = train_df[train_df['new_class_id'] == 0]  # Normal\ndf_class_1 = train_df[train_df['new_class_id'] == 1]  # Aortic enlargement\ndf_class_2 = train_df[train_df['new_class_id'] == 2]  # Other abnormality\n\n# Giảm số lượng mẫu của lớp 0 (Normal) để bằng với lớp 1\ndf_class_0_downsampled = resample(df_class_0, \n                                  replace=False,  # không sao chép\n                                  n_samples=len(df_class_1),  # số lượng mẫu bằng lớp 1\n                                  random_state=42)\n\n# Trộn các mẫu trong lớp 2 để có đủ các bệnh và số lượng bằng lớp 1\ngrouped_class_2 = df_class_2.groupby('class_id')\nsamples_per_class_2 = len(df_class_1) // len(grouped_class_2)\ndf_class_2_balanced = grouped_class_2.apply(lambda x: x.sample(samples_per_class_2, replace=True, random_state=42))\ndf_class_2_balanced = df_class_2_balanced.reset_index(drop=True, level=1)\ndf_class_2_balanced = df_class_2_balanced.reset_index(drop=True)\n\n# Kết hợp lại dữ liệu đã cân bằng\nbalanced_df = pd.concat([df_class_0_downsampled, df_class_1, df_class_2_balanced])\n\n# Chia dữ liệu thành tập huấn luyện (70%), tập tạm kiểm tra (30%)\ntrain_data, temp_data = train_test_split(balanced_df, test_size=0.3, random_state=42, stratify=balanced_df['new_class_id'])\n\n# Chia tập tạm kiểm tra thành tập validation (10%) và tập kiểm tra (20%)\nvalid_data, test_data = train_test_split(temp_data, test_size=2/3, random_state=42, stratify=temp_data['new_class_id'])\n\n# Đường dẫn đến thư mục YOLOv5 và các tập dữ liệu\nYOLO_DIR = Path('/kaggle/working/yolov5_data/')\nYOLO_DIR.mkdir(parents=True, exist_ok=True)\n\n# Tạo các thư mục cho YOLOv5\nfor folder in ['train/images', 'train/labels', 'val/images', 'val/labels', 'test/images', 'test/labels']:\n    (YOLO_DIR / folder).mkdir(parents=True, exist_ok=True)\n\n# Chức năng chuyển đổi sang định dạng YOLO\ndef convert_to_yolo_format(df, data_type):\n    for i, row in df.iterrows():\n        # Đường dẫn hình ảnh\n        img_path = ROOT / f\"train/train/{row['image_id']}.png\"\n        img_dest = YOLO_DIR / f\"{data_type}/images/{row['image_id']}.png\"\n        copyfile(img_path, img_dest)\n        \n        # Đường dẫn nhãn\n        label_path = YOLO_DIR / f\"{data_type}/labels/{row['image_id']}.txt\"\n        \n        # Chuyển đổi bounding box sang định dạng YOLO\n        if row['new_class_id'] == 0:\n            label_content = f\"0 0.5 0.5 1.0 1.0\\n\"\n        else:\n            x_center = (row['x_min'] + row['x_max']) / 2 / 1024\n            y_center = (row['y_min'] + row['y_max']) / 2 / 1024\n            width = (row['x_max'] - row['x_min']) / 1024\n            height = (row['y_max'] - row['y_min']) / 1024\n            label_content = f\"{row['new_class_id']} {x_center} {y_center} {width} {height}\\n\"\n        \n        # Ghi nhãn vào file\n        with open(label_path, 'w') as f:\n            f.write(label_content)\n\n# Chuyển đổi tất cả tập dữ liệu sang định dạng YOLO\nconvert_to_yolo_format(train_data, 'train')\nconvert_to_yolo_format(valid_data, 'val')\nconvert_to_yolo_format(test_data, 'test')\n\n# Tạo file cấu hình dữ liệu cho YOLO\ndata_yaml = dict(\n    train=str(YOLO_DIR / 'train/images'),\n    val=str(YOLO_DIR / 'val/images'),\n    nc=3,\n    names=['Normal', 'Aortic enlargement', 'Other abnormality']\n)\n\nwith open(YOLO_DIR / 'data.yaml', 'w') as outfile:\n    yaml.dump(data_yaml, outfile, default_flow_style=False)","metadata":{"execution":{"iopub.status.busy":"2024-08-27T13:29:11.710972Z","iopub.execute_input":"2024-08-27T13:29:11.711890Z","iopub.status.idle":"2024-08-27T13:31:24.024057Z","shell.execute_reply.started":"2024-08-27T13:29:11.711849Z","shell.execute_reply":"2024-08-27T13:31:24.022842Z"}}},{"cell_type":"code","source":"import pandas as pd\nimport os\nfrom pathlib import Path\nfrom sklearn.utils import resample\nfrom sklearn.model_selection import train_test_split\nimport yaml\nfrom shutil import copyfile\n\n# Path to the root directory of the dataset\nROOT = Path(\"/kaggle/input/amia-public-challenge-2024\")\n\n# Read train.csv and img_size.csv\ntrain_df = pd.read_csv(ROOT / \"train.csv\")\nimg_size_df = pd.read_csv(ROOT / \"img_size.csv\")\n\n# Merge the original image size information with the training data\ntrain_df = train_df.merge(img_size_df, on='image_id', how='left')\n\n# Function to map class IDs to new categories and set default bounding box for 'No finding'\ndef map_class(row):\n    if row['class_id'] == 14:  # \"No finding\"\n        row['new_class_id'] = 0\n        row['x_min'] = 0.0\n        row['y_min'] = 0.0\n        row['x_max'] = 1.0\n        row['y_max'] = 1.0\n    elif row['class_id'] == 0:  # \"Aortic enlargement\"\n        row['new_class_id'] = 1\n    else:  # \"Other abnormalities\"\n        row['new_class_id'] = 2\n    return row\n\n# Apply the class mapping function\ntrain_df = train_df.apply(map_class, axis=1)\n\n# Split the data according to the new labels\ndf_class_0 = train_df[train_df['new_class_id'] == 0]  # Normal\ndf_class_1 = train_df[train_df['new_class_id'] == 1]  # Aortic enlargement\ndf_class_2 = train_df[train_df['new_class_id'] == 2]  # Other abnormality\n\n# Downsample class 0 (Normal) to match the number of samples in class 1\ndf_class_0_downsampled = resample(df_class_0, \n                                  replace=False,  # no replacement\n                                  n_samples=len(df_class_1),  # match number of samples in class 1\n                                  random_state=42)\n\n# Balance class 2 by upsampling or downsampling to match class 1\ngrouped_class_2 = df_class_2.groupby('class_id')\nsamples_per_class_2 = len(df_class_1) // len(grouped_class_2)\ndf_class_2_balanced = grouped_class_2.apply(lambda x: x.sample(samples_per_class_2, replace=True, random_state=42))\ndf_class_2_balanced = df_class_2_balanced.reset_index(drop=True)\n\n# Combine the balanced datasets\nbalanced_df = pd.concat([df_class_0_downsampled, df_class_1, df_class_2_balanced])\n\n# Split the data into training, validation, and test sets\ntrain_data, temp_data = train_test_split(balanced_df, test_size=0.3, random_state=42, stratify=balanced_df['new_class_id'])\nvalid_data, test_data = train_test_split(temp_data, test_size=2/3, random_state=42, stratify=temp_data['new_class_id'])\n\n# Paths to YOLOv5 directories\nYOLO_DIR = Path('/kaggle/working/yolov5_data/')\nYOLO_DIR.mkdir(parents=True, exist_ok=True)\n\n# Create YOLOv5 directories for train, validation, and test sets\nfor folder in ['train/images', 'train/labels', 'val/images', 'val/labels', 'test/images', 'test/labels']:\n    (YOLO_DIR / folder).mkdir(parents=True, exist_ok=True)\n\n# Function to convert dataset to YOLO format\ndef convert_to_yolo_format(df, data_type):\n    for i, row in df.iterrows():\n        # Image path and destination\n        img_path = ROOT / f\"train/train/{row['image_id']}.png\"\n        img_dest = YOLO_DIR / f\"{data_type}/images/{row['image_id']}.png\"\n        copyfile(img_path, img_dest)\n        \n        # YOLO label path\n        label_path = YOLO_DIR / f\"{data_type}/labels/{row['image_id']}.txt\"\n        \n        # Normalize bounding box using original dimensions\n        if row['new_class_id'] == 0:  # Normal\n            label_content = \"0 0.5 0.5 1.0 1.0\\n\"\n        else:\n            orig_width, orig_height = row['dim1'], row['dim0']\n            x_center = (row['x_min'] + row['x_max']) / 2 / orig_width\n            y_center = (row['y_min'] + row['y_max']) / 2 / orig_height\n            width = (row['x_max'] - row['x_min']) / orig_width\n            height = (row['y_max'] - row['y_min']) / orig_height\n            \n            # Ensure the coordinates are within bounds [0, 1]\n            if not (0 <= x_center <= 1 and 0 <= y_center <= 1 and 0 <= width <= 1 and 0 <= height <= 1):\n                print(f\"Skipping image {row['image_id']} due to out of bounds coordinates.\")\n                continue\n\n            label_content = f\"{row['new_class_id']} {x_center} {y_center} {width} {height}\\n\"\n        \n        # Write label content to file\n        with open(label_path, 'w') as f:\n            f.write(label_content)\n\n# Convert all datasets to YOLO format\nconvert_to_yolo_format(train_data, 'train')\nconvert_to_yolo_format(valid_data, 'val')\nconvert_to_yolo_format(test_data, 'test')\n\n# Generate YOLO data config file\ndata_yaml = dict(\n    train=str(YOLO_DIR / 'train/images'),\n    val=str(YOLO_DIR / 'val/images'),\n    nc=3,\n    names=['Normal', 'Aortic enlargement', 'Other abnormality']\n)\n\nwith open(YOLO_DIR / 'data.yaml', 'w') as outfile:\n    yaml.dump(data_yaml, outfile, default_flow_style=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Huấn luyện mô hình YOLOv5\n!python train.py --img 640 --batch 16 --epochs 20 --data /kaggle/working/yolov5_data/data.yaml --weights yolov5s.pt --project /kaggle/working/yolov5_training --name yolov5s_xray --exist-ok","metadata":{"execution":{"iopub.status.busy":"2024-08-27T13:35:26.092230Z","iopub.execute_input":"2024-08-27T13:35:26.092991Z","iopub.status.idle":"2024-08-27T13:41:45.713735Z","shell.execute_reply.started":"2024-08-27T13:35:26.092945Z","shell.execute_reply":"2024-08-27T13:41:45.712611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Đánh giá mô hình YOLOv5\n!python val.py --weights /kaggle/working/yolov5_training/yolov5s_xray/weights/best.pt --data /kaggle/working/yolov5_data/data.yaml --img 640\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dự đoán với mô hình YOLOv5\n!python detect.py --weights /kaggle/working/yolov5_training/yolov5s_xray/weights/best.pt --img 640 --conf 0.25 --source /kaggle/working/yolov5_data/test/images\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}