{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":75176,"databundleVersionId":8252256,"sourceType":"competition"}],"dockerImageVersionId":30762,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install ultralytics","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-08-28T00:29:44.361265Z","iopub.execute_input":"2024-08-28T00:29:44.361602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install wandb","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import wandb\n\n# Thiết lập API key cho wandb\nwandb.login(key='4f88ff0bbc6e3485258bca7079d7da4c47798ccd')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom pathlib import Path\nfrom sklearn.utils import resample\nfrom sklearn.model_selection import train_test_split\nimport os\nfrom shutil import copyfile\nimport yaml\n\n# Paths to data directories\nROOT = Path(\"/kaggle/input/amia-public-challenge-2024\")\n\n# Read CSV files\ntrain_df = pd.read_csv(ROOT / \"train.csv\")\nimg_size_df = pd.read_csv(ROOT / \"img_size.csv\")\n\n# Merge the original image size information with the training data\ntrain_df = train_df.merge(img_size_df, on='image_id', how='left')\n\n# Map class IDs to new categories\ndef map_class(row):\n    if row['class_id'] == 14:  # \"No finding\"\n        row['new_class_id'] = 0\n        row['x_min'] = 0.0\n        row['y_min'] = 0.0\n        row['x_max'] = 1.0\n        row['y_max'] = 1.0\n    elif row['class_id'] == 0:  # \"Aortic enlargement\"\n        row['new_class_id'] = 1\n    else:  # \"Other abnormalities\"\n        row['new_class_id'] = 2\n    return row\n\n# Apply the class mapping function\ntrain_df = train_df.apply(map_class, axis=1)\n\n# Balance dataset by downsampling and upsampling\ndf_class_0 = train_df[train_df['new_class_id'] == 0]\ndf_class_1 = train_df[train_df['new_class_id'] == 1]\ndf_class_2 = train_df[train_df['new_class_id'] == 2]\n\n# Downsample class 0 (Normal)\ndf_class_0_downsampled = resample(df_class_0, replace=False, n_samples=len(df_class_1), random_state=42)\n\n# Upsample or downsample class 2 to match class 1\ngrouped_class_2 = df_class_2.groupby('class_id')\nsamples_per_class_2 = len(df_class_1) // len(grouped_class_2)\ndf_class_2_balanced = grouped_class_2.apply(lambda x: x.sample(samples_per_class_2, replace=True, random_state=42))\ndf_class_2_balanced = df_class_2_balanced.reset_index(drop=True)\n\n# Combine the balanced data\nbalanced_df = pd.concat([df_class_0_downsampled, df_class_1, df_class_2_balanced])\n\n# Split dataset into train, validation, and test sets\ntrain_data, temp_data = train_test_split(balanced_df, test_size=0.3, random_state=42, stratify=balanced_df['new_class_id'])\nvalid_data, test_data = train_test_split(temp_data, test_size=2/3, random_state=42, stratify=temp_data['new_class_id'])\n\n# YOLO directory paths\nYOLO_DIR = Path('/kaggle/working/yolov10_data/')\nYOLO_DIR.mkdir(parents=True, exist_ok=True)\n\n# Create necessary directories for YOLO\nfor folder in ['train/images', 'train/labels', 'val/images', 'val/labels', 'test/images', 'test/labels']:\n    (YOLO_DIR / folder).mkdir(parents=True, exist_ok=True)\n\n# Convert dataset to YOLO format\ndef convert_to_yolo_format(df, data_type):\n    for i, row in df.iterrows():\n        # Image path and destination\n        img_path = ROOT / f\"train/train/{row['image_id']}.png\"\n        img_dest = YOLO_DIR / f\"{data_type}/images/{row['image_id']}.png\"\n        copyfile(img_path, img_dest)\n        \n        # YOLO label path\n        label_path = YOLO_DIR / f\"{data_type}/labels/{row['image_id']}.txt\"\n        \n        # Normalize bounding box using original dimensions\n        if row['new_class_id'] == 0:  # Normal\n            label_content = \"0 0.5 0.5 1.0 1.0\\n\"\n        else:\n            orig_width, orig_height = row['dim1'], row['dim0']\n            x_center = (row['x_min'] + row['x_max']) / 2 / orig_width\n            y_center = (row['y_min'] + row['y_max']) / 2 / orig_height\n            width = (row['x_max'] - row['x_min']) / orig_width\n            height = (row['y_max'] - row['y_min']) / orig_height\n            \n            # Ensure the coordinates are within bounds [0, 1]\n            if not (0 <= x_center <= 1 and 0 <= y_center <= 1 and 0 <= width <= 1 and 0 <= height <= 1):\n                print(f\"Skipping image {row['image_id']} due to out of bounds coordinates.\")\n                continue\n\n            label_content = f\"{row['new_class_id']} {x_center} {y_center} {width} {height}\\n\"\n        \n        # Write label content to file\n        with open(label_path, 'w') as f:\n            f.write(label_content)\n\n# Convert all datasets to YOLO format\nconvert_to_yolo_format(train_data, 'train')\nconvert_to_yolo_format(valid_data, 'val')\nconvert_to_yolo_format(test_data, 'test')\n\n# Generate YOLO data config file\ndata_yaml = dict(\n    train=str(YOLO_DIR / 'train/images'),\n    val=str(YOLO_DIR / 'val/images'),\n    nc=3,\n    names=['Normal', 'Aortic enlargement', 'Other abnormality']\n)\n\nwith open(YOLO_DIR / 'data.yaml', 'w') as outfile:\n    yaml.dump(data_yaml, outfile, default_flow_style=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from ultralytics import YOLO\n\n# Load model YOLOv10\nmodel = YOLO('yolov10m.pt')  # Bạn có thể chọn các phiên bản khác như yolov10s.pt, yolov10m.pt, yolov10l.pt, hoặc yolov10x.pt\n\n# Huấn luyện mô hình\nmodel.train(data='/kaggle/working/yolov10_data/data.yaml', epochs=20, batch=16, imgsz=640)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Train the model\n# results = model.train(\n#     data='yolov10_data/meta.yaml',  # Point to the meta.yaml configuration file\n#     epochs=100,\n#     imgsz=640,\n#     batch=32,\n#     name='yolov10_lung_disease_detection',\n#     project='/kaggle/working/yolov10_results',  # Define where to save results\n#     save=True  # Save model checkpoints\n# )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Đánh giá mô hình\nmetrics = model.val()\nprint(metrics)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Optionally, perform predictions on a test set\n# test_images_path = '/kaggle/working/yolov10_data/test/images'\n# results = model.predict(source=test_images_path, save=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dự đoán với mô hình YOLOv8\nresults = model.predict(source='/kaggle/working/yolov10_data/test/images', save=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}