{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Cài thư viện (Kaggle thường đã có sẵn pydicom, opencv)\n!pip install -q pydicom pylibjpeg pylibjpeg-libjpeg pylibjpeg-openjpeg","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\nimport random\nimport yaml\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n\n# ============================== ĐƯỜNG DẪN ==============================\nRSNA_BASE = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\nDICOM_DIR = f'{RSNA_BASE}/stage_2_train_images'\nRSNA_CSV = f'{RSNA_BASE}/stage_2_train_labels.csv'\n\n# Đầu ra\nWORK_DIR = '/kaggle/working/processed'\nPNG_DIR = f'{WORK_DIR}/images_raw'              # Ảnh PNG sau pre-processing (chưa split)\nLABEL_DIR = f'{WORK_DIR}/labels_raw'             # Nhãn YOLO bbox (chưa split)\nDATASET_DIR = WORK_DIR                           # Sẽ chứa images/train, images/val, labels/train, labels/val\nCLINICAL_CSV = f'{WORK_DIR}/rsna_clinical_metadata.csv'\n\n# ============================== THAM SỐ ==============================\nIMG_SIZE = (640, 640)\nORIG_SIZE = 1024  # Ảnh RSNA gốc 1024x1024\nRANDOM_SEED = 42\nVAL_RATIO = 0.2\nNEG_POS_RATIO = 30 / 70  # ca không bệnh / ca bệnh trong tập train\n\nos.makedirs(PNG_DIR, exist_ok=True)\nos.makedirs(LABEL_DIR, exist_ok=True)\nrandom.seed(RANDOM_SEED)\nnp.random.seed(RANDOM_SEED)\n\nprint('✅ Cấu hình xong.')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def dicom_to_xray_png(dicom_path, output_path, size=IMG_SIZE):\n    \"\"\"\n    Convert một file DICOM X-quang sang PNG đã chuẩn hóa.\n    Pipeline: DICOM → VOI LUT → MONOCHROME1 fix → percentile normalize → resize → CLAHE\n    \"\"\"\n    ds = pydicom.dcmread(dicom_path)\n    \n    # 1. Apply VOI LUT (đúng chuẩn cho X-quang)\n    try:\n        img = apply_voi_lut(ds.pixel_array, ds)\n    except Exception:\n        img = ds.pixel_array.astype(np.float32)\n    \n    # 2. Fix MONOCHROME1 (DICOM có giá trị đảo ngược)\n    if hasattr(ds, \"PhotometricInterpretation\") and ds.PhotometricInterpretation == \"MONOCHROME1\":\n        img = np.amax(img) - img\n    \n    img = img.astype(np.float32)\n    \n    # 3. Percentile normalization (loại outlier)\n    p_low, p_high = np.percentile(img, (0.5, 99.5))\n    if p_high - p_low < 1e-6:\n        img_norm = np.zeros_like(img, dtype=np.uint8)\n    else:\n        img_clipped = np.clip(img, p_low, p_high)\n        img_norm = ((img_clipped - p_low) / (p_high - p_low) * 255.0).astype(np.uint8)\n    \n    # 4. Resize trước, CLAHE sau\n    img_resized = cv2.resize(img_norm, size, interpolation=cv2.INTER_AREA)\n    \n    # 5. CLAHE nhẹ\n    clahe = cv2.createCLAHE(clipLimit=1.5, tileGridSize=(8, 8))\n    img_enhanced = clahe.apply(img_resized)\n    \n    cv2.imwrite(output_path, img_enhanced)\n\n\ndef run_preprocessing(input_dir, output_dir):\n    \"\"\"Chạy pre-processing toàn bộ DICOM trong input_dir.\"\"\"\n    files = [f for f in os.listdir(input_dir) if f.endswith(\".dcm\")]\n    print(f\"Tổng số file DICOM: {len(files)}\")\n    \n    skipped, failed = 0, 0\n    for f in tqdm(files, desc=\"DICOM → PNG\"):\n        out_path = os.path.join(output_dir, f.replace(\".dcm\", \".png\"))\n        if os.path.exists(out_path):\n            skipped += 1\n            continue\n        try:\n            dicom_to_xray_png(os.path.join(input_dir, f), out_path)\n        except Exception as e:\n            failed += 1\n            if failed <= 3:\n                print(f\"⚠️ Lỗi {f}: {e}\")\n    \n    print(f\"✅ Hoàn tất. Skip {skipped} (đã có), thất bại {failed}.\")\n\n\nrun_preprocessing(DICOM_DIR, PNG_DIR)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_pngs = random.sample([f for f in os.listdir(PNG_DIR) if f.endswith(\".png\")], 6)\n\nfig, axes = plt.subplots(2, 3, figsize=(15, 10))\nfor ax, fname in zip(axes.flat, sample_pngs):\n    img = cv2.imread(os.path.join(PNG_DIR, fname), cv2.IMREAD_GRAYSCALE)\n    ax.imshow(img, cmap=\"gray\")\n    ax.set_title(f\"{fname[:14]}\\nshape={img.shape}, mean={img.mean():.1f}\", fontsize=10)\n    ax.axis(\"off\")\nplt.suptitle(\"Sample ảnh sau pre-processing (X-ray chuẩn)\", fontsize=14)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_yolo_bbox_labels(csv_path, output_dir, orig_size=ORIG_SIZE):\n    \"\"\"\n    Convert RSNA CSV (x, y, width, height ở pixel space 1024x1024) \n    sang YOLO format (xc, yc, w, h normalized [0,1]).\n    Chỉ tạo file cho ca dương tính (Target=1).\n    \"\"\"\n    df = pd.read_csv(csv_path)\n    df_pos = df[df[\"Target\"] == 1].copy()\n    \n    # Drop dòng có NaN\n    df_pos = df_pos.dropna(subset=[\"x\", \"y\", \"width\", \"height\"])\n    \n    print(f\"Số bounding box dương tính: {len(df_pos)}\")\n    print(f\"Số bệnh nhân dương tính (unique): {df_pos['patientId'].nunique()}\")\n    \n    count_files = 0\n    for pid, group in tqdm(df_pos.groupby(\"patientId\"), desc=\"CSV → YOLO labels\"):\n        lines = []\n        for _, row in group.iterrows():\n            xc = (row[\"x\"] + row[\"width\"] / 2) / orig_size\n            yc = (row[\"y\"] + row[\"height\"] / 2) / orig_size\n            w = row[\"width\"] / orig_size\n            h = row[\"height\"] / orig_size\n            \n            # Sanity: clip [0, 1]\n            xc, yc, w, h = [max(0.0, min(1.0, v)) for v in [xc, yc, w, h]]\n            \n            lines.append(f\"0 {xc:.6f} {yc:.6f} {w:.6f} {h:.6f}\")\n        \n        with open(os.path.join(output_dir, f\"{pid}.txt\"), \"w\") as f:\n            f.write(\"\\n\".join(lines))\n        count_files += 1\n    \n    print(f\"✅ Đã tạo {count_files} file nhãn YOLO bounding box.\")\n\n\ncreate_yolo_bbox_labels(RSNA_CSV, LABEL_DIR)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_severity(boxes, orig_size=ORIG_SIZE):\n    \"\"\"Tính tỷ lệ tổng diện tích box / diện tích ảnh, phân loại severity.\"\"\"\n    if len(boxes) == 0:\n        return 0.0, \"NORMAL\"\n    total_area = sum(row[\"width\"] * row[\"height\"] for _, row in boxes.iterrows())\n    ratio = total_area / (orig_size * orig_size)\n    if ratio < 0.05:\n        return ratio, \"MILD\"\n    elif ratio < 0.15:\n        return ratio, \"MODERATE\"\n    else:\n        return ratio, \"SEVERE\"\n\n\ndef generate_clinical_data(severity_label, target):\n    \"\"\"Sinh triệu chứng tương quan với severity.\"\"\"\n    if target == 0:\n        fever = round(max(35.5, min(41.5, np.random.normal(36.8, 0.3))), 1)\n        wbc = round(max(3.0, min(28.0, np.random.normal(7.0, 1.5))), 1)\n        cough = int(np.random.choice([0, 1], p=[0.9, 0.1]))\n        return fever, wbc, cough, 0, \"NONE\"\n    \n    sub_type = np.random.choice([\"BACTERIA\", \"VIRUS\"], p=[0.6, 0.4])\n    \n    if severity_label == \"MILD\":\n        fever_mu, wbc_mu, dys_p = 37.8, 10.0, [0.8, 0.2]\n    elif severity_label == \"MODERATE\":\n        fever_mu, wbc_mu, dys_p = 38.5, 14.0, [0.4, 0.6]\n    else:  # SEVERE\n        fever_mu, wbc_mu, dys_p = 39.5, 20.0, [0.0, 1.0]\n    \n    fever = round(max(35.5, min(41.5, np.random.normal(fever_mu, 0.6))), 1)\n    wbc = round(max(3.0, min(28.0, np.random.normal(wbc_mu, 3.0))), 1)\n    cough = 1\n    dyspnea = int(np.random.choice([0, 1], p=dys_p))\n    \n    return fever, wbc, cough, dyspnea, sub_type\n\n\ndef build_clinical_csv(rsna_csv, output_csv):\n    df = pd.read_csv(rsna_csv)\n    rows = []\n    \n    for pid, group in tqdm(df.groupby(\"patientId\"), desc=\"Clinical data\"):\n        target = int(group.iloc[0][\"Target\"])\n        boxes = group if target == 1 else pd.DataFrame()\n        ratio, sev = calculate_severity(boxes)\n        fever, wbc, cough, dys, st = generate_clinical_data(sev, target)\n        \n        rows.append({\n            \"PatientID\": pid, \"Target\": target, \"Severity\": sev,\n            \"Area_Ratio\": round(ratio, 4), \"Fever\": fever, \"WBC_Count\": wbc,\n            \"Cough\": cough, \"Dyspnea\": dys, \"Sub_Type\": st\n        })\n    \n    df_out = pd.DataFrame(rows)\n    df_out.to_csv(output_csv, index=False)\n    print(f\"\\n✅ Đã lưu {len(df_out)} dòng vào {output_csv}\")\n    print(\"\\nPhân bố Severity:\")\n    print(df_out[\"Severity\"].value_counts())\n    return df_out\n\n\ndf_clinical = build_clinical_csv(RSNA_CSV, CLINICAL_CSV)\ndf_clinical.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def setup_split_dirs(base_dir):\n    \"\"\"Xóa và tạo lại 4 thư mục cần cho YOLO.\"\"\"\n    for sub in [\"images/train\", \"images/val\", \"labels/train\", \"labels/val\"]:\n        path = os.path.join(base_dir, sub)\n        if os.path.exists(path):\n            shutil.rmtree(path)\n        os.makedirs(path, exist_ok=True)\n\n\ndef split_and_copy(rsna_csv, png_dir, label_dir, base_dir,\n                   val_ratio=VAL_RATIO, neg_pos_ratio=NEG_POS_RATIO):\n    setup_split_dirs(base_dir)\n    \n    df = pd.read_csv(rsna_csv)\n    sick_pids = set(df[df[\"Target\"] == 1][\"patientId\"].unique())\n    \n    all_pngs = [f for f in os.listdir(png_dir) if f.endswith(\".png\")]\n    pos_list = [f for f in all_pngs if f.replace(\".png\", \"\") in sick_pids]\n    neg_list = [f for f in all_pngs if f.replace(\".png\", \"\") not in sick_pids]\n    \n    print(f\"Tổng PNG: {len(all_pngs)} | Bệnh: {len(pos_list)} | Bình thường: {len(neg_list)}\")\n    \n    # Split val\n    p_val = random.sample(pos_list, int(len(pos_list) * val_ratio))\n    n_val = random.sample(neg_list, int(len(neg_list) * val_ratio))\n    p_train = [f for f in pos_list if f not in p_val]\n    n_train_pool = [f for f in neg_list if f not in n_val]\n    n_train_count = int(len(p_train) * neg_pos_ratio)\n    n_train = random.sample(n_train_pool, min(n_train_count, len(n_train_pool)))\n    \n    def deliver(img_list, split_name):\n        pos_count, neg_count = 0, 0\n        for img_name in tqdm(img_list, desc=f\"Copy {split_name}\"):\n            pid = img_name.replace(\".png\", \"\")\n            \n            # Copy ảnh\n            src_img = os.path.join(png_dir, img_name)\n            dst_img = os.path.join(base_dir, \"images\", split_name, img_name)\n            shutil.copy2(src_img, dst_img)\n            \n            # Copy nhãn (nếu là ca bệnh) hoặc tạo file rỗng (ca không bệnh)\n            dst_lbl = os.path.join(base_dir, \"labels\", split_name, f\"{pid}.txt\")\n            src_lbl = os.path.join(label_dir, f\"{pid}.txt\")\n            \n            if pid in sick_pids and os.path.exists(src_lbl):\n                shutil.copy2(src_lbl, dst_lbl)\n                pos_count += 1\n            else:\n                open(dst_lbl, \"w\").close()  # File rỗng = không có object\n                neg_count += 1\n        return pos_count, neg_count\n    \n    tr_p, tr_n = deliver(p_train + n_train, \"train\")\n    vl_p, vl_n = deliver(p_val + n_val, \"val\")\n    \n    print(f\"\\n📊 Train: {tr_p} bệnh / {tr_n} thường (tỷ lệ N/P = {tr_n/(tr_p+1e-6):.2f})\")\n    print(f\"📊 Val:   {vl_p} bệnh / {vl_n} thường (tỷ lệ N/P = {vl_n/(vl_p+1e-6):.2f})\")\n    \n    return {\"train_pos\": tr_p, \"train_neg\": tr_n, \"val_pos\": vl_p, \"val_neg\": vl_n}\n\n\nsplit_stats = split_and_copy(RSNA_CSV, PNG_DIR, LABEL_DIR, DATASET_DIR)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"yaml_path = os.path.join(DATASET_DIR, \"dataset.yaml\")\n\nconfig = {\n    \"path\": DATASET_DIR,\n    \"train\": \"images/train\",\n    \"val\": \"images/val\",\n    \"names\": {0: \"lesion\"}\n}\n\nwith open(yaml_path, \"w\") as f:\n    yaml.dump(config, f, sort_keys=False)\n\nprint(f\"✅ {yaml_path}\")\nwith open(yaml_path) as f:\n    print(f.read())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def visualize_labels(img_dir, lbl_dir, n=20, save_path=\"/kaggle/working/sanity_check.png\"):\n    \"\"\"Visualize ngẫu nhiên n ca bệnh + bounding box.\"\"\"\n    sick_labels = [\n        f for f in os.listdir(lbl_dir)\n        if f.endswith(\".txt\") and os.path.getsize(os.path.join(lbl_dir, f)) > 0\n    ]\n    print(f\"Số ca bệnh trong tập này: {len(sick_labels)}\")\n    \n    if len(sick_labels) < n:\n        n = len(sick_labels)\n    sample = random.sample(sick_labels, n)\n    \n    rows, cols = 4, 5\n    fig, axes = plt.subplots(rows, cols, figsize=(22, 18))\n    \n    for ax, lbl_file in zip(axes.flat, sample):\n        img_path = os.path.join(img_dir, lbl_file.replace(\".txt\", \".png\"))\n        img = cv2.imread(img_path)\n        if img is None:\n            ax.set_title(\"NOT FOUND\", color=\"red\")\n            ax.axis(\"off\")\n            continue\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        h, w = img.shape[:2]\n        \n        box_count = 0\n        with open(os.path.join(lbl_dir, lbl_file)) as f:\n            for line in f:\n                parts = line.strip().split()\n                if len(parts) != 5:\n                    continue\n                _, xc, yc, bw, bh = map(float, parts)\n                x1 = int((xc - bw / 2) * w)\n                y1 = int((yc - bh / 2) * h)\n                x2 = int((xc + bw / 2) * w)\n                y2 = int((yc + bh / 2) * h)\n                cv2.rectangle(img, (x1, y1), (x2, y2), (0, 255, 0), 3)\n                box_count += 1\n        \n        ax.imshow(img)\n        ax.set_title(f\"{lbl_file[:13]} ({box_count} box)\", fontsize=10)\n        ax.axis(\"off\")\n    \n    plt.suptitle(f\"Sanity Check: {n} ca bệnh ngẫu nhiên + bounding box\", fontsize=14, y=1.00)\n    plt.tight_layout()\n    plt.savefig(save_path, dpi=100, bbox_inches=\"tight\")\n    plt.show()\n    print(f\"✅ Đã lưu: {save_path}\")\n\n\nvisualize_labels(\n    img_dir=os.path.join(DATASET_DIR, \"images/train\"),\n    lbl_dir=os.path.join(DATASET_DIR, \"labels/train\"),\n    n=20\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}