{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042,"isSourceIdPinned":false},{"sourceType":"kernelVersion","sourceId":316053622,"isSourceIdPinned":false}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Cài torchxrayvision (chứa pretrained lung seg)\n!pip install -q torchxrayvision scikit-image","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport torch\nimport torchxrayvision as xrv\nimport skimage\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n\n# Đường dẫn (sửa nếu cần)\nPNG_DIR = \"/kaggle/input/notebooks/hiennguyen396/dataprocess/processed/images_raw/\"\nLUNG_MASK_DIR = \"/kaggle/working/processed/lung_masks\"\nLUNG_AREA_CSV = \"/kaggle/working/processed/lung_areas.csv\"\n\nos.makedirs(LUNG_MASK_DIR, exist_ok=True)\n\n# Load pretrained model: PSPNet trained on JSRT + Montgomery\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nprint(f\"Device: {device}\")\n\nmodel = xrv.baseline_models.chestx_det.PSPNet()\nmodel.eval()\nmodel.to(device)\n\nprint(\"✅ Model loaded:\")\nprint(f\"   Targets: {model.targets}\")\nprint(f\"   Input size: 512x512\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def segment_lungs(img_path, model, device, return_mask=True):\n    \"\"\"\n    Segment vùng phổi trên ảnh X-quang.\n    Returns:\n        mask_512: numpy array (512, 512) bool — mask phổi ở size 512\n        lung_area_ratio: float — diện tích phổi / diện tích ảnh (0 đến 1)\n    \"\"\"\n    # 1. Đọc ảnh grayscale\n    img = cv2.imread(img_path, cv2.IMREAD_GRAYSCALE)\n    if img is None:\n        return None, 0.0\n    \n    h_orig, w_orig = img.shape\n    \n    # 2. Chuẩn hóa cho torchxrayvision (yêu cầu range [-1024, 1024])\n    img_norm = xrv.datasets.normalize(img.astype(np.float32), 255)  # [-1024, 1024]\n    \n    # 3. Resize về 512x512 và thêm batch dim\n    img_resized = skimage.transform.resize(img_norm, (512, 512), \n                                            mode=\"constant\", anti_aliasing=False)\n    img_tensor = torch.from_numpy(img_resized).unsqueeze(0).unsqueeze(0).float().to(device)\n    \n    # 4. Inference\n    with torch.no_grad():\n        pred = model(img_tensor)  # shape: (1, 14, 512, 512)\n    \n    # 5. Lấy 2 channel phổi (Left Lung idx=4, Right Lung idx=5 — kiểm tra targets)\n    targets = model.targets\n    left_idx = targets.index(\"Left Lung\")\n    right_idx = targets.index(\"Right Lung\")\n    \n    # Sigmoid → binary mask (threshold 0.5)\n    pred_sigmoid = torch.sigmoid(pred).cpu().numpy()[0]\n    left_mask = pred_sigmoid[left_idx] > 0.5\n    right_mask = pred_sigmoid[right_idx] > 0.5\n    lung_mask_512 = left_mask | right_mask\n    \n    # 6. Tính tỷ lệ diện tích phổi (so với cả ảnh)\n    lung_area_ratio = lung_mask_512.sum() / (512 * 512)\n    \n    if return_mask:\n        # Resize mask về kích thước ảnh gốc để overlay\n        lung_mask_orig = cv2.resize(lung_mask_512.astype(np.uint8), \n                                     (w_orig, h_orig), \n                                     interpolation=cv2.INTER_NEAREST).astype(bool)\n        return lung_mask_orig, lung_area_ratio\n    \n    return lung_mask_512, lung_area_ratio\n\n\n# Test trên 1 ảnh\ntest_imgs = [f for f in os.listdir(PNG_DIR) if f.endswith(\".png\")]\ntest_path = os.path.join(PNG_DIR, test_imgs[0])\nmask, ratio = segment_lungs(test_path, model, device)\nprint(f\"Test ảnh: {test_imgs[0]}\")\nprint(f\"   Lung area ratio (so với ảnh): {ratio*100:.1f}%\")\nprint(f\"   Mask shape: {mask.shape}, dtype: {mask.dtype}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def visualize_lung_masks(img_dir, model, device, n=12, save_path=None):\n    \"\"\"Visualize n ảnh với lung mask overlay.\"\"\"\n    import random\n    random.seed(42)\n    \n    imgs = [f for f in os.listdir(img_dir) if f.endswith(\".png\")]\n    sample = random.sample(imgs, min(n, len(imgs)))\n    \n    rows = 3\n    cols = 4\n    fig, axes = plt.subplots(rows, cols, figsize=(20, 14))\n    \n    ratios = []\n    for ax, fname in zip(axes.flat, sample):\n        img_path = os.path.join(img_dir, fname)\n        mask, ratio = segment_lungs(img_path, model, device)\n        ratios.append(ratio)\n        \n        img = cv2.imread(img_path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        \n        # Overlay green mask\n        overlay = img.copy()\n        overlay[mask] = [0, 255, 0]\n        result = cv2.addWeighted(img, 0.6, overlay, 0.4, 0)\n        \n        ax.imshow(result)\n        ax.set_title(f\"{fname[:13]}\\nLung area = {ratio*100:.1f}% of image\", fontsize=10)\n        ax.axis(\"off\")\n    \n    plt.suptitle(f\"Lung Segmentation Sanity Check (mean lung area: {np.mean(ratios)*100:.1f}%, std: {np.std(ratios)*100:.1f}%)\", \n                 fontsize=13, y=1.00)\n    plt.tight_layout()\n    if save_path:\n        plt.savefig(save_path, dpi=100, bbox_inches=\"tight\")\n    plt.show()\n    \n    return ratios\n\n\nratios = visualize_lung_masks(PNG_DIR, model, device, n=12, \n                               save_path=\"/kaggle/working/lung_seg_check.png\")\nprint(f\"\\n📊 Trên 12 ảnh sample:\")\nprint(f\"   Mean lung area: {np.mean(ratios)*100:.1f}%\")\nprint(f\"   Min: {min(ratios)*100:.1f}%, Max: {max(ratios)*100:.1f}%\")\nprint(f\"   → Hằng số 0.4 (40%) là gần đúng nhưng thiếu cá nhân hóa.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_all_lungs(img_dir, model, device, output_csv, save_masks=False, mask_dir=None):\n    \"\"\"\n    Chạy lung seg cho toàn bộ ảnh, lưu lung_area_ratio vào CSV.\n    Nếu save_masks=True, lưu mask binary (.npz) vào mask_dir.\n    \"\"\"\n    img_files = [f for f in os.listdir(img_dir) if f.endswith(\".png\")]\n    print(f\"Processing {len(img_files)} images...\")\n    \n    rows = []\n    failed = 0\n    \n    for fname in tqdm(img_files, desc=\"Lung segmentation\"):\n        img_path = os.path.join(img_dir, fname)\n        try:\n            # Chỉ cần mask 512 cho việc tính tỷ lệ — nhanh hơn nhiều\n            mask_512, ratio = segment_lungs(img_path, model, device, return_mask=False)\n            \n            patient_id = fname.replace(\".png\", \"\")\n            rows.append({\n                \"PatientID\": patient_id,\n                \"Lung_Area_Ratio\": round(ratio, 4),  # so với toàn ảnh\n                \"Lung_Pixels_512\": int(mask_512.sum())  # px count ở size 512x512\n            })\n            \n            # Lưu mask nếu cần (cho XAI sau này)\n            if save_masks and mask_dir:\n                np.savez_compressed(os.path.join(mask_dir, f\"{patient_id}.npz\"), \n                                    mask=mask_512.astype(np.uint8))\n        except Exception as e:\n            failed += 1\n            if failed <= 3:\n                print(f\"⚠️ Lỗi {fname}: {e}\")\n    \n    df = pd.DataFrame(rows)\n    df.to_csv(output_csv, index=False)\n    \n    print(f\"\\n✅ Đã xử lý {len(df)} ảnh ({failed} lỗi)\")\n    print(f\"📊 Lung Area Ratio statistics:\")\n    print(df[\"Lung_Area_Ratio\"].describe())\n    \n    return df\n\n\ndf_lung = process_all_lungs(PNG_DIR, model, device, LUNG_AREA_CSV, \n                             save_masks=False)  # Đặt True nếu muốn lưu mask cho Grad-CAM sau\ndf_lung.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_severity_v2(boxes_df_row_or_lines, lung_area_ratio, orig_size=1024):\n    \"\"\"\n    Severity scoring v2: dùng diện tích phổi thật.\n    \n    Args:\n        boxes_df_row_or_lines: \n            - Nếu là DataFrame group (từ RSNA CSV): cột x, y, width, height (pixel space 1024)\n            - Nếu là list các tuple (xc, yc, w, h) đã chuẩn hóa [0,1] (YOLO format)\n        lung_area_ratio: float — tỷ lệ phổi/ảnh đã tính từ pretrained model\n        orig_size: kích thước ảnh gốc\n    \n    Returns:\n        (R, severity_label)\n        R: tỷ lệ vùng tổn thương / vùng phổi (0 đến 1)\n    \"\"\"\n    if isinstance(boxes_df_row_or_lines, pd.DataFrame):\n        if len(boxes_df_row_or_lines) == 0:\n            return 0.0, \"NORMAL\"\n        # Tổng diện tích box (pixel space) → tỷ lệ so với ảnh\n        total_lesion_pixels = sum(\n            row[\"width\"] * row[\"height\"] \n            for _, row in boxes_df_row_or_lines.iterrows()\n        )\n        lesion_ratio_to_img = total_lesion_pixels / (orig_size * orig_size)\n    else:\n        # YOLO format: list of (xc, yc, w, h) normalized\n        if len(boxes_df_row_or_lines) == 0:\n            return 0.0, \"NORMAL\"\n        lesion_ratio_to_img = sum(w * h for (_, _, w, h) in boxes_df_row_or_lines)\n    \n    # Avoid division by zero\n    if lung_area_ratio < 0.01:\n        return 0.0, \"ERROR_NO_LUNG\"\n    \n    # R = lesion / lung\n    R = lesion_ratio_to_img / lung_area_ratio\n    R = min(R, 1.0)  # Clip về [0, 1]\n    \n    # Phân ngưỡng dựa trên y khoa\n    if R < 0.25:\n        severity = \"MILD\"\n    elif R < 0.50:\n        severity = \"MODERATE\"\n    else:\n        severity = \"SEVERE\"\n    \n    return R, severity\n\n\n# Test với một bệnh nhân từ RSNA CSV\nRSNA_CSV = \"/kaggle/input/competitions/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv\"\ndf_rsna = pd.read_csv(RSNA_CSV)\ndf_lung_lookup = df_lung.set_index(\"PatientID\")[\"Lung_Area_Ratio\"].to_dict()\n\n# Demo: 5 bệnh nhân dương tính\ndf_pos = df_rsna[df_rsna[\"Target\"] == 1].dropna()\nsample_pids = df_pos[\"patientId\"].unique()[:5]\n\nprint(f\"{'PatientID':<25} {'Lesion%img':<12} {'Lung%img':<12} {'R (lesion/lung)':<18} {'Severity':<10}\")\nprint(\"-\" * 85)\n\nfor pid in sample_pids:\n    boxes = df_pos[df_pos[\"patientId\"] == pid]\n    lung_ratio = df_lung_lookup.get(pid, 0.4)\n    \n    lesion_pixels = sum(row[\"width\"] * row[\"height\"] for _, row in boxes.iterrows())\n    lesion_ratio_img = lesion_pixels / (1024 * 1024)\n    \n    R, sev = calculate_severity_v2(boxes, lung_ratio)\n    \n    print(f\"{pid[:24]:<25} {lesion_ratio_img*100:>8.1f}%    {lung_ratio*100:>8.1f}%    {R*100:>10.1f}%        {sev}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_clinical_csv_v2(rsna_csv, lung_csv, output_csv):\n    \"\"\"Build clinical CSV mới với severity dựa trên diện tích phổi thật.\"\"\"\n    df_rsna = pd.read_csv(rsna_csv)\n    df_lung = pd.read_csv(lung_csv).set_index(\"PatientID\")\n    \n    rows = []\n    for pid, group in tqdm(df_rsna.groupby(\"patientId\"), desc=\"Building v2 CSV\"):\n        target = int(group.iloc[0][\"Target\"])\n        lung_ratio = df_lung.loc[pid, \"Lung_Area_Ratio\"] if pid in df_lung.index else 0.4\n        \n        if target == 1:\n            boxes_clean = group.dropna(subset=[\"x\", \"y\", \"width\", \"height\"])\n            R, sev = calculate_severity_v2(boxes_clean, lung_ratio)\n            lesion_pixels = sum(r[\"width\"] * r[\"height\"] for _, r in boxes_clean.iterrows())\n            lesion_ratio_img = lesion_pixels / (1024 * 1024)\n        else:\n            R, sev = 0.0, \"NORMAL\"\n            lesion_ratio_img = 0.0\n        \n        # Sinh dữ liệu lâm sàng (giữ logic cũ, chỉ thay severity input)\n        if target == 0:\n            fever = round(max(35.5, min(41.5, np.random.normal(36.8, 0.3))), 1)\n            wbc = round(max(3.0, min(28.0, np.random.normal(7.0, 1.5))), 1)\n            cough = int(np.random.choice([0, 1], p=[0.9, 0.1]))\n            dys = 0\n            sub_type = \"NONE\"\n        else:\n            sub_type = np.random.choice([\"BACTERIA\", \"VIRUS\"], p=[0.6, 0.4])\n            if sev == \"MILD\":\n                fever_mu, wbc_mu, dys_p = 37.8, 10.0, [0.8, 0.2]\n            elif sev == \"MODERATE\":\n                fever_mu, wbc_mu, dys_p = 38.5, 14.0, [0.4, 0.6]\n            else:\n                fever_mu, wbc_mu, dys_p = 39.5, 20.0, [0.0, 1.0]\n            fever = round(max(35.5, min(41.5, np.random.normal(fever_mu, 0.6))), 1)\n            wbc = round(max(3.0, min(28.0, np.random.normal(wbc_mu, 3.0))), 1)\n            cough = 1\n            dys = int(np.random.choice([0, 1], p=dys_p))\n        \n        rows.append({\n            \"PatientID\": pid,\n            \"Target\": target,\n            \"Lesion_Area_Img_Ratio\": round(lesion_ratio_img, 4),  # vùng tổn thương / ảnh\n            \"Lung_Area_Img_Ratio\": round(lung_ratio, 4),           # phổi / ảnh\n            \"R_Lesion_Lung\": round(R, 4),                           # vùng tổn thương / phổi (mới)\n            \"Severity_v2\": sev,                                      # severity dựa trên R mới\n            \"Fever\": fever,\n            \"WBC_Count\": wbc,\n            \"Cough\": cough,\n            \"Dyspnea\": dys,\n            \"Sub_Type\": sub_type\n        })\n    \n    df_out = pd.DataFrame(rows)\n    df_out.to_csv(output_csv, index=False)\n    \n    print(f\"\\n✅ Đã lưu {output_csv}\")\n    print(f\"\\nPhân bố Severity_v2:\")\n    print(df_out[\"Severity_v2\"].value_counts())\n    print(f\"\\nPhân bố R (vùng tổn thương / phổi) cho ca dương:\")\n    print(df_out[df_out[\"Target\"] == 1][\"R_Lesion_Lung\"].describe())\n    \n    return df_out\n\n\nCLINICAL_CSV_V2 = \"/kaggle/working/processed/rsna_clinical_metadata_v2.csv\"\ndf_clinical_v2 = build_clinical_csv_v2(RSNA_CSV, LUNG_AREA_CSV, CLINICAL_CSV_V2)\ndf_clinical_v2.head(10)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# So sánh: severity tính theo cách cũ (hardcode 0.4) vs cách mới (lung mask thật)\ndf_compare = df_clinical_v2[df_clinical_v2[\"Target\"] == 1].copy()\n\n# Tính severity cũ (cùng công thức cũ với hardcode 0.4)\ndef severity_old(lesion_ratio_img):\n    if lesion_ratio_img == 0:\n        return \"NORMAL\"\n    R_old = lesion_ratio_img / 0.4  # hardcode\n    R_old = min(R_old, 1.0)\n    if R_old < 0.25:\n        return \"MILD\"\n    elif R_old < 0.50:\n        return \"MODERATE\"\n    else:\n        return \"SEVERE\"\n\n\ndf_compare[\"Severity_Old\"] = df_compare[\"Lesion_Area_Img_Ratio\"].apply(severity_old)\n\n# Cross-tab so sánh\nct = pd.crosstab(df_compare[\"Severity_Old\"], df_compare[\"Severity_v2\"], \n                  rownames=[\"Cũ (0.4)\"], colnames=[\"Mới (lung mask)\"])\nprint(\"Cross-tab phân bố severity:\")\nprint(ct)\n\n# Số ca thay đổi label\nchanged = (df_compare[\"Severity_Old\"] != df_compare[\"Severity_v2\"]).sum()\ntotal = len(df_compare)\nprint(f\"\\n📊 {changed}/{total} ca ({100*changed/total:.1f}%) thay đổi label severity sau khi dùng lung mask thật.\")\n\n# Visualize: distribution of lung area ratio\nfig, axes = plt.subplots(1, 2, figsize=(15, 5))\n\naxes[0].hist(df_clinical_v2[\"Lung_Area_Img_Ratio\"], bins=50, color=\"steelblue\", edgecolor=\"white\")\naxes[0].axvline(0.4, color=\"red\", linestyle=\"--\", linewidth=2, label=\"Hardcode cũ = 0.4\")\naxes[0].set_xlabel(\"Lung area / Image area\")\naxes[0].set_ylabel(\"Số bệnh nhân\")\naxes[0].set_title(\"Phân bố diện tích phổi thật\\n(Hằng số 0.4 không khớp thực tế)\")\naxes[0].legend()\n\n# Severity distribution: old vs new\nsev_order = [\"MILD\", \"MODERATE\", \"SEVERE\"]\nold_counts = [df_compare[\"Severity_Old\"].value_counts().get(s, 0) for s in sev_order]\nnew_counts = [df_compare[\"Severity_v2\"].value_counts().get(s, 0) for s in sev_order]\n\nx = np.arange(len(sev_order))\nwidth = 0.35\naxes[1].bar(x - width/2, old_counts, width, label=\"Cũ (hardcode 0.4)\", color=\"lightcoral\")\naxes[1].bar(x + width/2, new_counts, width, label=\"Mới (lung mask thật)\", color=\"seagreen\")\naxes[1].set_xticks(x)\naxes[1].set_xticklabels(sev_order)\naxes[1].set_ylabel(\"Số bệnh nhân\")\naxes[1].set_title(\"Phân bố Severity: Cũ vs Mới\")\naxes[1].legend()\n\nplt.tight_layout()\nplt.savefig(\"/kaggle/working/severity_old_vs_new.png\", dpi=100)\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}