{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042}],"dockerImageVersionId":31400,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Notebook 01 — Prepare Data (7 Clinical Features)\n**Ước tính thời gian: ~2–3 giờ**\n\n**Output lưu vào `/kaggle/working/`:**\n- `train_images_png/` — toàn bộ ảnh PNG\n- `splits/train.csv`, `splits/val.csv`, `splits/test.csv`\n- `splits/clinical_features.csv`\n\n> Sau khi chạy xong: **Save version → Save & Run All → Output → New Dataset** → đặt tên `pneumonia-splits`","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport pydicom\nfrom PIL import Image\nfrom pathlib import Path\nfrom sklearn.model_selection import train_test_split\nfrom tqdm.notebook import tqdm\n!pip install -q pylibjpeg pylibjpeg-libjpeg pylibjpeg-openjpeg\n\nRSNA_DIR  = Path('/kaggle/input/competitions/rsna-pneumonia-detection-challenge')\nDCM_DIR   = RSNA_DIR / 'stage_2_train_images'\nLABEL_CSV = RSNA_DIR / 'stage_2_train_labels.csv'\nOUT_DIR   = Path('/kaggle/working')\nPNG_DIR   = OUT_DIR / 'train_images_png'\nSPLIT_DIR = OUT_DIR / 'splits'\n\nPNG_DIR.mkdir(parents=True, exist_ok=True)\nSPLIT_DIR.mkdir(parents=True, exist_ok=True)\nprint('Paths OK')\nprint(f'DICOM files: {len(list(DCM_DIR.glob(\"*.dcm\")))}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-28T00:15:54.636471Z","iopub.execute_input":"2026-05-28T00:15:54.637379Z","iopub.status.idle":"2026-05-28T00:15:58.875497Z","shell.execute_reply.started":"2026-05-28T00:15:54.637335Z","shell.execute_reply":"2026-05-28T00:15:58.874341Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 2: Convert DICOM → PNG (Tối ưu dung lượng) ─────────────────────────\n# Định nghĩa danh sách file DICOM\ndcm_files = sorted(DCM_DIR.glob('*.dcm'))\nprint(f'Converting {len(dcm_files)} DICOM files...')\n\nerrors = [] # Cần khởi tạo danh sách chứa lỗi trước khi chạy vòng lặp\n\nfor dcm_path in tqdm(dcm_files, desc='DICOM→PNG'):\n    png_path = PNG_DIR / f'{dcm_path.stem}.png'\n    if png_path.exists():\n        continue\n    try:\n        ds  = pydicom.dcmread(str(dcm_path))\n        arr = ds.pixel_array.astype(np.float32)\n        lo, hi = arr.min(), arr.max()\n        if hi > lo:\n            arr = (arr - lo) / (hi - lo) * 255\n        \n        # Chuyển thành ảnh xám 'L' để tiết kiệm dung lượng\n        img = Image.fromarray(arr.astype(np.uint8)).convert('L')\n        \n        # Resize xuống 512x512 để tránh bị đầy bộ nhớ đĩa Kaggle\n        img = img.resize((512, 512), Image.Resampling.LANCZOS)\n        img.save(png_path)\n    except Exception as e:\n        errors.append((dcm_path.name, str(e)))\n\npng_count = len(list(PNG_DIR.glob('*.png')))\nprint(f'Done: {png_count} PNGs | Errors: {len(errors)}')\nif errors:\n    print('First 5 errors:', errors[:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-28T00:17:35.514720Z","iopub.execute_input":"2026-05-28T00:17:35.515047Z","iopub.status.idle":"2026-05-28T00:54:47.589467Z","shell.execute_reply.started":"2026-05-28T00:17:35.515021Z","shell.execute_reply":"2026-05-28T00:54:47.588386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 3: Đọc RSNA labels ─────────────────────────────────────────────────\nrsna = pd.read_csv(LABEL_CSV)\nprint(rsna.head())\nprint(f'Total rows: {len(rsna)}')\n\nlabels = rsna.groupby('patientId')['Target'].max().reset_index()\nlabels.columns = ['patientId', 'label']\n\nexisting = {p.stem for p in PNG_DIR.glob('*.png')}\nlabels   = labels[labels['patientId'].isin(existing)].reset_index(drop=True)\n\nprint(f'Patients co PNG: {len(labels)}')\nprint(f'Pneumonia (1): {labels[\"label\"].sum()}')\nprint(f'Normal    (0): {(labels[\"label\"]==0).sum()}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-28T00:55:03.238517Z","iopub.execute_input":"2026-05-28T00:55:03.238923Z","iopub.status.idle":"2026-05-28T00:55:03.494290Z","shell.execute_reply.started":"2026-05-28T00:55:03.238893Z","shell.execute_reply":"2026-05-28T00:55:03.493338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 4: Generate synthetic clinical features — 7 features ───────────────\n# Refs: Khader 2023 (MIMIC), Aksoy 2024 (MIMIC-IV-ED)\n# Clinical-only AUC target: 0.65–0.78  →  model buộc phải dùng cả X-ray\n\nRNG = np.random.default_rng(42)\n\ndef gen_clinical(labels_series):\n    n   = len(labels_series)\n    arr = labels_series.values\n\n    # Temperature (°C) — Thu hẹp khoảng cách trung bình (36.8 vs 37.1)\n    temp = np.where(arr==0,\n        np.clip(RNG.normal(36.8, 0.6, n), 35.5, 39.0),\n        np.clip(RNG.normal(37.1, 0.8, n), 35.5, 40.5))\n\n    # SpO2 (%) — Thu hẹp khoảng cách trung bình (97.0% vs 95.5%)\n    spo2 = np.where(arr==0,\n        np.clip(RNG.normal(97.0, 1.8, n), 86.0, 100.0),\n        np.clip(RNG.normal(95.5, 2.5, n), 82.0, 100.0))\n\n    # Age — Khá tương đồng giữa 2 nhóm (45 tuổi vs 48 tuổi)\n    age = np.where(arr==0,\n        np.clip(RNG.normal(45, 18, n), 1, 90),\n        np.clip(RNG.normal(48, 18, n), 1, 95)).astype(int)\n\n    # Cough (binary) — Giảm chênh lệch tỷ lệ xuất hiện ho (35% vs 45%)\n    cough = (RNG.random(n) < np.where(arr==0, 0.35, 0.42)).astype(int)\n\n    # Heart rate (bpm) — GIẢM mạnh chênh lệch trung bình (từ 17 xuống còn 6)\n    hr = np.where(arr==0,\n        np.clip(RNG.normal(78, 12, n), 50, 110),\n        np.clip(RNG.normal(84, 14, n), 55, 140))\n\n    # Respiratory rate (breaths/min) — GIẢM mạnh chênh lệch (từ 6 xuống còn 2)\n    rr = np.where(arr==0,\n        np.clip(RNG.normal(16, 3, n), 10, 24),\n        np.clip(RNG.normal(18, 4, n), 12, 35))\n\n    # Systolic blood pressure (mmHg) — Đưa về mức gần như trùng nhau (120 vs 118)\n    sbp = np.where(arr==0,\n        np.clip(RNG.normal(120, 15, n), 85, 165),\n        np.clip(RNG.normal(118, 15, n), 75, 160))\n\n    return pd.DataFrame({\n        'patientId':   labels_series.index.tolist(),\n        'label':       arr,\n        'temperature': np.round(temp, 1),\n        'spo2':        np.round(spo2, 1),\n        'age':         age,\n        'cough':       cough,\n        'heart_rate':  np.round(hr, 0).astype(int),\n        'resp_rate':   np.round(rr, 0).astype(int),\n        'sys_bp':      np.round(sbp, 0).astype(int),\n    })\n\n    return pd.DataFrame({\n        'patientId':   labels_series.index.tolist(),\n        'label':       arr,\n        'temperature': np.round(temp, 1),\n        'spo2':        np.round(spo2, 1),\n        'age':         age,\n        'cough':       cough,\n        'heart_rate':  np.round(hr, 0).astype(int),\n        'resp_rate':   np.round(rr, 0).astype(int),\n        'sys_bp':      np.round(sbp, 0).astype(int),\n    })\n\nlabel_series = labels.set_index('patientId')['label']\nclinical_df  = gen_clinical(label_series)\n\n# ── Sanity check ─────────────────────────────────────────────────────────────\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import roc_auc_score\n\nCLINICAL_COLS = ['temperature','spo2','age','cough','heart_rate','resp_rate','sys_bp']\nX     = clinical_df[CLINICAL_COLS].values\ny     = clinical_df['label'].values\nX_n   = (X - X.mean(0)) / (X.std(0) + 1e-8)\nclf   = LogisticRegression(max_iter=500).fit(X_n, y)\nauc_c = roc_auc_score(y, clf.predict_proba(X_n)[:,1])\n\nprint(f'Clinical-only AUC: {auc_c:.4f}  (target: 0.65–0.78)')\nif   auc_c > 0.82: print('WARNING: qua cao — can giam std them')\nelif auc_c < 0.60: print('WARNING: qua thap — tang std them')\nelse:              print('OK — distribution hop ly')\n\nprint()\nprint('Mean by class:')\nprint(clinical_df.groupby('label')[CLINICAL_COLS].mean().round(2).to_string())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-28T01:02:51.545818Z","iopub.execute_input":"2026-05-28T01:02:51.548106Z","iopub.status.idle":"2026-05-28T01:02:51.649616Z","shell.execute_reply.started":"2026-05-28T01:02:51.548036Z","shell.execute_reply":"2026-05-28T01:02:51.648695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 5: Train / Val / Test split 70/15/15 ───────────────────────────────\ntrain_df, temp_df = train_test_split(\n    clinical_df, test_size=0.30,\n    stratify=clinical_df['label'], random_state=42)\n\nval_df, test_df = train_test_split(\n    temp_df, test_size=0.50,\n    stratify=temp_df['label'], random_state=42)\n\nfor name, df in [('train', train_df), ('val', val_df), ('test', test_df)]:\n    df.to_csv(SPLIT_DIR / f'{name}.csv', index=False)\n    p = df['label'].sum()\n    n = (df['label']==0).sum()\n    print(f'{name:5s}: {len(df):5d} samples | Pneumonia={p} ({p/len(df):.1%}) | Normal={n}')\n\nclinical_df.to_csv(SPLIT_DIR / 'clinical_features.csv', index=False)\nprint('\\nAll splits saved.  Notebook 01 DONE.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-28T01:03:55.510462Z","iopub.execute_input":"2026-05-28T01:03:55.511135Z","iopub.status.idle":"2026-05-28T01:03:55.831732Z","shell.execute_reply.started":"2026-05-28T01:03:55.511087Z","shell.execute_reply":"2026-05-28T01:03:55.830869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 6: Verify ──────────────────────────────────────────────────────────\nimport matplotlib.pyplot as plt\n\nsample_id  = train_df.iloc[0]['patientId']\nsample_row = train_df.iloc[0]\nimg = Image.open(PNG_DIR / f'{sample_id}.png')\n\nfig, axes = plt.subplots(1, 2, figsize=(10, 4))\naxes[0].imshow(img, cmap='gray')\naxes[0].set_title(f'Label: {\"Pneumonia\" if sample_row[\"label\"]==1 else \"Normal\"}')\naxes[0].axis('off')\n\naxes[1].bar(['Normal','Pneumonia'],\n            [(clinical_df['label']==0).sum(), clinical_df['label'].sum()],\n            color=['#1D9E75','#D85A30'])\naxes[1].set_title('Class distribution')\naxes[1].set_ylabel('Count')\nplt.tight_layout()\nplt.savefig(OUT_DIR/'data_overview.png', dpi=120)\nplt.show()\n\ncols_show = ['label','temperature','spo2','age','cough','heart_rate','resp_rate','sys_bp']\nprint('Sample row:')\nprint(sample_row[cols_show].to_string())\nprint('\\n=> Publish output lam dataset \"pneumonia-splits\" roi chay NB02')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-28T01:04:58.358158Z","iopub.execute_input":"2026-05-28T01:04:58.358651Z","iopub.status.idle":"2026-05-28T01:04:59.008309Z","shell.execute_reply.started":"2026-05-28T01:04:58.358603Z","shell.execute_reply":"2026-05-28T01:04:59.007201Z"}},"outputs":[],"execution_count":null}]}