{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Step 1: Setup & Data Acquisition","metadata":{}},{"cell_type":"markdown","source":"## Install Needed Libraries","metadata":{}},{"cell_type":"code","source":"# Install all required third-party libraries:\n# - pydicom: read DICOM medical image files\n# - scikit-image: image processing utilities (GLCM texture features, region props)\n# - scikit-learn: ML models, preprocessing, evaluation metrics\n# - xgboost: gradient boosting classifier\n# - imbalanced-learn: SMOTE oversampling to handle class imbalance\n# - albumentations: fast image augmentation library\n!pip install -q pydicom scikit-image scikit-learn \\\n               xgboost imbalanced-learn albumentations","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:33:37.718924Z","iopub.execute_input":"2026-05-08T20:33:37.719232Z","iopub.status.idle":"2026-05-08T20:33:43.605068Z","shell.execute_reply.started":"2026-05-08T20:33:37.719201Z","shell.execute_reply":"2026-05-08T20:33:43.603399Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Set Path","metadata":{}},{"cell_type":"code","source":"import os\n\n# Root directory of the RSNA Pneumonia Detection Challenge dataset on Kaggle\nDATA_DIR  = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\n\n# Directory containing all training DICOM (.dcm) image files\nTRAIN_DIR = f'{DATA_DIR}/stage_2_train_images'\n\n# Directory where all output files (CSVs, plots, models) will be saved\nOUTPUT_DIR = '/kaggle/working'\n\n# List the top-level contents of the dataset folder to verify paths are correct\nprint('📁 Dataset contents:')\nfor item in sorted(os.listdir(DATA_DIR)):\n    print(f'   {item}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:33:43.607958Z","iopub.execute_input":"2026-05-08T20:33:43.610288Z","iopub.status.idle":"2026-05-08T20:33:43.619104Z","shell.execute_reply.started":"2026-05-08T20:33:43.610223Z","shell.execute_reply":"2026-05-08T20:33:43.617962Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load and Preview Data","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# Load the training labels CSV.\n# Each row = one bounding box annotation for a pneumonia region.\n# Normal patients (Target=0) have a single row with NaN for x/y/width/height.\n# Pneumonia patients (Target=1) may have multiple rows (one per detected region).\ndf = pd.read_csv(f'{DATA_DIR}/stage_2_train_labels.csv')\nprint('Shape:', df.shape)\nprint(df.head(8))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:33:43.620231Z","iopub.execute_input":"2026-05-08T20:33:43.620593Z","iopub.status.idle":"2026-05-08T20:33:44.165467Z","shell.execute_reply.started":"2026-05-08T20:33:43.620552Z","shell.execute_reply":"2026-05-08T20:33:44.164424Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Class Distribution","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Collapse the multi-row annotations to one row per patient by taking the max Target.\n# This ensures each patient is counted once regardless of how many bounding boxes they have.\npatients = df.groupby('patientId')['Target'].max().reset_index()\ncounts   = patients['Target'].value_counts()\n\n# Print raw counts and percentages for each class\nprint(f'Normal    (0): {counts[0]:,} ({counts[0]/len(patients)*100:.1f}%)')\nprint(f'Pneumonia (1): {counts[1]:,} ({counts[1]/len(patients)*100:.1f}%)')\nprint(f'Total        : {len(patients):,}')\n\n# Bar chart to visually confirm class imbalance\nfig, ax = plt.subplots(figsize=(5,3))\nax.bar(['Normal','Pneumonia'], [counts[0], counts[1]],\n       color=['steelblue','tomato'], edgecolor='white', width=0.5)\nax.set_title('Class Distribution — RSNA Dataset')\nax.set_ylabel('Patients')\n# Annotate exact counts on top of each bar\nfor i, v in enumerate([counts[0], counts[1]]):\n    ax.text(i, v+50, f'{v:,}', ha='center', fontweight='bold')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:33:44.166856Z","iopub.execute_input":"2026-05-08T20:33:44.167467Z","iopub.status.idle":"2026-05-08T20:33:44.440511Z","shell.execute_reply.started":"2026-05-08T20:33:44.167433Z","shell.execute_reply":"2026-05-08T20:33:44.439436Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualize Sample X-Rays with Bounding Boxes","metadata":{}},{"cell_type":"code","source":"import pydicom\nimport numpy as np\nimport matplotlib.patches as patches\n\ndef apply_window(img, level=-500, width=1500):\n    \"\"\"\n    Apply lung windowing to a raw CT/X-ray pixel array.\n    \n    Lung windowing (level=-500 HU, width=1500 HU) suppresses bone and\n    soft tissue, making pulmonary structures much more visible.\n    \n    Args:\n        img   : raw float32 pixel array\n        level : window center in Hounsfield Units (HU)\n        width : window width in HU\n    Returns:\n        img   : float32 array clipped and rescaled to [0, 1]\n    \"\"\"\n    low  = level - width // 2   # Lower bound of the HU window\n    high = level + width // 2   # Upper bound of the HU window\n    img  = np.clip(img, low, high)          # Clip pixels outside the window\n    img  = (img - low) / (high - low)       # Rescale to [0, 1]\n    return img\n\n# Create a 2x4 grid: top row = 4 Normal, bottom row = 4 Pneumonia samples\nfig, axes = plt.subplots(2, 4, figsize=(16, 8))\naxes = axes.flatten()\n\n# Randomly sample 4 normal and 4 pneumonia patient IDs\nnormal_ids    = patients[patients['Target']==0]['patientId'].sample(4, random_state=42).values\npneumonia_ids = patients[patients['Target']==1]['patientId'].sample(4, random_state=42).values\nsample_ids    = list(normal_ids) + list(pneumonia_ids)\nlabels_list   = ['Normal']*4 + ['Pneumonia']*4\n\nfor idx, (pid, label) in enumerate(zip(sample_ids, labels_list)):\n    # Read the DICOM file and apply lung windowing for better visibility\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = apply_window(dcm.pixel_array.astype(np.float32))\n\n    axes[idx].imshow(img, cmap='gray')\n    axes[idx].set_title(label,\n                        color='tomato' if label=='Pneumonia' else 'steelblue',\n                        fontweight='bold')\n    axes[idx].axis('off')\n\n    # For pneumonia cases, overlay all annotated bounding boxes in red\n    if label == 'Pneumonia':\n        for _, row in df[df['patientId']==pid][['x','y','width','height']].dropna().iterrows():\n            axes[idx].add_patch(patches.Rectangle(\n                (row['x'], row['y']), row['width'], row['height'],\n                linewidth=2, edgecolor='red', facecolor='none'))\n\nplt.suptitle('Sample Chest X-Rays — red boxes = pneumonia regions', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:33:44.443101Z","iopub.execute_input":"2026-05-08T20:33:44.443399Z","iopub.status.idle":"2026-05-08T20:33:47.858922Z","shell.execute_reply.started":"2026-05-08T20:33:44.443371Z","shell.execute_reply":"2026-05-08T20:33:47.857842Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Save Global Config","metadata":{}},{"cell_type":"code","source":"import json\n\n# Re-declare paths (needed if this cell is run independently)\nDATA_DIR  = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\nTRAIN_DIR = f'{DATA_DIR}/stage_2_train_images'\nOUTPUT_DIR = '/kaggle/working'\n\n# Central config dictionary used throughout the project.\n# Keeping all hyperparameters/paths in one place makes experiments easier to reproduce.\nCONFIG = {\n    'data_dir'    : DATA_DIR,\n    'train_dcm'   : TRAIN_DIR,\n    'labels_csv'  : f'{DATA_DIR}/stage_2_train_labels.csv',\n    'output_dir'  : OUTPUT_DIR,\n    'img_size'    : 512,          # All images are resized to 512x512 pixels\n    'window_level': -500,         # HU center for lung windowing\n    'window_width': 1500,         # HU width for lung windowing\n    'train_ratio' : 0.70,         # 70% of patients used for training\n    'val_ratio'   : 0.15,         # 15% used for validation / hyperparameter tuning\n    'test_ratio'  : 0.15,         # 15% held out for final evaluation\n    'random_seed' : 42,           # Fixed seed for reproducibility\n}\n\nprint('✅ Config ready.')\nprint(json.dumps(CONFIG, indent=2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:33:47.860140Z","iopub.execute_input":"2026-05-08T20:33:47.860442Z","iopub.status.idle":"2026-05-08T20:33:47.868190Z","shell.execute_reply.started":"2026-05-08T20:33:47.860413Z","shell.execute_reply":"2026-05-08T20:33:47.867032Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 2: EDA & Visualization","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"# Standard scientific and image libraries\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport pydicom\nimport cv2\nimport os\nfrom tqdm import tqdm   # Progress bar for long loops\n\n# Dataset paths\nDATA_DIR  = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\nTRAIN_DIR = f'{DATA_DIR}/stage_2_train_images'\n\n# Load two CSVs:\n# df       — bounding boxes + binary Target label per patient\n# df_class — detailed 3-class labels: 'Normal', 'No Lung Opacity / Not Normal', 'Lung Opacity'\ndf       = pd.read_csv(f'{DATA_DIR}/stage_2_train_labels.csv')\ndf_class = pd.read_csv(f'{DATA_DIR}/stage_2_detailed_class_info.csv')\n\n# One row per patient with the max Target (handles multi-box pneumonia rows)\npatients = df.groupby('patientId')['Target'].max().reset_index()\n\nprint('Labels shape    :', df.shape)\nprint('Class info shape:', df_class.shape)\nprint(df_class['class'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:33:47.869366Z","iopub.execute_input":"2026-05-08T20:33:47.869750Z","iopub.status.idle":"2026-05-08T20:33:48.410355Z","shell.execute_reply.started":"2026-05-08T20:33:47.869705Z","shell.execute_reply":"2026-05-08T20:33:48.408984Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Missing Values","metadata":{}},{"cell_type":"code","source":"# ── Labels CSV ──\n# Normal patients have no bounding box, so x/y/width/height will be NaN.\n# This is expected behavior, NOT a data quality issue.\nprint('=== Labels CSV ===')\nprint(df.isnull().sum())\nprint(f'\\nTotal missing: {df.isnull().sum().sum()}')\n\n# ── Class Info CSV ──\n# Should have no missing values (every patient has exactly one class label)\nprint('\\n=== Class Info CSV ===')\nprint(df_class.isnull().sum())\nprint(f'\\nTotal missing: {df_class.isnull().sum().sum()}')\n\n# Confirm how many rows are normal (NaN boxes) vs pneumonia (have boxes)\nprint('\\n=== Missing Bounding Boxes (Normal patients have NaN boxes) ===')\nprint(f'Rows with NaN boxes : {df[df[\"x\"].isnull()].shape[0]:,}')\nprint(f'Rows with boxes     : {df[df[\"x\"].notna()].shape[0]:,}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:33:48.411489Z","iopub.execute_input":"2026-05-08T20:33:48.411754Z","iopub.status.idle":"2026-05-08T20:33:48.439480Z","shell.execute_reply.started":"2026-05-08T20:33:48.411728Z","shell.execute_reply":"2026-05-08T20:33:48.438142Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Patient Age & Gender Distribution","metadata":{}},{"cell_type":"code","source":"# Deduplicate to one row per patient in the detailed class CSV\ndetail_df = df_class.drop_duplicates('patientId').copy()\n\n# Verify which demographic columns are available before trying to use them\nprint('Available columns:', detail_df.columns.tolist())\nprint(detail_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:33:48.440701Z","iopub.execute_input":"2026-05-08T20:33:48.441083Z","iopub.status.idle":"2026-05-08T20:33:48.456394Z","shell.execute_reply.started":"2026-05-08T20:33:48.440971Z","shell.execute_reply":"2026-05-08T20:33:48.454947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\n\nrecords = []\n\n# Sample 300 patients for a fast demographic overview\n# (reading all ~26k DICOMs would be too slow just for metadata)\nsample_pids = patients.sample(300, random_state=42)['patientId'].values\n\nfor pid in tqdm(sample_pids):\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    target = patients[patients['patientId']==pid]['Target'].values[0]\n    \n    # Use getattr with a default of None — some DICOM files omit these tags\n    age    = getattr(dcm, 'PatientAge',  None)\n    gender = getattr(dcm, 'PatientSex',  None)\n    \n    records.append({\n        'patientId': pid,\n        'age'      : age,\n        'gender'   : gender,\n        'target'   : target\n    })\n\ndf_demo = pd.DataFrame(records)\nprint('Sample demographics:')\nprint(df_demo.head(10))\nprint('\\nGender counts:')\nprint(df_demo['gender'].value_counts())\nprint('\\nAge sample:')\nprint(df_demo['age'].value_counts().head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:33:48.457720Z","iopub.execute_input":"2026-05-08T20:33:48.458107Z","iopub.status.idle":"2026-05-08T20:33:51.947387Z","shell.execute_reply.started":"2026-05-08T20:33:48.458073Z","shell.execute_reply":"2026-05-08T20:33:51.946391Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Plot Age & Gender","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize=(16, 4))\n\n# ── Plot 1: Overall gender distribution ──\ngender_counts = df_demo['gender'].value_counts()\naxes[0].bar(gender_counts.index, gender_counts.values,\n            color=['steelblue','tomato'], edgecolor='white', width=0.4)\naxes[0].set_title('Patient Gender Distribution')\naxes[0].set_ylabel('Count')\naxes[0].set_xlabel('Gender')\nfor i, v in enumerate(gender_counts.values):\n    axes[0].text(i, v+1, f'{v:,}', ha='center', fontweight='bold')\n\n# ── Plot 2: Age distribution broken down by class ──\n# Convert age strings (e.g. '045Y') to numeric; invalid entries become NaN\ndf_demo['age'] = pd.to_numeric(df_demo['age'], errors='coerce')\nnormal_ages    = df_demo[df_demo['target']==0]['age'].dropna()\npneumonia_ages = df_demo[df_demo['target']==1]['age'].dropna()\n\naxes[1].hist(normal_ages,    bins=20, alpha=0.7, color='steelblue', label='Normal')\naxes[1].hist(pneumonia_ages, bins=20, alpha=0.7, color='tomato',    label='Pneumonia')\naxes[1].set_title('Age Distribution by Class')\naxes[1].set_xlabel('Age (years)')\naxes[1].set_ylabel('Count')\naxes[1].legend()\n\n# ── Plot 3: Gender breakdown per class (to spot demographic bias) ──\ngender_class = df_demo.groupby(['gender','target']).size().unstack(fill_value=0)\ngender_class.plot(kind='bar', ax=axes[2],\n                  color=['steelblue','tomato'], edgecolor='white')\naxes[2].set_title('Gender by Class')\naxes[2].set_xlabel('Gender')\naxes[2].set_ylabel('Count')\naxes[2].legend(['Normal','Pneumonia'])\naxes[2].tick_params(axis='x', rotation=0)\n\nplt.suptitle('Patient Demographics — RSNA Dataset', fontsize=13)\nplt.tight_layout()\nplt.show()\n\n# Print summary statistics\nprint(f'Male   — Total: {gender_counts.get(\"M\",0):,}')\nprint(f'Female — Total: {gender_counts.get(\"F\",0):,}')\nprint(f'\\nNormal    — Mean age: {normal_ages.mean():.1f} yrs')\nprint(f'Pneumonia — Mean age: {pneumonia_ages.mean():.1f} yrs')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:33:51.948330Z","iopub.execute_input":"2026-05-08T20:33:51.948625Z","iopub.status.idle":"2026-05-08T20:33:52.552193Z","shell.execute_reply.started":"2026-05-08T20:33:51.948597Z","shell.execute_reply":"2026-05-08T20:33:52.551205Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Class Distribution","metadata":{}},{"cell_type":"code","source":"# Use the detailed 3-class labels (more informative than the binary Target)\n# Classes: 'Normal', 'No Lung Opacity / Not Normal', 'Lung Opacity'\nclass_counts = df_class.drop_duplicates('patientId')['class'].value_counts()\n\nfig, axes = plt.subplots(1, 2, figsize=(12, 4))\n\n# Bar chart — easy to compare absolute counts across classes\naxes[0].bar(class_counts.index, class_counts.values,\n            color=['steelblue', 'tomato', 'orange'], edgecolor='white')\naxes[0].set_title('Detailed Class Distribution')\naxes[0].set_ylabel('Number of Patients')\naxes[0].tick_params(axis='x', rotation=15)\nfor i, v in enumerate(class_counts.values):\n    axes[0].text(i, v+50, f'{v:,}', ha='center', fontweight='bold')\n\n# Pie chart — shows proportional class split at a glance\naxes[1].pie(class_counts.values, labels=class_counts.index,\n            colors=['steelblue', 'tomato', 'orange'],\n            autopct='%1.1f%%', startangle=90)\naxes[1].set_title('Class Proportions')\n\nplt.suptitle('RSNA Dataset — Class Distribution', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:33:52.553344Z","iopub.execute_input":"2026-05-08T20:33:52.553675Z","iopub.status.idle":"2026-05-08T20:33:52.786080Z","shell.execute_reply.started":"2026-05-08T20:33:52.553643Z","shell.execute_reply":"2026-05-08T20:33:52.784703Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Bounding Box Statistics","metadata":{}},{"cell_type":"code","source":"# Keep only pneumonia rows (bounding boxes exist only for Target==1)\ndf_pos = df[df['Target'] == 1].copy()\n\nprint('Total bounding boxes      :', len(df_pos))\nprint('Patients with pneumonia   :', df_pos['patientId'].nunique())\n# Average boxes per patient tells us how multi-focal the disease tends to be\nprint('Avg boxes per patient     :', round(len(df_pos)/df_pos['patientId'].nunique(), 2))\nprint()\nprint('Bounding Box Size Stats:')\nprint(df_pos[['width','height']].describe().round(1))\n\n# Count how many bounding boxes each pneumonia patient has\nbox_counts = df_pos.groupby('patientId').size()\n\nfig, axes = plt.subplots(1, 3, figsize=(15, 4))\n\n# Plot 1: How many regions per patient — reveals multi-focal pneumonia frequency\naxes[0].hist(box_counts.values, bins=range(1, box_counts.max()+2),\n             color='tomato', edgecolor='white', align='left')\naxes[0].set_title('Number of Boxes per Pneumonia Patient')\naxes[0].set_xlabel('Number of Bounding Boxes')\naxes[0].set_ylabel('Count')\n\n# Plot 2: Width spread — large variance means pneumonia region sizes vary a lot\naxes[1].hist(df_pos['width'].dropna(), bins=40, color='steelblue', edgecolor='white')\naxes[1].set_title('Bounding Box Width Distribution')\naxes[1].set_xlabel('Width (pixels)')\naxes[1].set_ylabel('Count')\n\n# Plot 3: Height spread — similar insight as width\naxes[2].hist(df_pos['height'].dropna(), bins=40, color='purple', edgecolor='white')\naxes[2].set_title('Bounding Box Height Distribution')\naxes[2].set_xlabel('Height (pixels)')\naxes[2].set_ylabel('Count')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:33:52.787625Z","iopub.execute_input":"2026-05-08T20:33:52.787934Z","iopub.status.idle":"2026-05-08T20:33:53.345854Z","shell.execute_reply.started":"2026-05-08T20:33:52.787903Z","shell.execute_reply":"2026-05-08T20:33:53.344847Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Bounding Box Area Distribution","metadata":{}},{"cell_type":"code","source":"# Compute area as width × height (assumes axis-aligned rectangle)\ndf_pos['area'] = df_pos['width'] * df_pos['height']\n\nfig, axes = plt.subplots(1, 2, figsize=(12, 4))\n\n# Plot 1: Area histogram with mean line — shows typical pneumonia lesion coverage\naxes[0].hist(df_pos['area'].dropna(), bins=40, color='steelblue', edgecolor='white')\naxes[0].set_title('Bounding Box Area Distribution')\naxes[0].set_xlabel('Area (pixels²)')\naxes[0].set_ylabel('Count')\naxes[0].axvline(df_pos['area'].mean(), color='red', linestyle='--',\n                label=f'Mean: {df_pos[\"area\"].mean():,.0f}')\naxes[0].legend()\n\n# Plot 2: Width vs Height scatter — check if boxes are roughly square or elongated\naxes[1].scatter(df_pos['width'], df_pos['height'],\n                alpha=0.2, s=10, color='tomato')\naxes[1].set_title('Bounding Box Width vs Height')\naxes[1].set_xlabel('Width (pixels)')\naxes[1].set_ylabel('Height (pixels)')\n\nprint(f'Mean area  : {df_pos[\"area\"].mean():,.0f} px²')\nprint(f'Median area: {df_pos[\"area\"].median():,.0f} px²')\nprint(f'Min area   : {df_pos[\"area\"].min():,.0f} px²')\nprint(f'Max area   : {df_pos[\"area\"].max():,.0f} px²')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:33:53.349970Z","iopub.execute_input":"2026-05-08T20:33:53.350299Z","iopub.status.idle":"2026-05-08T20:33:53.775461Z","shell.execute_reply.started":"2026-05-08T20:33:53.350267Z","shell.execute_reply":"2026-05-08T20:33:53.774306Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Image Metadata Analysis","metadata":{}},{"cell_type":"code","source":"import pydicom\n\n# Read one DICOM file to inspect shared metadata (all images should match)\nsample = pydicom.dcmread(f'{TRAIN_DIR}/{patients[\"patientId\"].iloc[0]}.dcm')\n\n# Rows x Columns tells us the native resolution before any resizing\nprint('Image size  :', sample.Rows, 'x', sample.Columns)\n# BitsAllocated shows pixel depth (16-bit is standard for X-ray DICOMs)\nprint('Bits        :', sample.BitsAllocated)\n# Modality confirms this is an X-ray (DX) and not CT or MRI\nprint('Modality    :', sample.Modality)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:33:53.776439Z","iopub.execute_input":"2026-05-08T20:33:53.776717Z","iopub.status.idle":"2026-05-08T20:33:53.795482Z","shell.execute_reply.started":"2026-05-08T20:33:53.776689Z","shell.execute_reply":"2026-05-08T20:33:53.794245Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Pixel Intensity Analysis","metadata":{}},{"cell_type":"code","source":"print('Analyzing pixel intensities from 100 samples (50 normal, 50 pneumonia)...')\n\ndef apply_window(img, level=-500, width=1500):\n    \"\"\"Apply lung windowing: clip HU range then rescale to [0, 1].\"\"\"\n    low  = level - width // 2\n    high = level + width // 2\n    img  = np.clip(img, low, high)\n    img  = (img - low) / (high - low)\n    return img\n\n# Sample 50 patients from each class for a quick representative comparison\nnormal_pids    = patients[patients['Target']==0]['patientId'].sample(50, random_state=42).values\npneumonia_pids = patients[patients['Target']==1]['patientId'].sample(50, random_state=42).values\n\nnormal_means, pneumonia_means = [], []\nnormal_stds,  pneumonia_stds  = [], []\n\n# Compute per-image mean and std after windowing\n# Pneumonia causes increased opacity (higher mean), while stds may differ too\nfor pid in tqdm(normal_pids):\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = apply_window(dcm.pixel_array.astype(np.float32))\n    normal_means.append(img.mean())\n    normal_stds.append(img.std())\n\nfor pid in tqdm(pneumonia_pids):\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = apply_window(dcm.pixel_array.astype(np.float32))\n    pneumonia_means.append(img.mean())\n    pneumonia_stds.append(img.std())\n\nfig, axes = plt.subplots(1, 2, figsize=(12, 4))\n\n# Plot 1: Mean intensity — pneumonia opacities should shift this distribution\naxes[0].hist(normal_means,    bins=20, alpha=0.7, color='steelblue', label='Normal')\naxes[0].hist(pneumonia_means, bins=20, alpha=0.7, color='tomato',    label='Pneumonia')\naxes[0].set_title('Mean Pixel Intensity Distribution')\naxes[0].set_xlabel('Mean Intensity')\naxes[0].set_ylabel('Count')\naxes[0].legend()\n\n# Plot 2: Std intensity — pneumonia often increases local contrast variation\naxes[1].hist(normal_stds,    bins=20, alpha=0.7, color='steelblue', label='Normal')\naxes[1].hist(pneumonia_stds, bins=20, alpha=0.7, color='tomato',    label='Pneumonia')\naxes[1].set_title('Pixel Intensity Std Distribution')\naxes[1].set_xlabel('Std of Intensity')\naxes[1].set_ylabel('Count')\naxes[1].legend()\n\nplt.tight_layout()\nplt.show()\n\nprint(f'\\nNormal    — Mean: {np.mean(normal_means):.3f}, Std: {np.mean(normal_stds):.3f}')\nprint(f'Pneumonia — Mean: {np.mean(pneumonia_means):.3f}, Std: {np.mean(pneumonia_stds):.3f}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:33:53.796705Z","iopub.execute_input":"2026-05-08T20:33:53.796988Z","iopub.status.idle":"2026-05-08T20:33:56.253724Z","shell.execute_reply.started":"2026-05-08T20:33:53.796958Z","shell.execute_reply":"2026-05-08T20:33:56.252389Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train / Val / Test","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nSEED = 42\n\n# ── Two-stage stratified split ──\n# Stratify=Target ensures each split keeps the same ~24% pneumonia ratio,\n# preventing train/val/test from having very different class distributions.\n\n# Stage 1: Train (70%) vs Temp (30%)\ntrain_df, temp_df = train_test_split(patients, test_size=0.30,\n                                     stratify=patients['Target'],\n                                     random_state=SEED)\n\n# Stage 2: Split the temp 30% evenly into Val (15%) and Test (15%)\nval_df, test_df = train_test_split(temp_df, test_size=0.50,\n                                   stratify=temp_df['Target'],\n                                   random_state=SEED)\n\nprint(f'Train : {len(train_df):,} ({len(train_df)/len(patients)*100:.1f}%)')\nprint(f'Val   : {len(val_df):,}  ({len(val_df)/len(patients)*100:.1f}%)')\nprint(f'Test  : {len(test_df):,}  ({len(test_df)/len(patients)*100:.1f}%)')\n\n# Verify that pneumonia rate is consistent across all three splits\nfor name, split in [('Train', train_df), ('Val', val_df), ('Test', test_df)]:\n    pos = split['Target'].sum()\n    print(f'{name} — Pneumonia: {pos} ({pos/len(split)*100:.1f}%)')\n\n# Persist the patient ID lists to CSV so later steps don't have to re-split\ntrain_df.to_csv('/kaggle/working/train_split.csv', index=False)\nval_df.to_csv('/kaggle/working/val_split.csv',   index=False)\ntest_df.to_csv('/kaggle/working/test_split.csv',  index=False)\nprint('\\n✅ Splits saved to /kaggle/working/')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:33:56.254842Z","iopub.execute_input":"2026-05-08T20:33:56.255150Z","iopub.status.idle":"2026-05-08T20:33:57.253474Z","shell.execute_reply.started":"2026-05-08T20:33:56.255120Z","shell.execute_reply":"2026-05-08T20:33:57.252561Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 3: Preprocessing","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport pydicom\nimport cv2\nimport os\nfrom tqdm import tqdm\n\n# Dataset paths\nDATA_DIR  = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\nTRAIN_DIR = f'{DATA_DIR}/stage_2_train_images'\nOUTPUT_DIR = '/kaggle/working'\n\n# Load labels and pre-computed patient-level splits from Step 1/2\ndf       = pd.read_csv(f'{DATA_DIR}/stage_2_train_labels.csv')\npatients = df.groupby('patientId')['Target'].max().reset_index()\n\ntrain_df = pd.read_csv(f'{OUTPUT_DIR}/train_split.csv')\nval_df   = pd.read_csv(f'{OUTPUT_DIR}/val_split.csv')\ntest_df  = pd.read_csv(f'{OUTPUT_DIR}/test_split.csv')\n\nprint('✅ Imports done.')\nprint(f'Train: {len(train_df):,} | Val: {len(val_df):,} | Test: {len(test_df):,}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:33:57.254731Z","iopub.execute_input":"2026-05-08T20:33:57.255041Z","iopub.status.idle":"2026-05-08T20:33:57.346711Z","shell.execute_reply.started":"2026-05-08T20:33:57.254988Z","shell.execute_reply":"2026-05-08T20:33:57.345559Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Functions","metadata":{}},{"cell_type":"code","source":"def load_dicom(pid):\n    \"\"\"Load raw DICOM pixel array for a given patient ID.\"\"\"\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = dcm.pixel_array.astype(np.float32)  # Cast to float32 for downstream math\n    return img\n\ndef apply_window(img, level=-500, width=1500):\n    \"\"\"Apply lung windowing to enhance pulmonary structures.\n    \n    Clips pixel values to the HU range [level-width/2, level+width/2]\n    and rescales to [0, 1]. Lung window (-500 HU ± 750) suppresses bone\n    and fat, making air spaces and consolidations clearly visible.\n    \"\"\"\n    low  = level - width // 2\n    high = level + width // 2\n    img  = np.clip(img, low, high)\n    img  = (img - low) / (high - low)\n    return img\n\ndef normalize(img):\n    \"\"\"Min-max normalize image to [0, 1].\n    \n    Shifts the minimum to 0 then divides by the range.\n    Guard against all-zero arrays (avoids division by zero).\n    \"\"\"\n    img = img - img.min()\n    if img.max() > 0:\n        img = img / img.max()\n    return img\n\ndef enhance_contrast(img):\n    \"\"\"Apply CLAHE (Contrast Limited Adaptive Histogram Equalization).\n    \n    CLAHE improves local contrast without over-amplifying noise.\n    - clipLimit=2.0: prevents extreme contrast amplification in uniform regions\n    - tileGridSize=(8,8): divides image into 8x8 tiles for local equalization\n    Converts to uint8 for OpenCV, applies CLAHE, then returns float32 [0, 1].\n    \"\"\"\n    img_uint8 = (img * 255).astype(np.uint8)\n\n    clahe = cv2.createCLAHE(\n        clipLimit=2.0,\n        tileGridSize=(8,8)\n    )\n\n    enhanced = clahe.apply(img_uint8)\n\n    return enhanced / 255.0     # Rescale back to [0, 1]\n\ndef denoise(img):\n    \"\"\"Apply Gaussian blur for noise reduction.\n    \n    3x3 kernel is small enough to preserve edges while smoothing\n    high-frequency noise introduced by the detector.\n    \"\"\"\n    return cv2.GaussianBlur(img, (3, 3), 0)\n\ndef resize_img(img, size=512):\n    \"\"\"Resize image to a fixed square size using area interpolation.\n    \n    INTER_AREA is preferred for downscaling — it averages pixel values\n    in each target pixel's footprint, preserving texture better than\n    nearest-neighbor or bilinear interpolation.\n    \"\"\"\n    return cv2.resize(img, (size, size), interpolation=cv2.INTER_AREA)\n\ndef preprocess(pid, size=512):\n    \"\"\"Full preprocessing pipeline for one patient.\n    \n    Pipeline order:\n      1. Load raw DICOM pixels\n      2. Normalize to [0, 1] (required before CLAHE which expects [0,1] input)\n      3. Enhance contrast with CLAHE (improves feature visibility)\n      4. Resize to fixed 512x512 (standardize spatial resolution)\n    \"\"\"\n    img = load_dicom(pid)       # Step 1: Load DICOM\n    img = normalize(img)        # Step 2: Normalize to [0,1] first\n    img = enhance_contrast(img) # Step 3: CLAHE contrast enhancement\n    img = resize_img(img, size) # Step 4: Resize to 512x512\n    return img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:33:57.347827Z","iopub.execute_input":"2026-05-08T20:33:57.348149Z","iopub.status.idle":"2026-05-08T20:33:57.360494Z","shell.execute_reply.started":"2026-05-08T20:33:57.348118Z","shell.execute_reply":"2026-05-08T20:33:57.358871Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Test Preprocessing on Single Image","metadata":{}},{"cell_type":"code","source":"# Run each preprocessing step in isolation to sanity-check shapes and value ranges\npid = patients['patientId'].iloc[0]\n\nraw     = load_dicom(pid)\nnormed  = normalize(raw)\nenhanced = enhance_contrast(normed)\nresized  = resize_img(enhanced)\n\n# Confirm:\n# - raw: 16-bit HU values (can be large positive/negative)\n# - normed: [0, 1] float\n# - enhanced: [0, 1] float after CLAHE\n# - resized: 512x512, still [0, 1]\nprint(f'Raw      — shape: {raw.shape},     min: {raw.min():.1f},   max: {raw.max():.1f}')\nprint(f'Normed   — shape: {normed.shape},   min: {normed.min():.2f},  max: {normed.max():.2f}')\nprint(f'Enhanced — shape: {enhanced.shape}, min: {enhanced.min():.2f},  max: {enhanced.max():.2f}')\nprint(f'Resized  — shape: {resized.shape},  min: {resized.min():.2f},  max: {resized.max():.2f}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:33:57.361901Z","iopub.execute_input":"2026-05-08T20:33:57.362451Z","iopub.status.idle":"2026-05-08T20:33:57.439627Z","shell.execute_reply.started":"2026-05-08T20:33:57.362415Z","shell.execute_reply":"2026-05-08T20:33:57.438443Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Augmentation","metadata":{}},{"cell_type":"code","source":"import albumentations as A\n\n# Define the augmentation pipeline.\n# Augmentation artificially expands the training set and helps the model\n# generalize to images captured at slightly different orientations / exposures.\n#\n# Chosen transforms and rationale:\n#   HorizontalFlip(p=0.5)   — valid for chest X-rays; left/right lung symmetry\n#   Rotate(limit=10)        — small rotation mimics patient positioning variation\n#   Affine(translate=0.1)   — slight shift simulates off-center positioning\n#   RandomBrightnessContrast — simulates varying exposure / detector sensitivity\naugment = A.Compose([\n    A.HorizontalFlip(p=0.5),\n    A.Rotate(limit=10, p=0.3),\n    A.Affine(translate_percent=0.1, p=0.3),\n    A.RandomBrightnessContrast(brightness_limit=0.2,\n                               contrast_limit=0.2, p=0.4),\n])\n\ndef augment_img(img):\n    \"\"\"Apply the augmentation pipeline to a single [0,1] float image.\n    \n    Albumentations expects uint8 input, so we convert before augmenting\n    and rescale back to float32 [0, 1] after.\n    \"\"\"\n    img_uint8 = (img * 255).astype(np.uint8)\n    augmented = augment(image=img_uint8)['image']\n    return augmented.astype(np.float32) / 255.0\n\nprint('✅ Augmentation pipeline defined.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:33:57.441154Z","iopub.execute_input":"2026-05-08T20:33:57.441683Z","iopub.status.idle":"2026-05-08T20:34:03.342974Z","shell.execute_reply.started":"2026-05-08T20:33:57.441626Z","shell.execute_reply":"2026-05-08T20:34:03.342066Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualize Augmentation Effects","metadata":{}},{"cell_type":"code","source":"# Use a pneumonia case so we can see how augmentations affect the lesion region\nsample_pid = patients[patients['Target']==1]['patientId'].sample(1, random_state=7).values[0]\nimg = preprocess(sample_pid)\n\nfig, axes = plt.subplots(2, 4, figsize=(16, 8))\naxes = axes.flatten()\n\n# First panel: the preprocessed (but un-augmented) baseline\naxes[0].imshow(img, cmap='gray')\naxes[0].set_title('Original (preprocessed)')\naxes[0].axis('off')\n\n# 7 randomly augmented variants to visually verify the transforms look realistic\nfor i in range(1, 8):\n    aug = augment_img(img)\n    axes[i].imshow(aug, cmap='gray')\n    axes[i].set_title(f'Augmented #{i}')\n    axes[i].axis('off')\n\nplt.suptitle('Data Augmentation Examples', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:34:03.344714Z","iopub.execute_input":"2026-05-08T20:34:03.345311Z","iopub.status.idle":"2026-05-08T20:34:04.726669Z","shell.execute_reply.started":"2026-05-08T20:34:03.345275Z","shell.execute_reply":"2026-05-08T20:34:04.725590Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Before VS After Preprocessing","metadata":{}},{"cell_type":"code","source":"# Visual comparison of the three preprocessing stages across 4 patients\nfig, axes = plt.subplots(3, 4, figsize=(16, 12))\n\nsample_pids = patients.sample(4, random_state=10)['patientId'].values\n\nfor idx, pid in enumerate(sample_pids):\n    raw      = load_dicom(pid)\n    windowed = apply_window(raw)             # Apply lung HU window\n    resized  = resize_img(normalize(windowed)) # Normalize then resize\n\n    # Row 1: Raw DICOM — wide HU range, not visually optimized\n    axes[0][idx].imshow(raw, cmap='gray')\n    axes[0][idx].set_title(f'Raw DICOM\\n{raw.shape}')\n    axes[0][idx].axis('off')\n\n    # Row 2: After lung windowing — lung structures become visible\n    axes[1][idx].imshow(windowed, cmap='gray')\n    axes[1][idx].set_title(f'After Windowing\\n[{windowed.min():.2f}, {windowed.max():.2f}]')\n    axes[1][idx].axis('off')\n\n    # Row 3: Final 512x512 — ready for feature extraction\n    axes[2][idx].imshow(resized, cmap='gray')\n    axes[2][idx].set_title(f'Final (512x512)\\n[{resized.min():.2f}, {resized.max():.2f}]')\n    axes[2][idx].axis('off')\n\nplt.suptitle('Preprocessing Pipeline: Raw → Windowed → Final', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:34:04.728060Z","iopub.execute_input":"2026-05-08T20:34:04.728435Z","iopub.status.idle":"2026-05-08T20:34:07.836052Z","shell.execute_reply.started":"2026-05-08T20:34:04.728400Z","shell.execute_reply":"2026-05-08T20:34:07.834486Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Histogram Update","metadata":{}},{"cell_type":"code","source":"# Show pixel value histograms at each preprocessing stage.\n# Histograms reveal how each step reshapes the intensity distribution.\nfig, axes = plt.subplots(3, 4, figsize=(16, 12))\n\nsample_pids = patients.sample(4, random_state=10)['patientId'].values\n\nfor idx, pid in enumerate(sample_pids):\n    raw      = load_dicom(pid)\n    normed   = normalize(raw)\n    # Final image = CLAHE contrast enhancement + resize\n    final    = resize_img(enhance_contrast(normed))\n\n    # Row 1: Raw histogram — bimodal (dark air vs bright tissue) with wide spread\n    axes[0][idx].hist(raw.flatten(), bins=50, color='gray', edgecolor='none')\n    axes[0][idx].set_title(f'Raw\\nmin={raw.min():.0f} max={raw.max():.0f}')\n    axes[0][idx].set_xlabel('Pixel Value')\n    axes[0][idx].set_ylabel('Count')\n\n    # Row 2: After min-max normalization — same shape, rescaled to [0, 1]\n    axes[1][idx].hist(normed.flatten(), bins=50, color='steelblue', edgecolor='none')\n    axes[1][idx].set_title(f'After Normalize\\nmin={normed.min():.2f} max={normed.max():.2f}')\n    axes[1][idx].set_xlabel('Pixel Value')\n    axes[1][idx].set_ylabel('Count')\n\n    # Row 3: After CLAHE + resize — more uniform distribution, better contrast\n    axes[2][idx].hist(final.flatten(), bins=50, color='tomato', edgecolor='none')\n    axes[2][idx].set_title(f'Final (enhanced+resized)\\nmin={final.min():.2f} max={final.max():.2f}')\n    axes[2][idx].set_xlabel('Pixel Value')\n    axes[2][idx].set_ylabel('Count')\n\nplt.suptitle('Preprocessing Pipeline: Raw → Normalized → Final', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:34:07.837670Z","iopub.execute_input":"2026-05-08T20:34:07.838023Z","iopub.status.idle":"2026-05-08T20:34:10.600702Z","shell.execute_reply.started":"2026-05-08T20:34:07.837961Z","shell.execute_reply":"2026-05-08T20:34:10.599509Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 4: Lung Segmentation","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport pydicom\nimport cv2\nfrom scipy import ndimage      # Used for morphological fill operations\nfrom tqdm import tqdm\nimport os\n\nDATA_DIR  = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\nTRAIN_DIR = f'{DATA_DIR}/stage_2_train_images'\nOUTPUT_DIR = '/kaggle/working'\n\ndf       = pd.read_csv(f'{DATA_DIR}/stage_2_train_labels.csv')\npatients = df.groupby('patientId')['Target'].max().reset_index()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:34:10.602156Z","iopub.execute_input":"2026-05-08T20:34:10.602746Z","iopub.status.idle":"2026-05-08T20:34:10.665943Z","shell.execute_reply.started":"2026-05-08T20:34:10.602688Z","shell.execute_reply":"2026-05-08T20:34:10.665160Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Segmentation Function","metadata":{}},{"cell_type":"code","source":"def segment_lungs(img):\n    \"\"\"\n    Segment the lung region from a preprocessed chest X-ray.\n    \n    Algorithm:\n      1. Convert to uint8\n      2. Otsu thresholding  — automatically finds the best global threshold\n      3. Invert             — lungs appear dark in X-ray, inversion makes them foreground\n      4. Morphological open — removes small noise blobs (specks) outside lungs\n      5. Morphological close— fills holes inside the lung region\n      6. Keep top-2 connected components — retains only left and right lung lobes\n    \n    Args:\n        img  : float32 array in [0, 1], preprocessed X-ray\n    Returns:\n        mask : uint8 binary mask (255 = lung, 0 = background)\n    \"\"\"\n    # Step 1: Convert float [0,1] → uint8 [0,255] for OpenCV compatibility\n    img_uint8 = (img * 255).astype(np.uint8)\n\n    # Step 2: Otsu thresholding — separates lung (dark) from background (bright)\n    # THRESH_BINARY + THRESH_OTSU automatically determines the optimal threshold value\n    _, thresh = cv2.threshold(img_uint8, 0, 255,\n                              cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n\n    # Step 3: Invert the binary mask — lungs are dark so Otsu marks them as 0;\n    # after inversion they become 255 (foreground)\n    thresh = cv2.bitwise_not(thresh)\n\n    # Step 4: Morphological opening (erode then dilate) with a 5x5 elliptical kernel\n    # Removes thin noise connected to the lung boundary\n    kernel  = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (5, 5))\n    opened  = cv2.morphologyEx(thresh, cv2.MORPH_OPEN, kernel)\n\n    # Step 5: Morphological closing (dilate then erode) with a larger 20x20 kernel\n    # Fills gaps/holes inside the lung region caused by blood vessels or opacities\n    kernel2 = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (20, 20))\n    closed  = cv2.morphologyEx(opened, cv2.MORPH_CLOSE, kernel2)\n\n    # Step 6: Connected component analysis — find all separate blobs\n    # We keep only the 2 largest components (left + right lung) and discard the rest\n    num_labels, labels, stats, _ = cv2.connectedComponentsWithStats(closed)\n\n    # Sort by area descending (skip label 0 = background, hence +1 offset)\n    areas = stats[1:, cv2.CC_STAT_AREA]\n    top2  = np.argsort(areas)[::-1][:2] + 1  # Indices of the 2 largest components\n\n    # Build the final binary mask containing only the two lung lobes\n    mask = np.zeros_like(closed)\n    for label in top2:\n        mask[labels == label] = 255\n\n    return mask\n\ndef apply_mask(img, mask):\n    \"\"\"Zero out non-lung regions by multiplying the image with the binary mask.\n    \n    Normalizes mask to [0, 1] before multiplying so pixel values are preserved\n    inside the lung and set to 0 outside.\n    \"\"\"\n    mask_norm = mask / 255.0\n    return img * mask_norm\n\nprint('✅ Segmentation functions defined.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:34:10.667227Z","iopub.execute_input":"2026-05-08T20:34:10.667598Z","iopub.status.idle":"2026-05-08T20:34:10.678411Z","shell.execute_reply.started":"2026-05-08T20:34:10.667556Z","shell.execute_reply":"2026-05-08T20:34:10.676926Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Test Segmentation on Sample Image","metadata":{}},{"cell_type":"code","source":"# Test segmentation on 4 random patients (mix of classes) and display side-by-side\nfig, axes = plt.subplots(3, 4, figsize=(16, 12))\n\nsample_pids = patients.sample(4, random_state=42)['patientId'].values\n\nfor idx, pid in enumerate(sample_pids):\n    # ── Inline preprocessing (mirrors the preprocess() function) ──\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = dcm.pixel_array.astype(np.float32)\n    img = img - img.min()\n    if img.max() > 0:\n        img = img / img.max()           # Normalize to [0, 1]\n    img = enhance_contrast(img)         # CLAHE\n    img = cv2.resize(img, (512, 512), interpolation=cv2.INTER_AREA)\n\n    # ── Segmentation ──\n    mask        = segment_lungs(img)    # Binary lung mask\n    masked_img  = apply_mask(img, mask) # Image with background zeroed out\n\n    # Row 1: Preprocessed image (input to segmentation)\n    axes[0][idx].imshow(img, cmap='gray')\n    axes[0][idx].set_title('Preprocessed')\n    axes[0][idx].axis('off')\n\n    # Row 2: Binary lung mask (white = lung, black = background)\n    axes[1][idx].imshow(mask, cmap='gray')\n    axes[1][idx].set_title('Lung Mask')\n    axes[1][idx].axis('off')\n\n    # Row 3: Masked image — only lung pixels remain, background is black\n    axes[2][idx].imshow(masked_img, cmap='gray')\n    axes[2][idx].set_title('Segmented Lung')\n    axes[2][idx].axis('off')\n\nplt.suptitle('Lung Segmentation: Thresholding + Morphology', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:34:10.679836Z","iopub.execute_input":"2026-05-08T20:34:10.680383Z","iopub.status.idle":"2026-05-08T20:34:12.426217Z","shell.execute_reply.started":"2026-05-08T20:34:10.680344Z","shell.execute_reply":"2026-05-08T20:34:12.425025Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Test segmentation on 4 random patients (mix of classes) and display side-by-side\nfig, axes = plt.subplots(3, 4, figsize=(16, 12))\n\nsample_pids = patients.sample(4, random_state=42)['patientId'].values\n\nfor idx, pid in enumerate(sample_pids):\n    # ── Inline preprocessing (mirrors the preprocess() function) ──\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = dcm.pixel_array.astype(np.float32)\n    img = img - img.min()\n    if img.max() > 0:\n        img = img / img.max()           # Normalize to [0, 1]\n    img = enhance_contrast(img)         # CLAHE\n    img = cv2.resize(img, (512, 512), interpolation=cv2.INTER_AREA)\n\n    # ── Segmentation ──\n    mask        = segment_lungs(img)    # Binary lung mask\n    masked_img  = apply_mask(img, mask) # Image with background zeroed out\n\n    # Row 1: Preprocessed image (input to segmentation)\n    axes[0][idx].imshow(img, cmap='gray')\n    axes[0][idx].set_title('Preprocessed')\n    axes[0][idx].axis('off')\n\n    # Row 2: Binary lung mask (white = lung, black = background)\n    axes[1][idx].imshow(mask, cmap='gray')\n    axes[1][idx].set_title('Lung Mask')\n    axes[1][idx].axis('off')\n\n    # Row 3: Masked image — only lung pixels remain, background is black\n    axes[2][idx].imshow(masked_img, cmap='gray')\n    axes[2][idx].set_title('Segmented Lung')\n    axes[2][idx].axis('off')\n\nplt.suptitle('Lung Segmentation: Thresholding + Morphology', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:34:12.427601Z","iopub.execute_input":"2026-05-08T20:34:12.427988Z","iopub.status.idle":"2026-05-08T20:34:14.346465Z","shell.execute_reply.started":"2026-05-08T20:34:12.427937Z","shell.execute_reply":"2026-05-08T20:34:14.345383Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Test Segmentation on Pneumonia Cases","metadata":{}},{"cell_type":"code","source":"# Repeat segmentation test specifically on pneumonia patients.\n# Important: pneumonia consolidations can confuse the thresholding step because\n# they appear brighter than normal lung tissue — this checks robustness.\nfig, axes = plt.subplots(3, 4, figsize=(16, 12))\n\npneumonia_pids = patients[patients['Target']==1]['patientId'].sample(4, random_state=42).values\n\nfor idx, pid in enumerate(pneumonia_pids):\n    # Preprocess (same inline pipeline as above)\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = dcm.pixel_array.astype(np.float32)\n    img = img - img.min()\n    if img.max() > 0:\n        img = img / img.max()\n    img = enhance_contrast(img)\n    img = cv2.resize(img, (512, 512), interpolation=cv2.INTER_AREA)\n\n    mask       = segment_lungs(img)\n    masked_img = apply_mask(img, mask)\n\n    # Row 1: Pneumonia original\n    axes[0][idx].imshow(img, cmap='gray')\n    axes[0][idx].set_title('Pneumonia — Original')\n    axes[0][idx].axis('off')\n\n    # Row 2: Mask — verify both lobes are captured even with opacities present\n    axes[1][idx].imshow(mask, cmap='gray')\n    axes[1][idx].set_title('Lung Mask')\n    axes[1][idx].axis('off')\n\n    # Row 3: Segmented lung — consolidation regions should still be inside the mask\n    axes[2][idx].imshow(masked_img, cmap='gray')\n    axes[2][idx].set_title('Segmented Lung')\n    axes[2][idx].axis('off')\n\nplt.suptitle('Lung Segmentation on Pneumonia Cases', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:34:14.347797Z","iopub.execute_input":"2026-05-08T20:34:14.348177Z","iopub.status.idle":"2026-05-08T20:34:16.207567Z","shell.execute_reply.started":"2026-05-08T20:34:14.348141Z","shell.execute_reply":"2026-05-08T20:34:16.206280Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 5: Feature Extraction","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"from scipy.stats import skew, kurtosis          # Statistical moment features\nfrom skimage.feature import graycomatrix, graycoprops  # GLCM texture features\nfrom skimage.measure import regionprops, label  # Shape/geometry features from binary mask\nimport pandas as pd\nfrom tqdm import tqdm\nimport pydicom\nimport cv2\nimport os","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:34:16.209036Z","iopub.execute_input":"2026-05-08T20:34:16.209671Z","iopub.status.idle":"2026-05-08T20:34:16.668600Z","shell.execute_reply.started":"2026-05-08T20:34:16.209622Z","shell.execute_reply":"2026-05-08T20:34:16.667458Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Extraction Logic","metadata":{}},{"cell_type":"code","source":"def extract_features(img, mask):\n    \"\"\"Extract a rich hand-crafted feature vector from a segmented lung image.\n    \n    Four feature groups are computed:\n      1. Intensity statistics  — mean, std, skewness, kurtosis + histogram bins\n      2. GLCM texture          — contrast, energy, homogeneity, correlation, etc.\n      3. Shape/geometry        — area, perimeter, circularity of the lung mask\n      4. Per-lobe (asymmetry)  — separate stats for left/right lung\n         Asymmetry is a key clinical indicator of pneumonia (one-sided infection).\n    \n    Args:\n        img  : float32 [0,1] preprocessed image (512x512)\n        mask : uint8 binary lung mask (255 = lung region)\n    Returns:\n        features : dict of {feature_name: float}\n    \"\"\"\n    features = {}\n    bool_mask   = mask > 0             # Boolean mask for indexing\n    lung_pixels = img[bool_mask]       # 1D array of pixels inside the lung\n\n    # ──────────────────────────────────────────────\n    # 1. INTENSITY STATISTICS (global across the whole lung)\n    # ──────────────────────────────────────────────\n    if len(lung_pixels) > 0:\n        features['mean']     = float(np.mean(lung_pixels))    # Average brightness\n        features['std']      = float(np.std(lung_pixels))     # Spread of intensity\n        features['skewness'] = float(skew(lung_pixels))       # Asymmetry of distribution\n        features['kurtosis'] = float(kurtosis(lung_pixels))   # Tail heaviness\n    else:\n        features['mean'] = features['std'] = features['skewness'] = features['kurtosis'] = 0.0\n\n    # 10-bin histogram — captures the shape of the intensity distribution\n    # Normalized so the total sums to 1 (relative frequency)\n    hist, _ = np.histogram(lung_pixels if len(lung_pixels) > 0 else [0], bins=10, range=(0, 1))\n    hist = hist / (hist.sum() + 1e-6)  # Avoid division by zero\n    for i, val in enumerate(hist):\n        features[f'hist_bin_{i}'] = float(val)\n\n    # ──────────────────────────────────────────────\n    # 2. GLCM TEXTURE (Gray-Level Co-occurrence Matrix)\n    # ──────────────────────────────────────────────\n    # GLCM captures spatial relationships between neighboring pixel intensities.\n    # Using multiple distances [1,2,3] and angles [0°,45°,90°,135°] makes the\n    # texture features rotation-invariant (averaged over all angles).\n    img_uint8    = (img * 255).astype(np.uint8)\n    labeled_mask = label(bool_mask)     # Label connected components in the mask\n    regions      = regionprops(labeled_mask)\n\n    if regions:\n        # Crop to the bounding box of all lung regions to reduce GLCM computation time\n        min_row = min([r.bbox[0] for r in regions])\n        min_col = min([r.bbox[1] for r in regions])\n        max_row = max([r.bbox[2] for r in regions])\n        max_col = max([r.bbox[3] for r in regions])\n\n        cropped_mask = bool_mask[min_row:max_row, min_col:max_col]\n        cropped_img  = img_uint8[min_row:max_row, min_col:max_col]\n        cropped_img  = cropped_img * cropped_mask   # Zero out non-lung pixels\n\n        # Quantize to 64 levels: reduces GLCM matrix size (64x64 vs 256x256)\n        # and speeds up computation while keeping enough texture resolution\n        img_quantized = (cropped_img // 4).astype(np.uint8)\n        glcm = graycomatrix(img_quantized,\n                            distances=[1, 2, 3],\n                            angles=[0, np.pi/4, np.pi/2, 3*np.pi/4],\n                            levels=64, symmetric=True, normed=True)\n\n        # Extract standard GLCM properties and average across all distances/angles\n        for prop in ['contrast', 'energy', 'homogeneity', 'correlation', 'dissimilarity', 'ASM']:\n            features[prop] = float(np.mean(graycoprops(glcm, prop)))\n    else:\n        for prop in ['contrast', 'energy', 'homogeneity', 'correlation', 'dissimilarity', 'ASM']:\n            features[prop] = 0.0\n\n    # ──────────────────────────────────────────────\n    # 3. SHAPE FEATURES (global lung geometry)\n    # ──────────────────────────────────────────────\n    if regions:\n        total_area      = sum([r.area for r in regions])\n        total_perimeter = sum([r.perimeter for r in regions])\n        features['area']      = float(total_area)\n        features['perimeter'] = float(total_perimeter)\n        # Circularity: 1.0 = perfect circle, lower = more elongated/irregular\n        features['circularity'] = (4 * np.pi * total_area) / (total_perimeter ** 2 + 1e-6)\n    else:\n        features['area'] = features['perimeter'] = features['circularity'] = 0.0\n\n    # ──────────────────────────────────────────────\n    # 4. PER-LOBE FEATURES (left vs right lung asymmetry)\n    # ──────────────────────────────────────────────\n    # Pneumonia often affects one lung more than the other.\n    # Sorting by centroid column (x-axis) separates left from right lobe.\n    if regions:\n        regions_sorted = sorted(regions, key=lambda r: r.centroid[1])\n        if len(regions_sorted) >= 2:\n            left_lung, right_lung = regions_sorted[0], regions_sorted[1]\n        else:\n            # Fallback: if only one component was detected, use it for both sides\n            left_lung = right_lung = regions_sorted[0]\n\n        for side, region in zip(['left', 'right'], [left_lung, right_lung]):\n            lobe_mask   = labeled_mask == region.label  # Isolate this lobe's pixels\n            lobe_pixels = img[lobe_mask]\n            features[f'{side}_mean'] = float(np.mean(lobe_pixels)) if len(lobe_pixels) > 0 else 0.0\n            features[f'{side}_std']  = float(np.std(lobe_pixels))  if len(lobe_pixels) > 0 else 0.0\n            features[f'{side}_area'] = float(region.area)\n            features[f'{side}_circularity'] = (4 * np.pi * region.area) / (region.perimeter ** 2 + 1e-6)\n\n        # Ratio of left to right lobe area — deviations from 1.0 indicate asymmetry\n        features['area_ratio'] = float(regions_sorted[0].area / (regions_sorted[-1].area + 1e-6))\n    else:\n        for side in ['left', 'right']:\n            features[f'{side}_mean'] = features[f'{side}_std'] = 0.0\n            features[f'{side}_area'] = features[f'{side}_circularity'] = 0.0\n        features['area_ratio'] = 1.0\n\n    return features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:34:16.669658Z","iopub.execute_input":"2026-05-08T20:34:16.670190Z","iopub.status.idle":"2026-05-08T20:34:16.688396Z","shell.execute_reply.started":"2026-05-08T20:34:16.670157Z","shell.execute_reply":"2026-05-08T20:34:16.687469Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Dataset Builder Pipeline","metadata":{}},{"cell_type":"code","source":"def build_feature_dataset(df_split, max_samples=None):\n    \"\"\"\n    Run the full preprocessing → segmentation → feature extraction pipeline\n    over an entire dataset split and return results as a tabular DataFrame.\n    \n    Args:\n        df_split    : DataFrame with columns ['patientId', 'Target']\n        max_samples : optional int — cap the number of patients processed\n                      (useful for quick debugging runs)\n    Returns:\n        pd.DataFrame with one row per patient and one column per feature\n    \"\"\"\n    data = []\n\n    if max_samples:\n        df_split = df_split.head(max_samples)\n\n    for _, row in tqdm(df_split.iterrows(), total=len(df_split), desc=\"Extracting Features\"):\n        pid    = row['patientId']\n        target = row['Target']\n\n        try:\n            # Step 1: Load and preprocess the DICOM image\n            dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n            img = dcm.pixel_array.astype(np.float32)\n            img = img - img.min()\n            if img.max() > 0:\n                img = img / img.max()          # Normalize to [0, 1]\n            img = enhance_contrast(img)        # CLAHE\n            img = cv2.resize(img, (512, 512), interpolation=cv2.INTER_AREA)\n\n            # Step 2: Segment the lung region\n            mask = segment_lungs(img)\n\n            # Step 3: Extract hand-crafted features from the segmented lung\n            feats = extract_features(img, mask)\n\n            # Add patient identifier and class label for later joining\n            feats['patientId'] = pid\n            feats['Target']    = target\n\n            data.append(feats)\n\n        except Exception as e:\n            # Skip corrupted/unreadable DICOM files without stopping the pipeline.\n            # In practice, a small fraction of DICOM files in the dataset are malformed.\n            continue\n\n    return pd.DataFrame(data)\n\nprint('✅ Pipeline function defined.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:34:16.689728Z","iopub.execute_input":"2026-05-08T20:34:16.690329Z","iopub.status.idle":"2026-05-08T20:34:16.717462Z","shell.execute_reply.started":"2026-05-08T20:34:16.690287Z","shell.execute_reply":"2026-05-08T20:34:16.716107Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Run Extraction & Save","metadata":{}},{"cell_type":"code","source":"# Extract features for all three splits and save to CSV.\n# This step is the most computationally expensive — expect 20-60 min on full dataset.\n# Saving to CSV decouples feature extraction from model training.\n\nprint(\"\\nExtracting Train Features...\")\ntrain_features = build_feature_dataset(train_df)\ntrain_features.to_csv(f'{OUTPUT_DIR}/train_features.csv', index=False)\n\nprint(\"\\nExtracting Val Features...\")\nval_features = build_feature_dataset(val_df)\nval_features.to_csv(f'{OUTPUT_DIR}/val_features.csv', index=False)\n\nprint(\"\\nExtracting Test Features...\")\ntest_features = build_feature_dataset(test_df)\ntest_features.to_csv(f'{OUTPUT_DIR}/test_features.csv', index=False)\n\nprint(\"\\n✅ All features extracted and saved to CSV!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T20:34:16.718789Z","iopub.execute_input":"2026-05-08T20:34:16.719155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Imports for Steps 6–7 (feature selection, scaling, modeling, evaluation) ──\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings('ignore')  # Suppress sklearn/xgboost deprecation warnings\n\n# Feature selection: Mutual Information ranks features by statistical dependence with Target\nfrom sklearn.feature_selection import SelectKBest, mutual_info_classif\n\n# Preprocessing\nfrom sklearn.preprocessing import StandardScaler  # Z-score normalization\nfrom sklearn.decomposition import PCA             # (loaded but skipped below)\n\n# Cross-validation\nfrom sklearn.model_selection import StratifiedKFold, cross_val_score\n\n# Classifiers\nfrom sklearn.svm import SVC\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.ensemble import RandomForestClassifier\nfrom xgboost import XGBClassifier\n\n# Evaluation metrics\nfrom sklearn.metrics import (accuracy_score, precision_score, recall_score,\n                             f1_score, roc_auc_score, classification_report,\n                             confusion_matrix, roc_curve)\n\n# Class imbalance handling\nfrom imblearn.over_sampling import SMOTE  # Synthetic Minority Over-sampling Technique","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load Pre-split Data ","metadata":{}},{"cell_type":"code","source":"# Load the pre-extracted feature CSVs saved in Step 5.\n# Loading from CSV avoids re-running the slow feature extraction every time.\nBASE = \"/kaggle/working\"\ntrain_df = pd.read_csv(f\"{BASE}/train_features.csv\")\nval_df   = pd.read_csv(f\"{BASE}/val_features.csv\")\ntest_df  = pd.read_csv(f\"{BASE}/test_features.csv\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Verify shapes and class balance across splits\nprint(f\"Train: {train_df.shape} | Val: {val_df.shape} | Test: {test_df.shape}\")\nprint(f\"Train class dist: {train_df.Target.value_counts().to_dict()}\")\nprint(f\"Val   class dist: {val_df.Target.value_counts().to_dict()}\")\nprint(f\"Test  class dist: {test_df.Target.value_counts().to_dict()}\")\n\n# Identify which columns are features vs metadata vs label\nDROP_COLS    = ['patientId']      # Patient ID is an identifier, not a predictive feature\nTARGET_COL   = 'Target'           # Binary label: 0 = Normal, 1 = Pneumonia\nFEATURE_COLS = [c for c in train_df.columns if c not in DROP_COLS + [TARGET_COL]]\n\n# Extract numpy arrays for sklearn compatibility\nX_train, y_train = train_df[FEATURE_COLS].values, train_df[TARGET_COL].values\nX_val,   y_val   = val_df[FEATURE_COLS].values,   val_df[TARGET_COL].values\nX_test,  y_test  = test_df[FEATURE_COLS].values,  test_df[TARGET_COL].values\n\nprint(f\"\\nFeatures ({len(FEATURE_COLS)}): {FEATURE_COLS}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Step 6: Feature Selection**","metadata":{}},{"cell_type":"code","source":"# Select the top-k most informative features using Mutual Information.\n# MI measures the non-linear statistical dependence between each feature and the label.\n# Higher MI = feature carries more information about pneumonia vs normal.\n#\n# k = min(20, total features) — take up to 20; fewer if dataset has fewer features.\n# Fitting on train only: prevents information leakage from val/test.\nk = min(20, len(FEATURE_COLS))\nselector = SelectKBest(score_func=mutual_info_classif, k=k)\nselector.fit(X_train, y_train)  # Fit on training data ONLY\n\nselected_features = [FEATURE_COLS[i] for i, m in enumerate(selector.get_support()) if m]\nprint(f\"\\nTop-{k} selected features: {selected_features}\")\n\n# Apply the same feature mask to all splits\nX_train_sel = selector.transform(X_train)\nX_val_sel   = selector.transform(X_val)\nX_test_sel  = selector.transform(X_test)\n\n# Bar chart of MI scores for all features — helps understand which signals matter most\nplt.figure(figsize=(10, 4))\nplt.bar(FEATURE_COLS, selector.scores_)\nplt.xticks(rotation=45, ha='right')\nplt.title('Mutual Information Scores')\nplt.tight_layout()\nplt.savefig('feature_importance.png', dpi=150)\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# SCALING","metadata":{}},{"cell_type":"code","source":"# Z-score standardization: subtract mean, divide by std → each feature ~ N(0,1)\n# Required for distance-based models (SVM, Logistic Regression).\n# fit_transform on train only; transform (no re-fit) on val and test to prevent leakage.\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train_sel)  # Fit on train; transform train\nX_val_scaled   = scaler.transform(X_val_sel)         # Transform val with train statistics\nX_test_scaled  = scaler.transform(X_test_sel)        # Transform test with train statistics","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PCA","metadata":{}},{"cell_type":"code","source":"# PCA dimensionality reduction was evaluated but skipped for this run.\n# With only 20 selected features, PCA offers minimal benefit and may\n# reduce interpretability, so we use the scaled features directly.\nX_train_pca = X_train_scaled\nX_val_pca   = X_val_scaled\nX_test_pca  = X_test_scaled\nprint(f\"Skipping PCA — using all {X_train_scaled.shape[1]} scaled features directly\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# SMOTE","metadata":{}},{"cell_type":"code","source":"# SMOTE (Synthetic Minority Over-sampling Technique) addresses class imbalance\n# (~76% Normal vs ~24% Pneumonia) by generating synthetic pneumonia examples\n# in feature space (interpolates between existing minority samples).\n#\n# Applied ONLY to training data — val and test remain unmodified to reflect\n# real-world class distribution during evaluation.\nsm = SMOTE(random_state=42)\nX_train_res, y_train_res = sm.fit_resample(X_train_pca, y_train)\nprint(f\"\\nAfter SMOTE — shape: {X_train_res.shape}, dist: {np.bincount(y_train_res)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Models","metadata":{}},{"cell_type":"markdown","source":"\n\n","metadata":{},"attachments":{}},{"cell_type":"markdown","source":"# TRAIN","metadata":{}},{"cell_type":"code","source":"# ── Define all classifiers with tuned hyperparameters ──\nmodels = {\n    # SVM with RBF kernel: effective for non-linearly separable features.\n    # C=10 allows some misclassification; class_weight='balanced' gives\n    # higher penalty for misclassifying the minority class (pneumonia).\n    'SVM': SVC(\n        C=10,\n        gamma='scale',\n        kernel='rbf',\n        probability=True,      # Required to compute ROC-AUC (needs predict_proba)\n        class_weight='balanced'\n    ),\n\n    # Logistic Regression: fast linear baseline.\n    # max_iter=1000 ensures convergence on this feature size.\n    'Logistic Reg': LogisticRegression(max_iter=1000, random_state=42),\n\n    # Naive Bayes: assumes feature independence; very fast.\n    # Useful as a lightweight lower-bound baseline.\n    'Naive Bayes': GaussianNB(),\n\n    # Random Forest: ensemble of decision trees.\n    # max_depth=15 prevents overfitting; class_weight='balanced' handles imbalance.\n    'Random Forest': RandomForestClassifier(\n        n_estimators=300,\n        max_depth=15,\n        min_samples_split=5,\n        class_weight='balanced'\n    ),\n\n    # XGBoost: gradient boosting — often the strongest classical ML model.\n    # Low learning_rate=0.03 with many trees prevents overfitting.\n    # subsample + colsample_bytree add stochastic regularization.\n    'XGBoost': XGBClassifier(\n        n_estimators=300,\n        max_depth=6,\n        learning_rate=0.03,\n        subsample=0.8,\n        colsample_bytree=0.8,\n        eval_metric='logloss'\n    ),\n}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results = {}\n\nprint(\"\\n\" + \"=\"*75)\nprint(f\"{'Model':<18} {'Val-F1':>7} {'Val-AUC':>8} {'Test-Acc':>9} {'Test-Rec':>9} {'Test-F1':>8} {'Test-AUC':>9}\")\nprint(\"=\"*75)\n\nfor name, clf in models.items():\n    # ── Training ──\n    # Train on SMOTE-resampled data so the model sees balanced classes\n    clf.fit(X_train_res, y_train_res)\n\n    # ── Validation ──\n    # Val set is unmodified (real distribution); used only for early model comparison\n    y_val_pred  = clf.predict(X_val_pca)\n    y_val_proba = clf.predict_proba(X_val_pca)[:, 1]  # Probability of pneumonia class\n    val_f1  = f1_score(y_val, y_val_pred)\n    val_auc = roc_auc_score(y_val, y_val_proba)\n\n    # ── Test evaluation ──\n    # Final metrics on the held-out test set (never seen during training or tuning)\n    y_pred  = clf.predict(X_test_pca)\n    y_proba = clf.predict_proba(X_test_pca)[:, 1]\n    acc  = accuracy_score(y_test, y_pred)\n    prec = precision_score(y_test, y_pred)   # TP / (TP + FP)\n    rec  = recall_score(y_test, y_pred)      # TP / (TP + FN) — clinically critical\n    f1   = f1_score(y_test, y_pred)          # Harmonic mean of precision and recall\n    auc  = roc_auc_score(y_test, y_proba)    # Area under ROC curve\n\n    # Store all metrics and predictions for later visualization\n    results[name] = dict(acc=acc, precision=prec, recall=rec, f1=f1, auc=auc,\n                         val_f1=val_f1, val_auc=val_auc,\n                         y_pred=y_pred, y_proba=y_proba)\n\n    print(f\"{name:<18} {val_f1:7.3f} {val_auc:8.3f} {acc:9.3f} {rec:9.3f} {f1:8.3f} {auc:9.3f}\")\n\nprint(\"=\"*75)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# DETAILED TEST REPORTS","metadata":{}},{"cell_type":"code","source":"# Full per-class classification report for each model.\n# Shows precision, recall, F1 and support separately for Normal and Pneumonia.\n# This is more informative than a single accuracy number,\n# especially for imbalanced datasets.\nfor name, res in results.items():\n    print(f\"\\n── {name} ──\")\n    print(classification_report(y_test, res['y_pred'],\n                                 target_names=['Normal', 'Pneumonia']))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# CONFUSION MATRICES","metadata":{}},{"cell_type":"code","source":"# Plot a confusion matrix for every model side-by-side.\n# Confusion matrix shows:\n#   - True Negatives  (top-left):  Correctly predicted Normal\n#   - False Positives (top-right): Normal predicted as Pneumonia (false alarm)\n#   - False Negatives (bottom-left): Pneumonia missed (most dangerous error clinically)\n#   - True Positives  (bottom-right): Correctly predicted Pneumonia\nfig, axes = plt.subplots(2, 3, figsize=(15, 9))\naxes = axes.flatten()\nfor ax, (name, res) in zip(axes, results.items()):\n    cm = confusion_matrix(y_test, res['y_pred'])\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', ax=ax,\n                xticklabels=['Normal', 'Pneumonia'],\n                yticklabels=['Normal', 'Pneumonia'])\n    ax.set_title(name); ax.set_xlabel('Predicted'); ax.set_ylabel('Actual')\naxes[-1].axis('off')  # Hide the unused 6th subplot\nplt.suptitle('Confusion Matrices (Test Set)', fontsize=14)\nplt.tight_layout(); plt.savefig('confusion_matrices.png', dpi=150); plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ROC CURVES","metadata":{}},{"cell_type":"code","source":"# ROC (Receiver Operating Characteristic) curves plot True Positive Rate vs\n# False Positive Rate at all possible classification thresholds.\n# AUC (Area Under Curve) summarizes the curve: 1.0 = perfect, 0.5 = random.\n# Plotting all models together makes it easy to compare their discrimination ability.\nplt.figure(figsize=(8, 6))\nfor name, res in results.items():\n    fpr, tpr, _ = roc_curve(y_test, res['y_proba'])\n    plt.plot(fpr, tpr, label=f\"{name} (AUC={res['auc']:.3f})\")\nplt.plot([0,1],[0,1],'k--')  # Diagonal = random classifier\nplt.xlabel('FPR'); plt.ylabel('TPR')\nplt.title('ROC Curves – Test Set'); plt.legend(loc='lower right')\nplt.tight_layout(); plt.savefig('roc_curves.png', dpi=150); plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# SUMMARY BAR CHART","metadata":{}},{"cell_type":"code","source":"# Aggregate all five test metrics into a single comparison bar chart.\n# Makes it easy to see which model is best overall vs best on recall vs best on AUC.\nmetrics = ['acc', 'precision', 'recall', 'f1', 'auc']\nsummary = pd.DataFrame({name: [res[m] for m in metrics]\n                        for name, res in results.items()}, index=metrics)\nsummary.T.plot(kind='bar', figsize=(12, 5), ylim=(0, 1.05))\nplt.title('Model Comparison (Test Set)'); plt.ylabel('Score')\nplt.xticks(rotation=20); plt.legend(loc='lower right')\nplt.tight_layout(); plt.savefig('model_comparison.png', dpi=150); plt.show()\n\n# Print the single best-performing model for each key metric\nprint(\"\\n── Best Models ──\")\nprint(\"Best F1:     \", max(results, key=lambda n: results[n]['f1']))\nprint(\"Best Recall: \", max(results, key=lambda n: results[n]['recall']))\nprint(\"Best AUC:    \", max(results, key=lambda n: results[n]['auc']))","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}