{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Step 1: Setup & Data Acquisition","metadata":{}},{"cell_type":"markdown","source":"## Install Needed Libraries","metadata":{}},{"cell_type":"code","source":"!pip install -q pydicom scikit-image scikit-learn \\\n               xgboost imbalanced-learn albumentations","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:55:01.863140Z","iopub.execute_input":"2026-05-09T13:55:01.863501Z","iopub.status.idle":"2026-05-09T13:55:06.663950Z","shell.execute_reply.started":"2026-05-09T13:55:01.863473Z","shell.execute_reply":"2026-05-09T13:55:06.662694Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Set Path","metadata":{}},{"cell_type":"code","source":"import os\n\nDATA_DIR  = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\nTRAIN_DIR = f'{DATA_DIR}/stage_2_train_images'\nOUTPUT_DIR = '/kaggle/working'\n\nprint('📁 Dataset contents:')\nfor item in sorted(os.listdir(DATA_DIR)):\n    print(f'   {item}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:55:06.666019Z","iopub.execute_input":"2026-05-09T13:55:06.666453Z","iopub.status.idle":"2026-05-09T13:55:06.673359Z","shell.execute_reply.started":"2026-05-09T13:55:06.666413Z","shell.execute_reply":"2026-05-09T13:55:06.672395Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load and Preview Data","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\ndf = pd.read_csv(f'{DATA_DIR}/stage_2_train_labels.csv')\nprint('Shape:', df.shape)\nprint(df.head(8))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:55:06.674463Z","iopub.execute_input":"2026-05-09T13:55:06.674774Z","iopub.status.idle":"2026-05-09T13:55:07.641193Z","shell.execute_reply.started":"2026-05-09T13:55:06.674740Z","shell.execute_reply":"2026-05-09T13:55:07.640226Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Class Distribution","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\npatients = df.groupby('patientId')['Target'].max().reset_index()\ncounts   = patients['Target'].value_counts()\n\nprint(f'Normal    (0): {counts[0]:,} ({counts[0]/len(patients)*100:.1f}%)')\nprint(f'Pneumonia (1): {counts[1]:,} ({counts[1]/len(patients)*100:.1f}%)')\nprint(f'Total        : {len(patients):,}')\n\nfig, ax = plt.subplots(figsize=(5,3))\nax.bar(['Normal','Pneumonia'], [counts[0], counts[1]],\n       color=['steelblue','tomato'], edgecolor='white', width=0.5)\nax.set_title('Class Distribution — RSNA Dataset')\nax.set_ylabel('Patients')\nfor i, v in enumerate([counts[0], counts[1]]):\n    ax.text(i, v+50, f'{v:,}', ha='center', fontweight='bold')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:55:10.626630Z","iopub.execute_input":"2026-05-09T13:55:10.627433Z","iopub.status.idle":"2026-05-09T13:55:10.769908Z","shell.execute_reply.started":"2026-05-09T13:55:10.627400Z","shell.execute_reply":"2026-05-09T13:55:10.769110Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualize Sample X-Rays with Bounding Boxes","metadata":{}},{"cell_type":"code","source":"import pydicom\nimport numpy as np\nimport matplotlib.patches as patches\n\ndef apply_window(img, level=-500, width=1500):\n    low  = level - width // 2\n    high = level + width // 2\n    img  = np.clip(img, low, high)\n    img  = (img - low) / (high - low)\n    return img\n\nfig, axes = plt.subplots(2, 4, figsize=(16, 8))\naxes = axes.flatten()\n\nnormal_ids    = patients[patients['Target']==0]['patientId'].sample(4, random_state=42).values\npneumonia_ids = patients[patients['Target']==1]['patientId'].sample(4, random_state=42).values\nsample_ids    = list(normal_ids) + list(pneumonia_ids)\nlabels_list   = ['Normal']*4 + ['Pneumonia']*4\n\nfor idx, (pid, label) in enumerate(zip(sample_ids, labels_list)):\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = apply_window(dcm.pixel_array.astype(np.float32))\n\n    axes[idx].imshow(img, cmap='gray')\n    axes[idx].set_title(label,\n                        color='tomato' if label=='Pneumonia' else 'steelblue',\n                        fontweight='bold')\n    axes[idx].axis('off')\n\n    if label == 'Pneumonia':\n        for _, row in df[df['patientId']==pid][['x','y','width','height']].dropna().iterrows():\n            axes[idx].add_patch(patches.Rectangle(\n                (row['x'], row['y']), row['width'], row['height'],\n                linewidth=2, edgecolor='red', facecolor='none'))\n\nplt.suptitle('Sample Chest X-Rays — red boxes = pneumonia regions', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:55:15.665939Z","iopub.execute_input":"2026-05-09T13:55:15.666729Z","iopub.status.idle":"2026-05-09T13:55:18.887312Z","shell.execute_reply.started":"2026-05-09T13:55:15.666697Z","shell.execute_reply":"2026-05-09T13:55:18.886159Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Save Global Config","metadata":{}},{"cell_type":"code","source":"import json\n\nDATA_DIR  = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\nTRAIN_DIR = f'{DATA_DIR}/stage_2_train_images'\nOUTPUT_DIR = '/kaggle/working'\n\nCONFIG = {\n    'data_dir'    : DATA_DIR,\n    'train_dcm'   : TRAIN_DIR,\n    'labels_csv'  : f'{DATA_DIR}/stage_2_train_labels.csv',\n    'output_dir'  : OUTPUT_DIR,\n    'img_size'    : 512,\n    'window_level': -500,\n    'window_width': 1500,\n    'train_ratio' : 0.70,\n    'val_ratio'   : 0.15,\n    'test_ratio'  : 0.15,\n    'random_seed' : 42,\n}\n\nprint('✅ Config ready.')\nprint(json.dumps(CONFIG, indent=2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:55:20.818933Z","iopub.execute_input":"2026-05-09T13:55:20.819915Z","iopub.status.idle":"2026-05-09T13:55:20.826816Z","shell.execute_reply.started":"2026-05-09T13:55:20.819881Z","shell.execute_reply":"2026-05-09T13:55:20.826041Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 2: EDA & Visualization","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport pydicom\nimport cv2\nimport os\nfrom tqdm import tqdm\n\nDATA_DIR  = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\nTRAIN_DIR = f'{DATA_DIR}/stage_2_train_images'\n\ndf       = pd.read_csv(f'{DATA_DIR}/stage_2_train_labels.csv')\ndf_class = pd.read_csv(f'{DATA_DIR}/stage_2_detailed_class_info.csv')\npatients = df.groupby('patientId')['Target'].max().reset_index()\n\nprint('Labels shape    :', df.shape)\nprint('Class info shape:', df_class.shape)\nprint(df_class['class'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:55:25.903439Z","iopub.execute_input":"2026-05-09T13:55:25.903742Z","iopub.status.idle":"2026-05-09T13:55:26.328266Z","shell.execute_reply.started":"2026-05-09T13:55:25.903717Z","shell.execute_reply":"2026-05-09T13:55:26.327152Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Missing Values","metadata":{}},{"cell_type":"code","source":"print('=== Labels CSV ===')\nprint(df.isnull().sum())\nprint(f'\\nTotal missing: {df.isnull().sum().sum()}')\n\nprint('\\n=== Class Info CSV ===')\nprint(df_class.isnull().sum())\nprint(f'\\nTotal missing: {df_class.isnull().sum().sum()}')\n\nprint('\\n=== Missing Bounding Boxes (Normal patients have NaN boxes) ===')\nprint(f'Rows with NaN boxes : {df[df[\"x\"].isnull()].shape[0]:,}')\nprint(f'Rows with boxes     : {df[df[\"x\"].notna()].shape[0]:,}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:55:31.035988Z","iopub.execute_input":"2026-05-09T13:55:31.036373Z","iopub.status.idle":"2026-05-09T13:55:31.061643Z","shell.execute_reply.started":"2026-05-09T13:55:31.036342Z","shell.execute_reply":"2026-05-09T13:55:31.060764Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Patient Age & Gender Distribution","metadata":{}},{"cell_type":"code","source":"detail_df = df_class.drop_duplicates('patientId').copy()\n\n# Check available columns\nprint('Available columns:', detail_df.columns.tolist())\nprint(detail_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:55:37.046102Z","iopub.execute_input":"2026-05-09T13:55:37.047035Z","iopub.status.idle":"2026-05-09T13:55:37.059420Z","shell.execute_reply.started":"2026-05-09T13:55:37.047002Z","shell.execute_reply":"2026-05-09T13:55:37.058638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\n\nrecords = []\nsample_pids = patients.sample(300, random_state=42)['patientId'].values\n\nfor pid in tqdm(sample_pids):\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    target = patients[patients['patientId']==pid]['Target'].values[0]\n    \n    age    = getattr(dcm, 'PatientAge',  None)\n    gender = getattr(dcm, 'PatientSex',  None)\n    \n    records.append({\n        'patientId': pid,\n        'age'      : age,\n        'gender'   : gender,\n        'target'   : target\n    })\n\ndf_demo = pd.DataFrame(records)\nprint('Sample demographics:')\nprint(df_demo.head(10))\nprint('\\nGender counts:')\nprint(df_demo['gender'].value_counts())\nprint('\\nAge sample:')\nprint(df_demo['age'].value_counts().head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:55:40.989980Z","iopub.execute_input":"2026-05-09T13:55:40.990599Z","iopub.status.idle":"2026-05-09T13:55:44.113479Z","shell.execute_reply.started":"2026-05-09T13:55:40.990568Z","shell.execute_reply":"2026-05-09T13:55:44.112507Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Plot Age & Gender","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize=(16, 4))\n\n# Gender distribution\ngender_counts = df_demo['gender'].value_counts()\naxes[0].bar(gender_counts.index, gender_counts.values,\n            color=['steelblue','tomato'], edgecolor='white', width=0.4)\naxes[0].set_title('Patient Gender Distribution')\naxes[0].set_ylabel('Count')\naxes[0].set_xlabel('Gender')\nfor i, v in enumerate(gender_counts.values):\n    axes[0].text(i, v+1, f'{v:,}', ha='center', fontweight='bold')\n\n# Age distribution by class\ndf_demo['age'] = pd.to_numeric(df_demo['age'], errors='coerce')\nnormal_ages    = df_demo[df_demo['target']==0]['age'].dropna()\npneumonia_ages = df_demo[df_demo['target']==1]['age'].dropna()\n\naxes[1].hist(normal_ages,    bins=20, alpha=0.7, color='steelblue', label='Normal')\naxes[1].hist(pneumonia_ages, bins=20, alpha=0.7, color='tomato',    label='Pneumonia')\naxes[1].set_title('Age Distribution by Class')\naxes[1].set_xlabel('Age (years)')\naxes[1].set_ylabel('Count')\naxes[1].legend()\n\n# Gender by class\ngender_class = df_demo.groupby(['gender','target']).size().unstack(fill_value=0)\ngender_class.plot(kind='bar', ax=axes[2],\n                  color=['steelblue','tomato'], edgecolor='white')\naxes[2].set_title('Gender by Class')\naxes[2].set_xlabel('Gender')\naxes[2].set_ylabel('Count')\naxes[2].legend(['Normal','Pneumonia'])\naxes[2].tick_params(axis='x', rotation=0)\n\nplt.suptitle('Patient Demographics — RSNA Dataset', fontsize=13)\nplt.tight_layout()\nplt.show()\n\nprint(f'Male   — Total: {gender_counts.get(\"M\",0):,}')\nprint(f'Female — Total: {gender_counts.get(\"F\",0):,}')\nprint(f'\\nNormal    — Mean age: {normal_ages.mean():.1f} yrs')\nprint(f'Pneumonia — Mean age: {pneumonia_ages.mean():.1f} yrs')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:56:03.795309Z","iopub.execute_input":"2026-05-09T13:56:03.795931Z","iopub.status.idle":"2026-05-09T13:56:04.381950Z","shell.execute_reply.started":"2026-05-09T13:56:03.795898Z","shell.execute_reply":"2026-05-09T13:56:04.381197Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Class Distribution","metadata":{}},{"cell_type":"code","source":"class_counts = df_class.drop_duplicates('patientId')['class'].value_counts()\n\nfig, axes = plt.subplots(1, 2, figsize=(12, 4))\n\n# Bar chart\naxes[0].bar(class_counts.index, class_counts.values,\n            color=['steelblue', 'tomato', 'orange'], edgecolor='white')\naxes[0].set_title('Detailed Class Distribution')\naxes[0].set_ylabel('Number of Patients')\naxes[0].tick_params(axis='x', rotation=15)\nfor i, v in enumerate(class_counts.values):\n    axes[0].text(i, v+50, f'{v:,}', ha='center', fontweight='bold')\n\n# Pie chart\naxes[1].pie(class_counts.values, labels=class_counts.index,\n            colors=['steelblue', 'tomato', 'orange'],\n            autopct='%1.1f%%', startangle=90)\naxes[1].set_title('Class Proportions')\n\nplt.suptitle('RSNA Dataset — Class Distribution', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:56:09.277305Z","iopub.execute_input":"2026-05-09T13:56:09.277768Z","iopub.status.idle":"2026-05-09T13:56:09.513101Z","shell.execute_reply.started":"2026-05-09T13:56:09.277737Z","shell.execute_reply":"2026-05-09T13:56:09.512409Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Bounding Box Statistics","metadata":{}},{"cell_type":"code","source":"df_pos = df[df['Target'] == 1].copy()\n\nprint('Total bounding boxes      :', len(df_pos))\nprint('Patients with pneumonia   :', df_pos['patientId'].nunique())\nprint('Avg boxes per patient     :', round(len(df_pos)/df_pos['patientId'].nunique(), 2))\nprint()\nprint('Bounding Box Size Stats:')\nprint(df_pos[['width','height']].describe().round(1))\n\n# Box count per patient\nbox_counts = df_pos.groupby('patientId').size()\n\nfig, axes = plt.subplots(1, 3, figsize=(15, 4))\n\n# Box count distribution\naxes[0].hist(box_counts.values, bins=range(1, box_counts.max()+2),\n             color='tomato', edgecolor='white', align='left')\naxes[0].set_title('Number of Boxes per Pneumonia Patient')\naxes[0].set_xlabel('Number of Bounding Boxes')\naxes[0].set_ylabel('Count')\n\n# Box width distribution\naxes[1].hist(df_pos['width'].dropna(), bins=40, color='steelblue', edgecolor='white')\naxes[1].set_title('Bounding Box Width Distribution')\naxes[1].set_xlabel('Width (pixels)')\naxes[1].set_ylabel('Count')\n\n# Box height distribution\naxes[2].hist(df_pos['height'].dropna(), bins=40, color='purple', edgecolor='white')\naxes[2].set_title('Bounding Box Height Distribution')\naxes[2].set_xlabel('Height (pixels)')\naxes[2].set_ylabel('Count')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:56:14.801315Z","iopub.execute_input":"2026-05-09T13:56:14.801982Z","iopub.status.idle":"2026-05-09T13:56:15.352714Z","shell.execute_reply.started":"2026-05-09T13:56:14.801952Z","shell.execute_reply":"2026-05-09T13:56:15.351745Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Bounding Box Area Distribution","metadata":{}},{"cell_type":"code","source":"df_pos['area'] = df_pos['width'] * df_pos['height']\n\nfig, axes = plt.subplots(1, 2, figsize=(12, 4))\n\n# Area histogram\naxes[0].hist(df_pos['area'].dropna(), bins=40, color='steelblue', edgecolor='white')\naxes[0].set_title('Bounding Box Area Distribution')\naxes[0].set_xlabel('Area (pixels²)')\naxes[0].set_ylabel('Count')\naxes[0].axvline(df_pos['area'].mean(), color='red', linestyle='--',\n                label=f'Mean: {df_pos[\"area\"].mean():,.0f}')\naxes[0].legend()\n\n# Width vs Height scatter\naxes[1].scatter(df_pos['width'], df_pos['height'],\n                alpha=0.2, s=10, color='tomato')\naxes[1].set_title('Bounding Box Width vs Height')\naxes[1].set_xlabel('Width (pixels)')\naxes[1].set_ylabel('Height (pixels)')\n\nprint(f'Mean area  : {df_pos[\"area\"].mean():,.0f} px²')\nprint(f'Median area: {df_pos[\"area\"].median():,.0f} px²')\nprint(f'Min area   : {df_pos[\"area\"].min():,.0f} px²')\nprint(f'Max area   : {df_pos[\"area\"].max():,.0f} px²')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:56:40.349862Z","iopub.execute_input":"2026-05-09T13:56:40.350583Z","iopub.status.idle":"2026-05-09T13:56:40.737370Z","shell.execute_reply.started":"2026-05-09T13:56:40.350550Z","shell.execute_reply":"2026-05-09T13:56:40.736533Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Image Metadata Analysis","metadata":{}},{"cell_type":"code","source":"import pydicom\n\nsample = pydicom.dcmread(f'{TRAIN_DIR}/{patients[\"patientId\"].iloc[0]}.dcm')\nprint('Image size  :', sample.Rows, 'x', sample.Columns)\nprint('Bits        :', sample.BitsAllocated)\nprint('Modality    :', sample.Modality)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:56:45.246046Z","iopub.execute_input":"2026-05-09T13:56:45.246432Z","iopub.status.idle":"2026-05-09T13:56:45.261381Z","shell.execute_reply.started":"2026-05-09T13:56:45.246401Z","shell.execute_reply":"2026-05-09T13:56:45.260329Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Pixel Intensity Analysis","metadata":{}},{"cell_type":"code","source":"print('Analyzing pixel intensities from 100 samples (50 normal, 50 pneumonia)...')\n\ndef apply_window(img, level=-500, width=1500):\n    low  = level - width // 2\n    high = level + width // 2\n    img  = np.clip(img, low, high)\n    img  = (img - low) / (high - low)\n    return img\n\nnormal_pids    = patients[patients['Target']==0]['patientId'].sample(50, random_state=42).values\npneumonia_pids = patients[patients['Target']==1]['patientId'].sample(50, random_state=42).values\n\nnormal_means, pneumonia_means = [], []\nnormal_stds,  pneumonia_stds  = [], []\n\nfor pid in tqdm(normal_pids):\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = apply_window(dcm.pixel_array.astype(np.float32))\n    normal_means.append(img.mean())\n    normal_stds.append(img.std())\n\nfor pid in tqdm(pneumonia_pids):\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = apply_window(dcm.pixel_array.astype(np.float32))\n    pneumonia_means.append(img.mean())\n    pneumonia_stds.append(img.std())\n\nfig, axes = plt.subplots(1, 2, figsize=(12, 4))\n\n# Mean intensity\naxes[0].hist(normal_means,    bins=20, alpha=0.7, color='steelblue', label='Normal')\naxes[0].hist(pneumonia_means, bins=20, alpha=0.7, color='tomato',    label='Pneumonia')\naxes[0].set_title('Mean Pixel Intensity Distribution')\naxes[0].set_xlabel('Mean Intensity')\naxes[0].set_ylabel('Count')\naxes[0].legend()\n\n# Std intensity\naxes[1].hist(normal_stds,    bins=20, alpha=0.7, color='steelblue', label='Normal')\naxes[1].hist(pneumonia_stds, bins=20, alpha=0.7, color='tomato',    label='Pneumonia')\naxes[1].set_title('Pixel Intensity Std Distribution')\naxes[1].set_xlabel('Std of Intensity')\naxes[1].set_ylabel('Count')\naxes[1].legend()\n\nplt.tight_layout()\nplt.show()\n\nprint(f'\\nNormal    — Mean: {np.mean(normal_means):.3f}, Std: {np.mean(normal_stds):.3f}')\nprint(f'Pneumonia — Mean: {np.mean(pneumonia_means):.3f}, Std: {np.mean(pneumonia_stds):.3f}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:56:48.766427Z","iopub.execute_input":"2026-05-09T13:56:48.767065Z","iopub.status.idle":"2026-05-09T13:56:51.092348Z","shell.execute_reply.started":"2026-05-09T13:56:48.767035Z","shell.execute_reply":"2026-05-09T13:56:51.091581Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train / Val / Test","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nSEED = 42\n\n# First split: train vs temp (val+test)\ntrain_df, temp_df = train_test_split(patients, test_size=0.30,\n                                     stratify=patients['Target'],\n                                     random_state=SEED)\n\n# Second split: val vs test\nval_df, test_df = train_test_split(temp_df, test_size=0.50,\n                                   stratify=temp_df['Target'],\n                                   random_state=SEED)\n\nprint(f'Train : {len(train_df):,} ({len(train_df)/len(patients)*100:.1f}%)')\nprint(f'Val   : {len(val_df):,}  ({len(val_df)/len(patients)*100:.1f}%)')\nprint(f'Test  : {len(test_df):,}  ({len(test_df)/len(patients)*100:.1f}%)')\n\n# Verify stratification\nfor name, split in [('Train', train_df), ('Val', val_df), ('Test', test_df)]:\n    pos = split['Target'].sum()\n    print(f'{name} — Pneumonia: {pos} ({pos/len(split)*100:.1f}%)')\n\n# Save splits\ntrain_df.to_csv('/kaggle/working/train_split.csv', index=False)\nval_df.to_csv('/kaggle/working/val_split.csv',   index=False)\ntest_df.to_csv('/kaggle/working/test_split.csv',  index=False)\nprint('\\n✅ Splits saved to /kaggle/working/')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:56:56.156985Z","iopub.execute_input":"2026-05-09T13:56:56.158404Z","iopub.status.idle":"2026-05-09T13:56:57.792606Z","shell.execute_reply.started":"2026-05-09T13:56:56.158367Z","shell.execute_reply":"2026-05-09T13:56:57.791831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport pydicom\nimport cv2\nimport os\nfrom tqdm import tqdm\n\nDATA_DIR  = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\nTRAIN_DIR = f'{DATA_DIR}/stage_2_train_images'\nOUTPUT_DIR = '/kaggle/working'\ndf       = pd.read_csv(f'{DATA_DIR}/stage_2_train_labels.csv')\npatients = df.groupby('patientId')['Target'].max().reset_index()\n\ntrain_df = pd.read_csv(f'{OUTPUT_DIR}/train_split.csv')\nval_df   = pd.read_csv(f'{OUTPUT_DIR}/val_split.csv')\ntest_df  = pd.read_csv(f'{OUTPUT_DIR}/test_split.csv')\n\nprint('✅ Imports done.')\nprint(f'Train: {len(train_df):,} | Val: {len(val_df):,} | Test: {len(test_df):,}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:57:53.902948Z","iopub.execute_input":"2026-05-09T13:57:53.903387Z","iopub.status.idle":"2026-05-09T13:57:53.986026Z","shell.execute_reply.started":"2026-05-09T13:57:53.903355Z","shell.execute_reply":"2026-05-09T13:57:53.985155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_dicom(pid):\n    \"\"\"Load raw DICOM pixel array.\"\"\"\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = dcm.pixel_array.astype(np.float32)\n    return img\n\ndef apply_window(img, level=-500, width=1500):\n    \"\"\"Apply lung windowing to enhance pulmonary structures.\"\"\"\n    low  = level - width // 2\n    high = level + width // 2\n    img  = np.clip(img, low, high)\n    img  = (img - low) / (high - low)\n    return img\n\ndef normalize(img):\n    \"\"\"Normalize image to [0, 1].\"\"\"\n    img = img - img.min()\n    if img.max() > 0:\n        img = img / img.max()\n    return img\ndef enhance_contrast(img):\n\n    img_uint8 = (img * 255).astype(np.uint8)\n\n    clahe = cv2.createCLAHE(\n        clipLimit=2.0,\n        tileGridSize=(8,8)\n    )\n\n    enhanced = clahe.apply(img_uint8)\n\n    return enhanced / 255.0\n    \ndef denoise(img):\n    \"\"\"Apply Gaussian blur for noise reduction.\"\"\"\n    return cv2.GaussianBlur(img, (3, 3), 0)\n    \ndef resize_img(img, size=512):\n    \"\"\"Resize image to fixed size.\"\"\"\n    return cv2.resize(img, (size, size), interpolation=cv2.INTER_AREA)\n\ndef preprocess(pid, size=512):\n    \"\"\"Full preprocessing pipeline for one patient.\"\"\"\n    img = load_dicom(pid)       # Step 1: Load DICOM\n    img = normalize(img)        # Step 2: Normalize to [0,1] first\n    img = enhance_contrast(img)          # Step 3: enhance_contrast\n    img = resize_img(img, size) # Step 4: Resize to 512x512\n    return img\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:59:08.539555Z","iopub.execute_input":"2026-05-09T13:59:08.540465Z","iopub.status.idle":"2026-05-09T13:59:08.549810Z","shell.execute_reply.started":"2026-05-09T13:59:08.540434Z","shell.execute_reply":"2026-05-09T13:59:08.549075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pid = patients['patientId'].iloc[0]\n\nraw     = load_dicom(pid)\nnormed  = normalize(raw)\nenhanced = enhance_contrast(normed)\nresized  = resize_img(enhanced)\n\nprint(f'Raw      — shape: {raw.shape},     min: {raw.min():.1f},   max: {raw.max():.1f}')\nprint(f'Normed   — shape: {normed.shape},   min: {normed.min():.2f},  max: {normed.max():.2f}')\nprint(f'Enhanced — shape: {enhanced.shape}, min: {enhanced.min():.2f},  max: {enhanced.max():.2f}')\nprint(f'Resized  — shape: {resized.shape},  min: {resized.min():.2f},  max: {resized.max():.2f}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:59:22.500223Z","iopub.execute_input":"2026-05-09T13:59:22.501164Z","iopub.status.idle":"2026-05-09T13:59:22.550034Z","shell.execute_reply.started":"2026-05-09T13:59:22.501131Z","shell.execute_reply":"2026-05-09T13:59:22.549213Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import albumentations as A\n\naugment = A.Compose([\n    A.HorizontalFlip(p=0.5),\n    A.Rotate(limit=10, p=0.3),\n    A.Affine(translate_percent=0.1, p=0.3),\n    A.RandomBrightnessContrast(brightness_limit=0.2,\n                               contrast_limit=0.2, p=0.4),\n])\n\ndef augment_img(img):\n    \"\"\"Apply augmentation to a single image.\"\"\"\n    img_uint8 = (img * 255).astype(np.uint8)\n    augmented = augment(image=img_uint8)['image']\n    return augmented.astype(np.float32) / 255.0\n\nprint('✅ Augmentation pipeline defined.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:59:39.572347Z","iopub.execute_input":"2026-05-09T13:59:39.572980Z","iopub.status.idle":"2026-05-09T13:59:45.874860Z","shell.execute_reply.started":"2026-05-09T13:59:39.572949Z","shell.execute_reply":"2026-05-09T13:59:45.873848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_pid = patients[patients['Target']==1]['patientId'].sample(1, random_state=7).values[0]\nimg = preprocess(sample_pid)\n\nfig, axes = plt.subplots(2, 4, figsize=(16, 8))\naxes = axes.flatten()\n\n# Original\naxes[0].imshow(img, cmap='gray')\naxes[0].set_title('Original (preprocessed)')\naxes[0].axis('off')\n\n# 7 augmented versions\nfor i in range(1, 8):\n    aug = augment_img(img)\n    axes[i].imshow(aug, cmap='gray')\n    axes[i].set_title(f'Augmented #{i}')\n    axes[i].axis('off')\n\nplt.suptitle('Data Augmentation Examples', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T13:59:58.231329Z","iopub.execute_input":"2026-05-09T13:59:58.232423Z","iopub.status.idle":"2026-05-09T13:59:59.616623Z","shell.execute_reply.started":"2026-05-09T13:59:58.232390Z","shell.execute_reply":"2026-05-09T13:59:59.615571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 4, figsize=(16, 12))\n\nsample_pids = patients.sample(4, random_state=10)['patientId'].values\nfor idx, pid in enumerate(sample_pids):\n    raw      = load_dicom(pid)\n    windowed = apply_window(raw)\n    resized  = resize_img(normalize(windowed))\n\n    # Row 1: Raw\n    axes[0][idx].imshow(raw, cmap='gray')\n    axes[0][idx].set_title(f'Raw DICOM\\n{raw.shape}')\n    axes[0][idx].axis('off')\n\n    # Row 2: After windowing\n    axes[1][idx].imshow(windowed, cmap='gray')\n    axes[1][idx].set_title(f'After Windowing\\n[{windowed.min():.2f}, {windowed.max():.2f}]')\n    axes[1][idx].axis('off')\n\n    # Row 3: After resize\n    axes[2][idx].imshow(resized, cmap='gray')\n    axes[2][idx].set_title(f'Final (512x512)\\n[{resized.min():.2f}, {resized.max():.2f}]')\n    axes[2][idx].axis('off')\n\nplt.suptitle('Preprocessing Pipeline: Raw → Windowed → Final', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T14:00:56.325148Z","iopub.execute_input":"2026-05-09T14:00:56.326064Z","iopub.status.idle":"2026-05-09T14:00:59.563049Z","shell.execute_reply.started":"2026-05-09T14:00:56.326029Z","shell.execute_reply":"2026-05-09T14:00:59.561997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport cv2\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import RobustScaler\nfrom sklearn.feature_selection import VarianceThreshold, SelectKBest, f_classif, mutual_info_classif\nfrom scipy import stats\nfrom skimage.feature import graycomatrix, graycoprops, local_binary_pattern\nfrom skimage.measure import shannon_entropy\nimport pywt  # Add this import\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Configuration\nIMG_SIZE = 512\nN_JOBS = -1\nRANDOM_STATE = 42\n\ndef load_and_preprocess(pid, train_dir):\n    \"\"\"Load and preprocess DICOM image with clinical windowing\"\"\"\n    dcm = pydicom.dcmread(f'{train_dir}/{pid}.dcm')\n    img = dcm.pixel_array.astype(np.float32)\n    \n    # Lung window (standard for pneumonia detection)\n    level, width = -500, 1500\n    low = level - width // 2\n    high = level + width // 2\n    img = np.clip(img, low, high)\n    img = (img - low) / (high - low)\n    \n    # Resize\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    return img\n\ndef extract_radiomic_features(img):\n    \"\"\"Extract comprehensive radiomic features from image\"\"\"\n    features = {}\n    \n    # 1. First-order statistics\n    features['mean'] = np.mean(img)\n    features['std'] = np.std(img)\n    features['skewness'] = stats.skew(img.flatten())\n    features['kurtosis'] = stats.kurtosis(img.flatten())\n    features['median'] = np.median(img)\n    features['iqr'] = np.percentile(img, 75) - np.percentile(img, 25)\n    features['entropy'] = shannon_entropy(img)\n    features['energy'] = np.sum(img ** 2)\n    \n    # Percentiles\n    for p in [10, 25, 75, 90]:\n        features[f'percentile_{p}'] = np.percentile(img, p)\n    \n    # 2. GLCM Features (Texture)\n    img_uint8 = (img * 255).astype(np.uint8)\n    glcm = graycomatrix(img_uint8, distances=[1], angles=[0, np.pi/4, np.pi/2, 3*np.pi/4], \n                       levels=256, symmetric=True, normed=True)\n    \n    for prop in ['contrast', 'dissimilarity', 'homogeneity', 'energy', 'correlation']:\n        for i, angle in enumerate(['0', '45', '90', '135']):\n            features[f'glcm_{prop}_{angle}'] = graycoprops(glcm, prop)[0, i]\n    \n    # 3. Local Binary Pattern (Texture descriptor)\n    lbp = local_binary_pattern(img, P=8, R=1, method='uniform')\n    lbp_hist, _ = np.histogram(lbp.ravel(), bins=np.arange(0, 11), density=True)\n    for i, val in enumerate(lbp_hist):\n        features[f'lbp_bin_{i}'] = val\n    \n    # 4. Shape features (from thresholded image)\n    threshold = np.percentile(img, 30)\n    binary = (img > threshold).astype(np.uint8)\n    contours, _ = cv2.findContours(binary, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n    \n    if contours:\n        largest_contour = max(contours, key=cv2.contourArea)\n        area = cv2.contourArea(largest_contour)\n        perimeter = cv2.arcLength(largest_contour, True)\n        if perimeter > 0:\n            features['shape_compactness'] = (4 * np.pi * area) / (perimeter ** 2)\n        features['shape_area_ratio'] = area / (IMG_SIZE ** 2)\n        \n        # Bounding box features\n        x, y, w, h = cv2.boundingRect(largest_contour)\n        features['shape_aspect_ratio'] = w / h if h > 0 else 1\n        features['shape_extent'] = area / (w * h) if w * h > 0 else 0\n    else:\n        features.update({\n            'shape_compactness': 0, 'shape_area_ratio': 0,\n            'shape_aspect_ratio': 1, 'shape_extent': 0\n        })\n    \n    # 5. Gradient features\n    grad_x = cv2.Sobel(img, cv2.CV_64F, 1, 0, ksize=3)\n    grad_y = cv2.Sobel(img, cv2.CV_64F, 0, 1, ksize=3)\n    grad_magnitude = np.sqrt(grad_x**2 + grad_y**2)\n    \n    features['gradient_mean'] = np.mean(grad_magnitude)\n    features['gradient_std'] = np.std(grad_magnitude)\n    features['gradient_entropy'] = shannon_entropy(grad_magnitude)\n    \n    # 6. Wavelet features (Haar-like)\n    coeffs = pywt.wavedec2(img, 'haar', level=2)\n    for level in range(1, 3):\n        cA, (cH, cV, cD) = coeffs[0], coeffs[level]\n        features[f'wavelet_cA_energy_lvl{level}'] = np.sum(cA ** 2)\n        features[f'wavelet_cH_energy_lvl{level}'] = np.sum(cH ** 2)\n        features[f'wavelet_cV_energy_lvl{level}'] = np.sum(cV ** 2)\n        features[f'wavelet_cD_energy_lvl{level}'] = np.sum(cD ** 2)\n    \n    return features\n\n# Extract features for all splits\ndef extract_features_for_split(df_split, split_name, train_dir):\n    \"\"\"Extract features for a data split with progress bar\"\"\"\n    features_list = []\n    labels_list = []\n    \n    for _, row in tqdm(df_split.iterrows(), total=len(df_split), desc=f'Extracting {split_name}'):\n        try:\n            img = load_and_preprocess(row['patientId'], train_dir)\n            features = extract_radiomic_features(img)\n            features_list.append(features)\n            labels_list.append(row['Target'])\n        except Exception as e:\n            print(f\"Error processing {row['patientId']}: {e}\")\n            continue\n    \n    X = pd.DataFrame(features_list)\n    y = np.array(labels_list)\n    \n    # Save\n    X.to_csv(f'/kaggle/working/X_{split_name}.csv', index=False)\n    np.save(f'/kaggle/working/y_{split_name}.npy', y)\n    \n    return X, y\n\n# Extract features (this will take time - run once)\nprint(\"=\"*60)\nprint(\"EXTRACTING RADIOMIC FEATURES\")\nprint(\"=\"*60)\n\n# Use your existing train_df, val_df, test_df from earlier\nX_train, y_train = extract_features_for_split(train_df, 'train', TRAIN_DIR)\nX_val, y_val = extract_features_for_split(val_df, 'val', TRAIN_DIR)\nX_test, y_test = extract_features_for_split(test_df, 'test', TRAIN_DIR)\n\nprint(f\"\\n✅ Feature extraction complete!\")\nprint(f\"Train shape: {X_train.shape}\")\nprint(f\"Val shape: {X_val.shape}\")\nprint(f\"Test shape: {X_test.shape}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T14:01:09.970200Z","iopub.execute_input":"2026-05-09T14:01:09.970569Z","iopub.status.idle":"2026-05-09T15:43:46.492369Z","shell.execute_reply.started":"2026-05-09T14:01:09.970541Z","shell.execute_reply":"2026-05-09T15:43:46.491663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================================\n# PART 2: FEATURE ENGINEERING & PREPROCESSING\n# RUN THIS AFTER FEATURE EXTRACTION\n# ============================================================================\n\nimport numpy as np\nimport pandas as pd\nfrom sklearn.preprocessing import RobustScaler, StandardScaler\nfrom sklearn.feature_selection import VarianceThreshold, SelectKBest, f_classif\nimport matplotlib.pyplot as plt\nimport warnings\nwarnings.filterwarnings('ignore')\n\nprint(\"\\n\" + \"=\"*70)\nprint(\"PART 2: FEATURE ENGINEERING & PREPROCESSING\")\nprint(\"=\"*70)\n\n# Load the features you just extracted\nX_train = pd.read_csv('/kaggle/working/X_train.csv')\nX_val = pd.read_csv('/kaggle/working/X_val.csv')\nX_test = pd.read_csv('/kaggle/working/X_test.csv')\n\n# Load labels\ny_train = np.load('/kaggle/working/y_train.npy')\ny_val = np.load('/kaggle/working/y_val.npy')\ny_test = np.load('/kaggle/working/y_test.npy')\n\nprint(f\"\\n📊 Data shapes after extraction:\")\nprint(f\"   X_train: {X_train.shape}\")\nprint(f\"   X_val: {X_val.shape}\")\nprint(f\"   X_test: {X_test.shape}\")\nprint(f\"   y_train: {y_train.shape} (Positives: {y_train.sum()})\")\nprint(f\"   y_val: {y_val.shape} (Positives: {y_val.sum()})\")\nprint(f\"   y_test: {y_test.shape} (Positives: {y_test.sum()})\")\n\n# Handle missing values\nprint(\"\\n1. Handling missing values...\")\nX_train = X_train.fillna(X_train.median())\nX_val = X_val.fillna(X_train.median())\nX_test = X_test.fillna(X_train.median())\n\n# Remove constant features\nprint(\"2. Removing constant features...\")\nselector_var = VarianceThreshold(threshold=0.01)\nX_train = selector_var.fit_transform(X_train)\nX_val = selector_var.transform(X_val)\nX_test = selector_var.transform(X_test)\nprint(f\"   Features after variance filter: {X_train.shape[1]}\")\n\n# Scale features\nprint(\"3. Scaling features...\")\nscaler = RobustScaler()  # Robust to outliers\nX_train = scaler.fit_transform(X_train)\nX_val = scaler.transform(X_val)\nX_test = scaler.transform(X_test)\n\n# Select top features\nprint(\"4. Selecting top features...\")\nn_features = min(100, X_train.shape[1])\nselector_kbest = SelectKBest(f_classif, k=n_features)\nX_train = selector_kbest.fit_transform(X_train, y_train)\nX_val = selector_kbest.transform(X_val)\nX_test = selector_kbest.transform(X_test)\nprint(f\"   Selected top {n_features} features\")\n\nprint(f\"\\n✅ Feature engineering complete!\")\nprint(f\"   Final training shape: {X_train.shape}\")\nprint(f\"   Final validation shape: {X_val.shape}\")\nprint(f\"   Final test shape: {X_test.shape}\")\n\n# Save processed features\nnp.save('/kaggle/working/X_train_processed.npy', X_train)\nnp.save('/kaggle/working/X_val_processed.npy', X_val)\nnp.save('/kaggle/working/X_test_processed.npy', X_test)\nnp.save('/kaggle/working/y_train.npy', y_train)\nnp.save('/kaggle/working/y_val.npy', y_val)\nnp.save('/kaggle/working/y_test.npy', y_test)\nprint(\"\\n✅ Processed features saved to /kaggle/working/\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T15:47:16.734669Z","iopub.execute_input":"2026-05-09T15:47:16.735634Z","iopub.status.idle":"2026-05-09T15:47:17.177606Z","shell.execute_reply.started":"2026-05-09T15:47:16.735601Z","shell.execute_reply":"2026-05-09T15:47:17.176793Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================================\n# PART 3: HANDLE CLASS IMBALANCE WITH SMOTE\n# ============================================================================\n\nfrom imblearn.over_sampling import SMOTE\nfrom collections import Counter\n\nprint(\"\\n\" + \"=\"*70)\nprint(\"PART 3: HANDLING CLASS IMBALANCE\")\nprint(\"=\"*70)\n\n# Load processed features\nX_train = np.load('/kaggle/working/X_train_processed.npy')\ny_train = np.load('/kaggle/working/y_train.npy')\n\nprint(f\"Original class distribution: {Counter(y_train)}\")\n\n# Apply SMOTE\nsmote = SMOTE(random_state=42)\nX_train_balanced, y_train_balanced = smote.fit_resample(X_train, y_train)\n\nprint(f\"Balanced class distribution: {Counter(y_train_balanced)}\")\nprint(f\"Training samples after SMOTE: {len(y_train_balanced)}\")\n\n# Save balanced data\nnp.save('/kaggle/working/X_train_balanced.npy', X_train_balanced)\nnp.save('/kaggle/working/y_train_balanced.npy', y_train_balanced)\nprint(\"\\n✅ Balanced data saved!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T15:47:29.889652Z","iopub.execute_input":"2026-05-09T15:47:29.890513Z","iopub.status.idle":"2026-05-09T15:47:30.661681Z","shell.execute_reply.started":"2026-05-09T15:47:29.890482Z","shell.execute_reply":"2026-05-09T15:47:30.660877Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# STEP 4: Advanced Feature Engineering & Selection","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# STEP 5: Hyperparameter Optimization with Cross-Validation","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-09T15:50:15.642913Z","iopub.execute_input":"2026-05-09T15:50:15.643689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================================\n# STEP 5: PROBABILITY CALIBRATION & ENSEMBLE\n# ============================================================================\n\nfrom sklearn.calibration import CalibratedClassifierCV\nfrom sklearn.ensemble import VotingClassifier, StackingClassifier\nfrom sklearn.linear_model import LogisticRegression\n\nprint(\"\\n\" + \"=\"*70)\nprint(\"STEP 5: PROBABILITY CALIBRATION & ENSEMBLE\")\nprint(\"=\"*70)\n\n# Load data\nX_train_balanced = np.load('/kaggle/working/X_train_balanced.npy')\ny_train_balanced = np.load('/kaggle/working/y_train_balanced.npy')\nX_val = np.load('/kaggle/working/X_val_processed.npy')\ny_val = np.load('/kaggle/working/y_val.npy')\n\n# Calibrate each model\ncalibrated_models = {}\n\nfor name, model in best_models.items():\n    print(f\"\\nCalibrating {name}...\")\n    calibrated = CalibratedClassifierCV(model, method='sigmoid', cv=5)\n    calibrated.fit(X_train_balanced, y_train_balanced)\n    calibrated_models[name] = calibrated\n    print(f\"✅ {name} calibrated\")\n\n# Create Voting Ensemble\nprint(\"\\nCreating Voting Ensemble...\")\nmodels_list = [(name, model) for name, model in calibrated_models.items()]\nvoting_ensemble = VotingClassifier(estimators=models_list, voting='soft')\nvoting_ensemble.fit(X_train_balanced, y_train_balanced)\n\n# Create Stacking Ensemble\nprint(\"Creating Stacking Ensemble...\")\nmeta_learner = LogisticRegression(C=1.0, class_weight='balanced', max_iter=1000)\nstacking_ensemble = StackingClassifier(\n    estimators=models_list,\n    final_estimator=meta_learner,\n    cv=5,\n    stack_method='predict_proba'\n)\nstacking_ensemble.fit(X_train_balanced, y_train_balanced)\n\nprint(\"✅ Ensembles created!\")\n\n# Save models\nimport joblib\njoblib.dump(best_models, '/kaggle/working/best_models.pkl')\njoblib.dump(calibrated_models, '/kaggle/working/calibrated_models.pkl')\njoblib.dump(voting_ensemble, '/kaggle/working/voting_ensemble.pkl')\njoblib.dump(stacking_ensemble, '/kaggle/working/stacking_ensemble.pkl')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================================\n# STEP 6: MODEL EVALUATION\n# ============================================================================\n\nfrom sklearn.metrics import (roc_auc_score, average_precision_score, brier_score_loss,\n                             log_loss, accuracy_score, precision_score, recall_score,\n                             f1_score, confusion_matrix, roc_curve, \n                             precision_recall_curve, calibration_curve)\nimport matplotlib.pyplot as plt\n\nprint(\"\\n\" + \"=\"*70)\nprint(\"STEP 6: MODEL EVALUATION\")\nprint(\"=\"*70)\n\n# Load test data\nX_test = np.load('/kaggle/working/X_test_processed.npy')\ny_test = np.load('/kaggle/working/y_test.npy')\n\n# Combine all models for evaluation\nall_models = {\n    'Random Forest': best_models['Random Forest'],\n    'XGBoost': best_models['XGBoost'],\n    'Gradient Boosting': best_models['Gradient Boosting'],\n    'SVM': best_models['SVM'],\n    'Voting Ensemble': voting_ensemble,\n    'Stacking Ensemble': stacking_ensemble\n}\n\n# Evaluate each model\ntest_results = {}\n\nfor name, model in all_models.items():\n    print(f\"\\n{'='*60}\")\n    print(f\"Evaluating: {name}\")\n    print(f\"{'='*60}\")\n    \n    y_pred = model.predict(X_test)\n    y_pred_proba = model.predict_proba(X_test)[:, 1]\n    \n    metrics = {\n        'Accuracy': accuracy_score(y_test, y_pred),\n        'Precision': precision_score(y_test, y_pred),\n        'Recall': recall_score(y_test, y_pred),\n        'F1-Score': f1_score(y_test, y_pred),\n        'ROC-AUC': roc_auc_score(y_test, y_pred_proba),\n        'PR-AUC': average_precision_score(y_test, y_pred_proba),\n        'Log-Loss': log_loss(y_test, y_pred_proba),\n        'Brier Score': brier_score_loss(y_test, y_pred_proba)\n    }\n    \n    test_results[name] = metrics\n    \n    for metric_name, value in metrics.items():\n        print(f\"{metric_name:15s}: {value:.4f}\")\n    \n    # Confusion matrix\n    cm = confusion_matrix(y_test, y_pred)\n    tn, fp, fn, tp = cm.ravel()\n    print(f\"\\nConfusion Matrix: TN={tn}, FP={fp}, FN={fn}, TP={tp}\")\n    print(f\"Sensitivity: {tp/(tp+fn):.4f}\")\n    print(f\"Specificity: {tn/(tn+fp):.4f}\")\n\n# Find best model\nbest_name = max(test_results.items(), key=lambda x: x[1]['ROC-AUC'])[0]\nprint(f\"\\n{'='*60}\")\nprint(f\"🏆 BEST MODEL: {best_name}\")\nprint(f\"   ROC-AUC: {test_results[best_name]['ROC-AUC']:.4f}\")\nprint(f\"   Sensitivity: {test_results[best_name]['Recall']:.4f}\")\nprint(f\"{'='*60}\")\n\n# Save results\nresults_df = pd.DataFrame(test_results).T\nresults_df.to_csv('/kaggle/working/model_results.csv')\nprint(\"\\n✅ Results saved to model_results.csv\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================================\n# STEP 7: VISUALIZATIONS\n# ============================================================================\n\nprint(\"\\n\" + \"=\"*70)\nprint(\"STEP 7: GENERATING VISUALIZATIONS\")\nprint(\"=\"*70)\n\n# ROC Curves\nplt.figure(figsize=(10, 8))\nfor name, model in all_models.items():\n    y_pred_proba = model.predict_proba(X_test)[:, 1]\n    fpr, tpr, _ = roc_curve(y_test, y_pred_proba)\n    auc = test_results[name]['ROC-AUC']\n    plt.plot(fpr, tpr, lw=2, label=f'{name} (AUC = {auc:.3f})')\n\nplt.plot([0, 1], [0, 1], 'k--', lw=2, label='Random')\nplt.xlabel('False Positive Rate (1 - Specificity)', fontsize=12)\nplt.ylabel('True Positive Rate (Sensitivity)', fontsize=12)\nplt.title('ROC Curves - Pneumonia Detection', fontsize=14, fontweight='bold')\nplt.legend(loc='lower right')\nplt.grid(alpha=0.3)\nplt.tight_layout()\nplt.savefig('/kaggle/working/roc_curves.png', dpi=150)\nplt.show()\n\n# Performance Comparison Bar Chart\nmetrics_to_plot = ['ROC-AUC', 'F1-Score', 'Recall', 'Precision']\nfig, ax = plt.subplots(figsize=(12, 6))\nx = np.arange(len(metrics_to_plot))\nwidth = 0.15\ncolors = ['#2E86AB', '#A23B72', '#F18F01', '#C73E1D', '#6A994E', '#FF6B35']\n\nfor i, (name, metrics) in enumerate(test_results.items()):\n    values = [metrics[m] for m in metrics_to_plot]\n    offset = (i - len(all_models)/2) * width\n    ax.bar(x + offset, values, width, label=name, color=colors[i % len(colors)])\n\nax.set_ylabel('Score', fontsize=12)\nax.set_title('Model Performance Comparison', fontsize=14, fontweight='bold')\nax.set_xticks(x)\nax.set_xticklabels(metrics_to_plot)\nax.legend(loc='lower right', fontsize=8)\nax.set_ylim([0, 1.05])\nax.grid(alpha=0.3, axis='y')\nplt.tight_layout()\nplt.savefig('/kaggle/working/model_comparison.png', dpi=150)\nplt.show()\n\nprint(\"\\n✅ All visualizations saved!\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# STEP 8: SHAP Explainability (Model Interpretability)","metadata":{}},{"cell_type":"code","source":"# ============================================================================\n# STEP 7: Comprehensive Model Evaluation\n# ============================================================================\n\nfrom sklearn.metrics import (roc_auc_score, average_precision_score, brier_score_loss,\n                             log_loss, accuracy_score, precision_score, recall_score,\n                             f1_score, confusion_matrix, classification_report,\n                             roc_curve, precision_recall_curve, calibration_curve)\nfrom scipy import interp\nfrom itertools import cycle\n\nclass ModelEvaluator:\n    \"\"\"Comprehensive model evaluation with clinical metrics\"\"\"\n    \n    def __init__(self, models, X_test, y_test):\n        self.models = models\n        self.X_test = X_test\n        self.y_test = y_test\n        self.results = {}\n        \n    def evaluate_all(self):\n        \"\"\"Evaluate all models with multiple metrics\"\"\"\n        print(\"\\n\" + \"=\"*70)\n        print(\"COMPREHENSIVE MODEL EVALUATION\")\n        print(\"=\"*70)\n        \n        for name, model in self.models.items():\n            print(f\"\\n{'='*60}\")\n            print(f\"EVALUATING: {name}\")\n            print(f\"{'='*60}\")\n            \n            # Predictions\n            y_pred = model.predict(self.X_test)\n            y_pred_proba = model.predict_proba(self.X_test)[:, 1]\n            \n            # Core metrics\n            metrics = {\n                'Accuracy': accuracy_score(self.y_test, y_pred),\n                'Precision': precision_score(self.y_test, y_pred),\n                'Recall': recall_score(self.y_test, y_pred),\n                'F1-Score': f1_score(self.y_test, y_pred),\n                'ROC-AUC': roc_auc_score(self.y_test, y_pred_proba),\n                'PR-AUC': average_precision_score(self.y_test, y_pred_proba),\n                'Log-Loss': log_loss(self.y_test, y_pred_proba),\n                'Brier Score': brier_score_loss(self.y_test, y_pred_proba)\n            }\n            \n            self.results[name] = metrics\n            \n            # Print metrics\n            for metric_name, value in metrics.items():\n                print(f\"{metric_name:15s}: {value:.4f}\")\n            \n            # Confusion Matrix\n            cm = confusion_matrix(self.y_test, y_pred)\n            print(f\"\\nConfusion Matrix:\")\n            print(f\"TN: {cm[0,0]:5d}  FP: {cm[0,1]:5d}\")\n            print(f\"FN: {cm[1,0]:5d}  TP: {cm[1,1]:5d}\")\n            \n            # Additional clinical metrics\n            tn, fp, fn, tp = cm.ravel()\n            specificity = tn / (tn + fp) if (tn + fp) > 0 else 0\n            npv = tn / (tn + fn) if (tn + fn) > 0 else 0  # Negative Predictive Value\n            print(f\"\\nClinical Metrics:\")\n            print(f\"Sensitivity (Recall): {metrics['Recall']:.4f}\")\n            print(f\"Specificity: {specificity:.4f}\")\n            print(f\"NPV: {npv:.4f}\")\n            print(f\"PPV (Precision): {metrics['Precision']:.4f}\")\n        \n        return self.results\n    \n    def plot_roc_curves(self):\n        \"\"\"Plot ROC curves for all models\"\"\"\n        plt.figure(figsize=(10, 8))\n        \n        colors = cycle(['blue', 'red', 'green', 'orange', 'purple', 'brown'])\n        \n        for (name, model), color in zip(self.models.items(), colors):\n            y_pred_proba = model.predict_proba(self.X_test)[:, 1]\n            fpr, tpr, _ = roc_curve(self.y_test, y_pred_proba)\n            auc = roc_auc_score(self.y_test, y_pred_proba)\n            plt.plot(fpr, tpr, color=color, lw=2, \n                    label=f'{name} (AUC = {auc:.3f})')\n        \n        plt.plot([0, 1], [0, 1], 'k--', lw=2, label='Random Classifier')\n        plt.xlim([0.0, 1.0])\n        plt.ylim([0.0, 1.05])\n        plt.xlabel('False Positive Rate (1 - Specificity)', fontsize=12)\n        plt.ylabel('True Positive Rate (Sensitivity)', fontsize=12)\n        plt.title('ROC Curves - Pneumonia Detection', fontsize=14, fontweight='bold')\n        plt.legend(loc=\"lower right\", fontsize=10)\n        plt.grid(alpha=0.3)\n        plt.tight_layout()\n        plt.show()\n        \n    def plot_precision_recall_curves(self):\n        \"\"\"Plot Precision-Recall curves\"\"\"\n        plt.figure(figsize=(10, 8))\n        \n        colors = cycle(['blue', 'red', 'green', 'orange', 'purple', 'brown'])\n        \n        for (name, model), color in zip(self.models.items(), colors):\n            y_pred_proba = model.predict_proba(self.X_test)[:, 1]\n            precision, recall, _ = precision_recall_curve(self.y_test, y_pred_proba)\n            ap = average_precision_score(self.y_test, y_pred_proba)\n            plt.plot(recall, precision, color=color, lw=2,\n                    label=f'{name} (AP = {ap:.3f})')\n        \n        plt.xlim([0.0, 1.0])\n        plt.ylim([0.0, 1.05])\n        plt.xlabel('Recall (Sensitivity)', fontsize=12)\n        plt.ylabel('Precision (PPV)', fontsize=12)\n        plt.title('Precision-Recall Curves', fontsize=14, fontweight='bold')\n        plt.legend(loc=\"lower left\", fontsize=10)\n        plt.grid(alpha=0.3)\n        plt.tight_layout()\n        plt.show()\n    \n    def plot_calibration_curves(self):\n        \"\"\"Plot calibration curves (critical for medical diagnosis)\"\"\"\n        plt.figure(figsize=(10, 8))\n        \n        colors = cycle(['blue', 'red', 'green', 'orange', 'purple', 'brown'])\n        \n        for (name, model), color in zip(self.models.items(), colors):\n            y_pred_proba = model.predict_proba(self.X_test)[:, 1]\n            prob_true, prob_pred = calibration_curve(self.y_test, y_pred_proba, n_bins=10)\n            brier = brier_score_loss(self.y_test, y_pred_proba)\n            plt.plot(prob_pred, prob_true, marker='o', linewidth=2, color=color,\n                    label=f'{name} (Brier = {brier:.3f})')\n        \n        plt.plot([0, 1], [0, 1], 'k--', lw=2, label='Perfectly Calibrated')\n        plt.xlim([0.0, 1.0])\n        plt.ylim([0.0, 1.05])\n        plt.xlabel('Mean Predicted Probability', fontsize=12)\n        plt.ylabel('Fraction of Positives', fontsize=12)\n        plt.title('Calibration Curves - Critical for Clinical Decision Making', \n                 fontsize=14, fontweight='bold')\n        plt.legend(loc=\"lower right\", fontsize=10)\n        plt.grid(alpha=0.3)\n        plt.tight_layout()\n        plt.show()\n    \n    def plot_metrics_comparison(self):\n        \"\"\"Bar plot comparing key metrics across models\"\"\"\n        metrics_to_plot = ['ROC-AUC', 'PR-AUC', 'F1-Score']\n        \n        fig, ax = plt.subplots(figsize=(12, 6))\n        x = np.arange(len(metrics_to_plot))\n        width = 0.2\n        colors = ['#2E86AB', '#A23B72', '#F18F01', '#C73E1D', '#6A994E']\n        \n        for i, (name, metrics) in enumerate(self.results.items()):\n            values = [metrics[m] for m in metrics_to_plot]\n            offset = (i - len(self.models)/2) * width\n            bars = ax.bar(x + offset, values, width, label=name, color=colors[i % len(colors)])\n        \n        ax.set_ylabel('Score', fontsize=12)\n        ax.set_title('Model Performance Comparison', fontsize=14, fontweight='bold')\n        ax.set_xticks(x)\n        ax.set_xticklabels(metrics_to_plot)\n        ax.legend(loc='lower right')\n        ax.set_ylim([0, 1.05])\n        ax.grid(alpha=0.3, axis='y')\n        \n        plt.tight_layout()\n        plt.show()\n    \n    def generate_clinical_report(self):\n        \"\"\"Generate a clinical-style evaluation report\"\"\"\n        print(\"\\n\" + \"=\"*70)\n        print(\"CLINICAL DECISION SUPPORT REPORT\")\n        print(\"=\"*70)\n        \n        # Find best model based on ROC-AUC\n        best_model = max(self.results.items(), key=lambda x: x[1]['ROC-AUC'])\n        \n        print(f\"\\n🏆 RECOMMENDED MODEL: {best_model[0]}\")\n        print(f\"   ROC-AUC: {best_model[1]['ROC-AUC']:.4f}\")\n        print(f\"   Sensitivity: {best_model[1]['Recall']:.4f}\")\n        \n        # Calculate specificity for best model\n        best_model_obj = self.models[best_model[0]]\n        y_pred = best_model_obj.predict(self.X_test)\n        tn, fp, fn, tp = confusion_matrix(self.y_test, y_pred).ravel()\n        specificity = tn / (tn + fp) if (tn + fp) > 0 else 0\n        print(f\"   Specificity: {specificity:.4f}\")\n        \n        print(\"\\n📊 CLINICAL INTERPRETATION:\")\n        print(\"-\" * 40)\n        \n        if best_model[1]['ROC-AUC'] > 0.9:\n            print(\"✓ EXCELLENT discriminatory ability\")\n        elif best_model[1]['ROC-AUC'] > 0.8:\n            print(\"✓ GOOD discriminatory ability\")\n        else:\n            print(\"⚠️ MODERATE discriminatory ability - consider additional features\")\n        \n        if best_model[1]['Recall'] > 0.8:\n            print(\"✓ HIGH sensitivity - good at identifying positive cases\")\n        else:\n            print(\"⚠️ MODERATE sensitivity - risk of false negatives\")\n        \n        if best_model[1]['Precision'] > 0.8:\n            print(\"✓ HIGH precision - low false positive rate\")\n        else:\n            print(\"⚠️ MODERATE precision - some false positives expected\")\n        \n        print(\"\\n💡 RECOMMENDATIONS:\")\n        print(\"-\" * 40)\n        print(\"1. Use calibrated probabilities for clinical decision thresholds\")\n        print(\"2. Consider operating point based on prevalence and cost of errors\")\n        print(\"3. Validate on external dataset before clinical deployment\")\n\n# Evaluate all models\nall_models = {\n    'Random Forest': optimizer.best_models['Random Forest'],\n    'XGBoost': optimizer.best_models['XGBoost'],\n    'Gradient Boosting': optimizer.best_models['Gradient Boosting'],\n    'SVM': optimizer.best_models['SVM'],\n    'Voting Ensemble': voting_ensemble,\n    'Stacking Ensemble': stacking_ensemble\n}}\n\nevaluator = ModelEvaluator(all_models, X_test_processed, y_test)\nresults = evaluator.evaluate_all()\n\n# Generate all visualizations\nevaluator.plot_roc_curves()\nevaluator.plot_precision_recall_curves()\nevaluator.plot_calibration_curves()\nevaluator.plot_metrics_comparison()\nevaluator.generate_clinical_report()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# STEP 7: Comprehensive Model Evaluation","metadata":{}},{"cell_type":"code","source":"# ============================================================================\n# STEP 8: SHAP Explainability (Model Interpretability)\n# ============================================================================\n\nimport shap\n\nclass ModelExplainer:\n    \"\"\"SHAP-based model explainability for clinical interpretability\"\"\"\n    \n    def __init__(self, model, X_train, feature_names):\n        self.model = model\n        self.X_train = X_train\n        self.feature_names = feature_names\n        self.explainer = None\n        self.shap_values = None\n        \n    def explain_random_forest(self):\n        \"\"\"Explain Random Forest predictions with TreeExplainer\"\"\"\n        print(\"\\n\" + \"=\"*60)\n        print(\"SHAP ANALYSIS - MODEL INTERPRETABILITY\")\n        print(\"=\"*60)\n        \n        # Create explainer\n        self.explainer = shap.TreeExplainer(self.model)\n        \n        # Calculate SHAP values (use subset for speed)\n        sample_size = min(500, len(self.X_train))\n        X_sample = self.X_train[:sample_size]\n        self.shap_values = self.explainer.shap_values(X_sample)\n        \n        return self.shap_values\n    \n    def plot_global_importance(self, max_display=20):\n        \"\"\"Global feature importance plot\"\"\"\n        plt.figure(figsize=(12, 8))\n        \n        # For binary classification, shap_values[1] is for positive class\n        shap.summary_plot(self.shap_values[1], self.X_train[:500], \n                         feature_names=self.feature_names,\n                         max_display=max_display, show=False)\n        \n        plt.title('Global Feature Importance - SHAP Values', fontsize=14, fontweight='bold')\n        plt.tight_layout()\n        plt.show()\n    \n    def plot_summary(self):\n        \"\"\"Summary plot showing feature impact\"\"\"\n        plt.figure(figsize=(10, 8))\n        shap.summary_plot(self.shap_values[1], self.X_train[:500], \n                         feature_names=self.feature_names, show=False)\n        plt.title('SHAP Summary Plot - Feature Impact on Predictions', \n                 fontsize=14, fontweight='bold')\n        plt.tight_layout()\n        plt.show()\n    \n    def explain_single_prediction(self, X_sample, sample_idx=0):\n        \"\"\"Explain a single prediction (waterfall plot)\"\"\"\n        # Get prediction for single sample\n        sample = X_sample[sample_idx:sample_idx+1]\n        prediction = self.model.predict_proba(sample)[0, 1]\n        \n        print(f\"\\n🔍 EXPLANATION FOR SINGLE PREDICTION\")\n        print(f\"   Predicted Probability: {prediction:.3f}\")\n        print(f\"   Risk Category: {'HIGH' if prediction > 0.5 else 'LOW'}\")\n        \n        # Waterfall plot\n        shap.waterfall_plot(\n            shap.Explanation(values=self.shap_values[1][sample_idx],\n                           base_values=self.explainer.expected_value[1],\n                           data=sample.flatten(),\n                           feature_names=self.feature_names),\n            max_display=10,\n            show=False\n        )\n        plt.title(f'Single Prediction Explanation (Probability = {prediction:.3f})', \n                 fontsize=12, fontweight='bold')\n        plt.tight_layout()\n        plt.show()\n\n# Explain best model (Random Forest)\nexplainer = ModelExplainer(\n    optimizer.best_models['Random Forest'], \n    X_train_processed, \n    feature_names_list\n)\n\n# Generate explanations\nshap_values = explainer.explain_random_forest()\nexplainer.plot_global_importance(max_display=20)\nexplainer.plot_summary()\n\n# Explain a correct prediction\ncorrect_idx = np.where(y_test == 0)[0][0]  # Normal case\nexplainer.explain_single_prediction(X_test_processed, correct_idx)\n\n    ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# STEP 9: Decision Threshold Optimization (FOR CLINICAL USE)","metadata":{}},{"cell_type":"code","source":"# ============================================================================\n# STEP 9: Decision Threshold Optimization (FOR CLINICAL USE)\n# ============================================================================\n\nfrom sklearn.metrics import fbeta_score\n\nclass ThresholdOptimizer:\n    \"\"\"Optimize decision threshold for clinical utility\"\"\"\n    \n    def __init__(self, model, X_val, y_val):\n        self.model = model\n        self.X_val = X_val\n        self.y_val = y_val\n        self.probabilities = model.predict_proba(X_val)[:, 1]\n        \n    def find_optimal_threshold(self):\n        \"\"\"Find threshold that maximizes F2-score (prioritizing recall)\"\"\"\n        thresholds = np.arange(0.1, 0.9, 0.01)\n        f2_scores = []\n        \n        for threshold in thresholds:\n            y_pred = (self.probabilities >= threshold).astype(int)\n            f2_scores.append(fbeta_score(self.y_val, y_pred, beta=2))\n        \n        optimal_idx = np.argmax(f2_scores)\n        optimal_threshold = thresholds[optimal_idx]\n        \n        # Plot optimization curve\n        plt.figure(figsize=(10, 6))\n        plt.plot(thresholds, f2_scores, 'b-', linewidth=2, label='F2-Score')\n        plt.axvline(optimal_threshold, color='green', linestyle='--', \n                   label=f'Optimal Threshold = {optimal_threshold:.3f}')\n        plt.xlabel('Decision Threshold', fontsize=12)\n        plt.ylabel('F2-Score', fontsize=12)\n        plt.title('Threshold Optimization (F2-Score prioritizes Recall)', \n                 fontsize=14, fontweight='bold')\n        plt.legend()\n        plt.grid(alpha=0.3)\n        plt.tight_layout()\n        plt.show()\n        \n        print(f\"\\n✅ Optimal Threshold: {optimal_threshold:.3f}\")\n        print(f\"   F2-Score at threshold: {f2_scores[optimal_idx]:.4f}\")\n        \n        return optimal_threshold\n\n# Optimize threshold for voting ensemble\nthreshold_optimizer = ThresholdOptimizer(voting_ensemble, X_val_processed, y_val)\noptimal_threshold = threshold_optimizer.find_optimal_threshold()\n\n\n\nprint(\"✅ Optimal decision threshold found\")\nprint(\"\\n📁 All files saved to /kaggle/working/\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# STEP 10: Final Model Saving","metadata":{}},{"cell_type":"code","source":"# ============================================================================\n# STEP 10: Final Model Saving\n# ============================================================================\n\nimport joblib\nfrom datetime import datetime\n\nprint(\"\\n\" + \"=\"*60)\nprint(\"SAVING MODELS\")\nprint(\"=\"*60)\n\n# Save best models\njoblib.dump(voting_ensemble, '/kaggle/working/voting_ensemble_model.pkl')\njoblib.dump(stacking_ensemble, '/kaggle/working/stacking_ensemble_model.pkl')\njoblib.dump(optimizer.best_models['Random Forest'], '/kaggle/working/random_forest_model.pkl')\njoblib.dump(optimizer.best_models['XGBoost'], '/kaggle/working/xgboost_model.pkl')\n\n# Save threshold and scalers\nnp.save('/kaggle/working/optimal_threshold.npy', optimal_threshold)\njoblib.dump(engineer.scaler, '/kaggle/working/feature_scaler.pkl')\njoblib.dump(engineer.selector_rf, '/kaggle/working/feature_selector.pkl')\n\nprint(\"✅ All models and preprocessing objects saved to /kaggle/working/\")\n\nprint(\"\\n\" + \"=\"*70)\nprint(\"🏆 PROJECT COMPLETED SUCCESSFULLY 🏆\")\nprint(\"=\"*70)\nprint(\"\\n✅ Models trained, optimized, and saved\")\nprint(\"✅ Comprehensive evaluation completed\")\nprint(\"✅ SHAP explanations generated\")\nprint(\"✅ Optimal decision threshold found\")\nprint(\"\\n📁 All files saved to /kaggle/working/\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 3: Preprocessing","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport pydicom\nimport cv2\nimport os\nfrom tqdm import tqdm\n\nDATA_DIR  = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\nTRAIN_DIR = f'{DATA_DIR}/stage_2_train_images'\nOUTPUT_DIR = '/kaggle/working'\n\ndf       = pd.read_csv(f'{DATA_DIR}/stage_2_train_labels.csv')\npatients = df.groupby('patientId')['Target'].max().reset_index()\n\ntrain_df = pd.read_csv(f'{OUTPUT_DIR}/train_split.csv')\nval_df   = pd.read_csv(f'{OUTPUT_DIR}/val_split.csv')\ntest_df  = pd.read_csv(f'{OUTPUT_DIR}/test_split.csv')\n\nprint('✅ Imports done.')\nprint(f'Train: {len(train_df):,} | Val: {len(val_df):,} | Test: {len(test_df):,}')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Functions","metadata":{}},{"cell_type":"code","source":"def load_dicom(pid):\n    \"\"\"Load raw DICOM pixel array.\"\"\"\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = dcm.pixel_array.astype(np.float32)\n    return img\n\ndef apply_window(img, level=-500, width=1500):\n    \"\"\"Apply lung windowing to enhance pulmonary structures.\"\"\"\n    low  = level - width // 2\n    high = level + width // 2\n    img  = np.clip(img, low, high)\n    img  = (img - low) / (high - low)\n    return img\n\ndef normalize(img):\n    \"\"\"Normalize image to [0, 1].\"\"\"\n    img = img - img.min()\n    if img.max() > 0:\n        img = img / img.max()\n    return img\n    \ndef enhance_contrast(img):\n\n    img_uint8 = (img * 255).astype(np.uint8)\n\n    clahe = cv2.createCLAHE(\n        clipLimit=2.0,\n        tileGridSize=(8,8)\n    )\n\n    enhanced = clahe.apply(img_uint8)\n\n    return enhanced / 255.0\n    \ndef denoise(img):\n    \"\"\"Apply Gaussian blur for noise reduction.\"\"\"\n    return cv2.GaussianBlur(img, (3, 3), 0)\n    \ndef resize_img(img, size=512):\n    \"\"\"Resize image to fixed size.\"\"\"\n    return cv2.resize(img, (size, size), interpolation=cv2.INTER_AREA)\n\ndef preprocess(pid, size=512):\n    \"\"\"Full preprocessing pipeline for one patient.\"\"\"\n    img = load_dicom(pid)       # Step 1: Load DICOM\n    img = normalize(img)        # Step 2: Normalize to [0,1] first\n    img = enhance_contrast(img)          # Step 3: enhance_contrast\n    img = resize_img(img, size) # Step 4: Resize to 512x512\n    return img","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Test Preprocessing on Single Image","metadata":{}},{"cell_type":"code","source":"pid = patients['patientId'].iloc[0]\n\nraw     = load_dicom(pid)\nnormed  = normalize(raw)\nenhanced = enhance_contrast(normed)\nresized  = resize_img(enhanced)\n\nprint(f'Raw      — shape: {raw.shape},     min: {raw.min():.1f},   max: {raw.max():.1f}')\nprint(f'Normed   — shape: {normed.shape},   min: {normed.min():.2f},  max: {normed.max():.2f}')\nprint(f'Enhanced — shape: {enhanced.shape}, min: {enhanced.min():.2f},  max: {enhanced.max():.2f}')\nprint(f'Resized  — shape: {resized.shape},  min: {resized.min():.2f},  max: {resized.max():.2f}')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Augmentation","metadata":{}},{"cell_type":"code","source":"import albumentations as A\n\naugment = A.Compose([\n    A.HorizontalFlip(p=0.5),\n    A.Rotate(limit=10, p=0.3),\n    A.Affine(translate_percent=0.1, p=0.3),\n    A.RandomBrightnessContrast(brightness_limit=0.2,\n                               contrast_limit=0.2, p=0.4),\n])\n\ndef augment_img(img):\n    \"\"\"Apply augmentation to a single image.\"\"\"\n    img_uint8 = (img * 255).astype(np.uint8)\n    augmented = augment(image=img_uint8)['image']\n    return augmented.astype(np.float32) / 255.0\n\nprint('✅ Augmentation pipeline defined.')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualize Augmentation Effects","metadata":{}},{"cell_type":"code","source":"sample_pid = patients[patients['Target']==1]['patientId'].sample(1, random_state=7).values[0]\nimg = preprocess(sample_pid)\n\nfig, axes = plt.subplots(2, 4, figsize=(16, 8))\naxes = axes.flatten()\n\n# Original\naxes[0].imshow(img, cmap='gray')\naxes[0].set_title('Original (preprocessed)')\naxes[0].axis('off')\n\n# 7 augmented versions\nfor i in range(1, 8):\n    aug = augment_img(img)\n    axes[i].imshow(aug, cmap='gray')\n    axes[i].set_title(f'Augmented #{i}')\n    axes[i].axis('off')\n\nplt.suptitle('Data Augmentation Examples', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Before VS After Preprocessing","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 4, figsize=(16, 12))\n\nsample_pids = patients.sample(4, random_state=10)['patientId'].values\n\nfor idx, pid in enumerate(sample_pids):\n    raw      = load_dicom(pid)\n    windowed = apply_window(raw)\n    resized  = resize_img(normalize(windowed))\n\n    # Row 1: Raw\n    axes[0][idx].imshow(raw, cmap='gray')\n    axes[0][idx].set_title(f'Raw DICOM\\n{raw.shape}')\n    axes[0][idx].axis('off')\n\n    # Row 2: After windowing\n    axes[1][idx].imshow(windowed, cmap='gray')\n    axes[1][idx].set_title(f'After Windowing\\n[{windowed.min():.2f}, {windowed.max():.2f}]')\n    axes[1][idx].axis('off')\n\n    # Row 3: After resize\n    axes[2][idx].imshow(resized, cmap='gray')\n    axes[2][idx].set_title(f'Final (512x512)\\n[{resized.min():.2f}, {resized.max():.2f}]')\n    axes[2][idx].axis('off')\n\nplt.suptitle('Preprocessing Pipeline: Raw → Windowed → Final', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Histogram Update","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 4, figsize=(16, 12))\n\nsample_pids = patients.sample(4, random_state=10)['patientId'].values\n\nfor idx, pid in enumerate(sample_pids):\n    raw      = load_dicom(pid)\n    normed   = normalize(raw)\n    final    = resize_img(enhance_contrast(normed))\n\n    # Row 1: Raw histogram\n    axes[0][idx].hist(raw.flatten(), bins=50, color='gray', edgecolor='none')\n    axes[0][idx].set_title(f'Raw\\nmin={raw.min():.0f} max={raw.max():.0f}')\n    axes[0][idx].set_xlabel('Pixel Value')\n    axes[0][idx].set_ylabel('Count')\n\n    # Row 2: After normalization\n    axes[1][idx].hist(normed.flatten(), bins=50, color='steelblue', edgecolor='none')\n    axes[1][idx].set_title(f'After Normalize\\nmin={normed.min():.2f} max={normed.max():.2f}')\n    axes[1][idx].set_xlabel('Pixel Value')\n    axes[1][idx].set_ylabel('Count')\n\n    # Row 3: Final (enhanced + resized)\n    axes[2][idx].hist(final.flatten(), bins=50, color='tomato', edgecolor='none')\n    axes[2][idx].set_title(f'Final (enhanced+resized)\\nmin={final.min():.2f} max={final.max():.2f}')\n    axes[2][idx].set_xlabel('Pixel Value')\n    axes[2][idx].set_ylabel('Count')\n\nplt.suptitle('Preprocessing Pipeline: Raw → Normalized → Final', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 4: Lung Segmentation","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport pydicom\nimport cv2\nfrom scipy import ndimage\nfrom tqdm import tqdm\nimport os\n\nDATA_DIR  = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\nTRAIN_DIR = f'{DATA_DIR}/stage_2_train_images'\nOUTPUT_DIR = '/kaggle/working'\n\ndf       = pd.read_csv(f'{DATA_DIR}/stage_2_train_labels.csv')\npatients = df.groupby('patientId')['Target'].max().reset_index()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Segmentation Function","metadata":{}},{"cell_type":"code","source":"def segment_lungs(img):\n    \"\"\"\n    Segment lung region from chest X-ray using\n    thresholding + morphological operations.\n    Returns: binary lung mask\n    \"\"\"\n    # Step 1: Convert to uint8\n    img_uint8 = (img * 255).astype(np.uint8)\n\n    # Step 2: Otsu thresholding — separates lung from background\n    _, thresh = cv2.threshold(img_uint8, 0, 255,\n                              cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n\n    # Step 3: Invert — lungs are dark in X-ray\n    thresh = cv2.bitwise_not(thresh)\n\n    # Step 4: Morphological opening — remove small noise\n    kernel  = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (5, 5))\n    opened  = cv2.morphologyEx(thresh, cv2.MORPH_OPEN, kernel)\n\n    # Step 5: Morphological closing — fill holes inside lungs\n    kernel2 = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (20, 20))\n    closed  = cv2.morphologyEx(opened, cv2.MORPH_CLOSE, kernel2)\n\n    # Step 6: Keep only the 2 largest components (left + right lung)\n    num_labels, labels, stats, _ = cv2.connectedComponentsWithStats(closed)\n    \n    # Sort by area (skip background = label 0)\n    areas = stats[1:, cv2.CC_STAT_AREA]\n    top2  = np.argsort(areas)[::-1][:2] + 1  # top 2 largest\n\n    mask = np.zeros_like(closed)\n    for label in top2:\n        mask[labels == label] = 255\n\n    return mask\n\ndef apply_mask(img, mask):\n    \"\"\"Apply lung mask to image — zero out background.\"\"\"\n    mask_norm = mask / 255.0\n    return img * mask_norm\n\nprint('✅ Segmentation functions defined.')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Test Segmentation on Sample Image","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 4, figsize=(16, 12))\n\nsample_pids = patients.sample(4, random_state=42)['patientId'].values\n\nfor idx, pid in enumerate(sample_pids):\n    # Preprocess\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = dcm.pixel_array.astype(np.float32)\n    img = img - img.min()\n    if img.max() > 0:\n        img = img / img.max()\n    img = enhance_contrast(img)\n    img = cv2.resize(img, (512, 512), interpolation=cv2.INTER_AREA)\n\n    # Segment\n    mask        = segment_lungs(img)\n    masked_img  = apply_mask(img, mask)\n\n    # Row 1: Original\n    axes[0][idx].imshow(img, cmap='gray')\n    axes[0][idx].set_title('Preprocessed')\n    axes[0][idx].axis('off')\n\n    # Row 2: Mask\n    axes[1][idx].imshow(mask, cmap='gray')\n    axes[1][idx].set_title('Lung Mask')\n    axes[1][idx].axis('off')\n\n    # Row 3: Masked image\n    axes[2][idx].imshow(masked_img, cmap='gray')\n    axes[2][idx].set_title('Segmented Lung')\n    axes[2][idx].axis('off')\n\nplt.suptitle('Lung Segmentation: Thresholding + Morphology', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 4, figsize=(16, 12))\n\npneumonia_pids = patients[patients['Target']==1]['patientId'].sample(4, random_state=42).values\n\nfor idx, pid in enumerate(pneumonia_pids):\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = dcm.pixel_array.astype(np.float32)\n    img = img - img.min()\n    if img.max() > 0:\n        img = img / img.max()\n    img = enhance_contrast(img)\n    img = cv2.resize(img, (512, 512), interpolation=cv2.INTER_AREA)\n\n    mask       = segment_lungs(img)\n    masked_img = apply_mask(img, mask)\n\n    axes[0][idx].imshow(img, cmap='gray')\n    axes[0][idx].set_title('Pneumonia — Original')\n    axes[0][idx].axis('off')\n\n    axes[1][idx].imshow(mask, cmap='gray')\n    axes[1][idx].set_title('Lung Mask')\n    axes[1][idx].axis('off')\n\n    axes[2][idx].imshow(masked_img, cmap='gray')\n    axes[2][idx].set_title('Segmented Lung')\n    axes[2][idx].axis('off')\n\nplt.suptitle('Lung Segmentation on Pneumonia Cases', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Test Segmentation on Pneumonia Cases","metadata":{}},{"cell_type":"markdown","source":"# Step 5: Feature Extraction","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"from scipy.stats import skew, kurtosis\nfrom skimage.feature import graycomatrix, graycoprops\nfrom skimage.measure import regionprops, label\nimport pandas as pd\nfrom tqdm import tqdm\nimport pydicom\nimport cv2\nimport os","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Extraction Logic","metadata":{}},{"cell_type":"code","source":"def extract_features(img, mask):\n    features = {}\n    bool_mask = mask > 0\n    lung_pixels = img[bool_mask]\n\n    # 1. Intensity (global)\n    if len(lung_pixels) > 0:\n        features['mean']     = float(np.mean(lung_pixels))\n        features['std']      = float(np.std(lung_pixels))\n        features['skewness'] = float(skew(lung_pixels))\n        features['kurtosis'] = float(kurtosis(lung_pixels))\n    else:\n        features['mean'] = features['std'] = features['skewness'] = features['kurtosis'] = 0.0\n\n    # Histogram bins\n    hist, _ = np.histogram(lung_pixels if len(lung_pixels) > 0 else [0], bins=10, range=(0, 1))\n    hist = hist / (hist.sum() + 1e-6)\n    for i, val in enumerate(hist):\n        features[f'hist_bin_{i}'] = float(val)\n\n    # 2. GLCM Texture (multiple angles + distances)\n    img_uint8 = (img * 255).astype(np.uint8)\n    labeled_mask = label(bool_mask)\n    regions = regionprops(labeled_mask)\n\n    if regions:\n        min_row = min([r.bbox[0] for r in regions])\n        min_col = min([r.bbox[1] for r in regions])\n        max_row = max([r.bbox[2] for r in regions])\n        max_col = max([r.bbox[3] for r in regions])\n        \n        cropped_mask = bool_mask[min_row:max_row, min_col:max_col]\n        cropped_img = img_uint8[min_row:max_row, min_col:max_col]\n        cropped_img = cropped_img * cropped_mask\n\n        # Quantize to 64 levels for speed + stability\n        img_quantized = (cropped_img // 4).astype(np.uint8)\n        glcm = graycomatrix(img_quantized,\n                            distances=[1, 2, 3],\n                            angles=[0, np.pi/4, np.pi/2, 3*np.pi/4],\n                            levels=64, symmetric=True, normed=True)\n\n        for prop in ['contrast', 'energy', 'homogeneity', 'correlation', 'dissimilarity', 'ASM']:\n            features[prop] = float(np.mean(graycoprops(glcm, prop)))\n    else:\n        for prop in ['contrast', 'energy', 'homogeneity', 'correlation', 'dissimilarity', 'ASM']:\n            features[prop] = 0.0\n\n    # 3. Shape (global)\n    if regions:\n        total_area = sum([r.area for r in regions])\n        total_perimeter = sum([r.perimeter for r in regions])\n        features['area']        = float(total_area)\n        features['perimeter']   = float(total_perimeter)\n        features['circularity'] = (4 * np.pi * total_area) / (total_perimeter ** 2 + 1e-6)\n    else:\n        features['area'] = features['perimeter'] = features['circularity'] = 0.0\n\n    # 4. Per-lobe features (asymmetry is a key pneumonia indicator)\n    if regions:\n        regions_sorted = sorted(regions, key=lambda r: r.centroid[1])\n        if len(regions_sorted) >= 2:\n            left_lung, right_lung = regions_sorted[0], regions_sorted[1]\n        else:\n            left_lung = right_lung = regions_sorted[0]\n\n        for side, region in zip(['left', 'right'], [left_lung, right_lung]):\n            lobe_mask   = labeled_mask == region.label\n            lobe_pixels = img[lobe_mask]\n            features[f'{side}_mean'] = float(np.mean(lobe_pixels)) if len(lobe_pixels) > 0 else 0.0\n            features[f'{side}_std']  = float(np.std(lobe_pixels))  if len(lobe_pixels) > 0 else 0.0\n            features[f'{side}_area'] = float(region.area)\n            features[f'{side}_circularity'] = (4 * np.pi * region.area) / (region.perimeter ** 2 + 1e-6)\n\n        features['area_ratio'] = float(regions_sorted[0].area / (regions_sorted[-1].area + 1e-6))\n    else:\n        for side in ['left', 'right']:\n            features[f'{side}_mean'] = features[f'{side}_std'] = 0.0\n            features[f'{side}_area'] = features[f'{side}_circularity'] = 0.0\n        features['area_ratio'] = 1.0\n\n    return features","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Dataset Builder Pipeline","metadata":{}},{"cell_type":"code","source":"def build_feature_dataset(df_split, max_samples=None):\n    \"\"\"\n    Applies preprocessing, Chan-Vese segmentation, and feature extraction \n    to a dataset split and returns a tabular DataFrame.\n    \"\"\"\n    data = []\n    \n    if max_samples:\n        df_split = df_split.head(max_samples)\n        \n    for _, row in tqdm(df_split.iterrows(), total=len(df_split), desc=\"Extracting Features\"):\n        pid = row['patientId']\n        target = row['Target']\n        \n        try:\n            # 1. Preprocess\n            dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n            img = dcm.pixel_array.astype(np.float32)\n            img = img - img.min()\n            if img.max() > 0:\n                img = img / img.max()\n            img = enhance_contrast(img)\n            img = cv2.resize(img, (512, 512), interpolation=cv2.INTER_AREA)\n            \n            # 2. Segment (Chan-Vese)\n            mask = segment_lungs(img)\n            \n            # 3. Extract\n            feats = extract_features(img, mask)\n            \n            # Add identifiers\n            feats['patientId'] = pid\n            feats['Target'] = target\n            \n            data.append(feats)\n        except Exception as e:\n            # Skip corrupted images quietly\n            continue\n            \n    return pd.DataFrame(data)\n\nprint('✅ Pipeline function defined.')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Run Extraction & Save","metadata":{}},{"cell_type":"code","source":"print(\"\\nExtracting Train Features...\")\ntrain_features = build_feature_dataset(train_df)\ntrain_features.to_csv(f'{OUTPUT_DIR}/train_features.csv', index=False)\n\nprint(\"\\nExtracting Val Features...\")\nval_features = build_feature_dataset(val_df)\nval_features.to_csv(f'{OUTPUT_DIR}/val_features.csv', index=False)\n\nprint(\"\\nExtracting Test Features...\")\ntest_features = build_feature_dataset(test_df)\ntest_features.to_csv(f'{OUTPUT_DIR}/test_features.csv', index=False)\n\nprint(\"\\n✅ All features extracted and saved to CSV!\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings('ignore')\nfrom sklearn.feature_selection import SelectKBest, mutual_info_classif\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.feature_selection import SelectKBest, mutual_info_classif\nfrom sklearn.model_selection import StratifiedKFold, cross_val_score\n \nfrom sklearn.svm import SVC\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.ensemble import RandomForestClassifier\nfrom xgboost import XGBClassifier\n \nfrom sklearn.metrics import (accuracy_score, precision_score, recall_score,\n                             f1_score, roc_auc_score, classification_report,\n                             confusion_matrix, roc_curve)\n \nfrom imblearn.over_sampling import SMOTE","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load Pre-split Data ","metadata":{}},{"cell_type":"code","source":"BASE = \"/kaggle/working\"\ntrain_df = pd.read_csv(f\"{BASE}/train_features.csv\")\nval_df   = pd.read_csv(f\"{BASE}/val_features.csv\")\ntest_df  = pd.read_csv(f\"{BASE}/test_features.csv\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Train: {train_df.shape} | Val: {val_df.shape} | Test: {test_df.shape}\")\nprint(f\"Train class dist: {train_df.Target.value_counts().to_dict()}\")\nprint(f\"Val   class dist: {val_df.Target.value_counts().to_dict()}\")\nprint(f\"Test  class dist: {test_df.Target.value_counts().to_dict()}\")\n \nDROP_COLS    = ['patientId']\nTARGET_COL   = 'Target'\nFEATURE_COLS = [c for c in train_df.columns if c not in DROP_COLS + [TARGET_COL]]\n \nX_train, y_train = train_df[FEATURE_COLS].values, train_df[TARGET_COL].values\nX_val,   y_val   = val_df[FEATURE_COLS].values,   val_df[TARGET_COL].values\nX_test,  y_test  = test_df[FEATURE_COLS].values,  test_df[TARGET_COL].values\n \nprint(f\"\\nFeatures ({len(FEATURE_COLS)}): {FEATURE_COLS}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Step 6: Feature Selection**","metadata":{}},{"cell_type":"code","source":"k = min(20, len(FEATURE_COLS))\nselector = SelectKBest(score_func=mutual_info_classif, k=k)\nselector.fit(X_train, y_train)\n \nselected_features = [FEATURE_COLS[i] for i, m in enumerate(selector.get_support()) if m]\nprint(f\"\\nTop-{k} selected features: {selected_features}\")\n \nX_train_sel = selector.transform(X_train)\nX_val_sel   = selector.transform(X_val)\nX_test_sel  = selector.transform(X_test)\n \n# Feature importance plot\nplt.figure(figsize=(10, 4))\nplt.bar(FEATURE_COLS, selector.scores_)\nplt.xticks(rotation=45, ha='right')\nplt.title('Mutual Information Scores')\nplt.tight_layout()\nplt.savefig('feature_importance.png', dpi=150)\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# SCALING","metadata":{}},{"cell_type":"code","source":"scaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train_sel)\nX_val_scaled   = scaler.transform(X_val_sel)\nX_test_scaled  = scaler.transform(X_test_sel)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PCA","metadata":{}},{"cell_type":"code","source":"X_train_pca = X_train_scaled\nX_val_pca   = X_val_scaled\nX_test_pca  = X_test_scaled\nprint(f\"Skipping PCA — using all {X_train_scaled.shape[1]} scaled features directly\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# SMOTE","metadata":{}},{"cell_type":"code","source":"sm = SMOTE(random_state=42)\nX_train_res, y_train_res = sm.fit_resample(X_train_pca, y_train)\nprint(f\"\\nAfter SMOTE — shape: {X_train_res.shape}, dist: {np.bincount(y_train_res)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Models","metadata":{}},{"cell_type":"markdown","source":"\n\n","metadata":{},"attachments":{}},{"cell_type":"markdown","source":"# TRAIN","metadata":{}},{"cell_type":"code","source":"models = {\n    'SVM':           SVC(\n    C=10,\n    gamma='scale',\n    kernel='rbf',\n    probability=True,\n    class_weight='balanced'\n),\n    'Logistic Reg':  LogisticRegression(max_iter=1000, random_state=42),\n    'Naive Bayes':   GaussianNB(),\n    'Random Forest': RandomForestClassifier(\n    n_estimators=300,\n    max_depth=15,\n    min_samples_split=5,\n    class_weight='balanced'\n),\n    'XGBoost':       XGBClassifier(\n    n_estimators=300,\n    max_depth=6,\n    learning_rate=0.03,\n    subsample=0.8,\n    colsample_bytree=0.8,\n    eval_metric='logloss'\n),\n}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results = {}\n \nprint(\"\\n\" + \"=\"*75)\nprint(f\"{'Model':<18} {'Val-F1':>7} {'Val-AUC':>8} {'Test-Acc':>9} {'Test-Rec':>9} {'Test-F1':>8} {'Test-AUC':>9}\")\nprint(\"=\"*75)\n \nfor name, clf in models.items():\n    # Train\n    clf.fit(X_train_res, y_train_res)\n \n    # Validation\n    y_val_pred  = clf.predict(X_val_pca)\n    y_val_proba = clf.predict_proba(X_val_pca)[:, 1]\n    val_f1  = f1_score(y_val, y_val_pred)\n    val_auc = roc_auc_score(y_val, y_val_proba)\n \n    # Test\n    y_pred  = clf.predict(X_test_pca)\n    y_proba = clf.predict_proba(X_test_pca)[:, 1]\n    acc  = accuracy_score(y_test, y_pred)\n    prec = precision_score(y_test, y_pred)\n    rec  = recall_score(y_test, y_pred)\n    f1   = f1_score(y_test, y_pred)\n    auc  = roc_auc_score(y_test, y_proba)\n \n    results[name] = dict(acc=acc, precision=prec, recall=rec, f1=f1, auc=auc,\n                         val_f1=val_f1, val_auc=val_auc,\n                         y_pred=y_pred, y_proba=y_proba)\n \n    print(f\"{name:<18} {val_f1:7.3f} {val_auc:8.3f} {acc:9.3f} {rec:9.3f} {f1:8.3f} {auc:9.3f}\")\n \nprint(\"=\"*75)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# DETAILED TEST REPORTS","metadata":{}},{"cell_type":"code","source":"for name, res in results.items():\n    print(f\"\\n── {name} ──\")\n    print(classification_report(y_test, res['y_pred'],\n                                 target_names=['Normal', 'Pneumonia']))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# CONFUSION MATRICES","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 3, figsize=(15, 9))\naxes = axes.flatten()\nfor ax, (name, res) in zip(axes, results.items()):\n    cm = confusion_matrix(y_test, res['y_pred'])\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', ax=ax,\n                xticklabels=['Normal', 'Pneumonia'],\n                yticklabels=['Normal', 'Pneumonia'])\n    ax.set_title(name); ax.set_xlabel('Predicted'); ax.set_ylabel('Actual')\naxes[-1].axis('off')\nplt.suptitle('Confusion Matrices (Test Set)', fontsize=14)\nplt.tight_layout(); plt.savefig('confusion_matrices.png', dpi=150); plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ROC CURVES","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nfor name, res in results.items():\n    fpr, tpr, _ = roc_curve(y_test, res['y_proba'])\n    plt.plot(fpr, tpr, label=f\"{name} (AUC={res['auc']:.3f})\")\nplt.plot([0,1],[0,1],'k--'); plt.xlabel('FPR'); plt.ylabel('TPR')\nplt.title('ROC Curves – Test Set'); plt.legend(loc='lower right')\nplt.tight_layout(); plt.savefig('roc_curves.png', dpi=150); plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# SUMMARY BAR CHART","metadata":{}},{"cell_type":"code","source":"metrics = ['acc', 'precision', 'recall', 'f1', 'auc']\nsummary = pd.DataFrame({name: [res[m] for m in metrics]\n                        for name, res in results.items()}, index=metrics)\nsummary.T.plot(kind='bar', figsize=(12, 5), ylim=(0, 1.05))\nplt.title('Model Comparison (Test Set)'); plt.ylabel('Score')\nplt.xticks(rotation=20); plt.legend(loc='lower right')\nplt.tight_layout(); plt.savefig('model_comparison.png', dpi=150); plt.show()\n \nprint(\"\\n── Best Models ──\")\nprint(\"Best F1:     \", max(results, key=lambda n: results[n]['f1']))\nprint(\"Best Recall: \", max(results, key=lambda n: results[n]['recall']))\nprint(\"Best AUC:    \", max(results, key=lambda n: results[n]['auc']))","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}