{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Step 1: Setup & Data Acquisition","metadata":{}},{"cell_type":"markdown","source":"## Install Needed Libraries","metadata":{}},{"cell_type":"code","source":"!pip install -q pydicom scikit-image scikit-learn \\\n               xgboost imbalanced-learn albumentations","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:43.556399Z","iopub.execute_input":"2026-05-08T19:19:43.557007Z","iopub.status.idle":"2026-05-08T19:19:47.042308Z","shell.execute_reply.started":"2026-05-08T19:19:43.556974Z","shell.execute_reply":"2026-05-08T19:19:47.041366Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Set Path","metadata":{}},{"cell_type":"code","source":"import os\n\nDATA_DIR  = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\nTRAIN_DIR = f'{DATA_DIR}/stage_2_train_images'\nOUTPUT_DIR = '/kaggle/working'\n\nprint('📁 Dataset contents:')\nfor item in sorted(os.listdir(DATA_DIR)):\n    print(f'   {item}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:47.043926Z","iopub.execute_input":"2026-05-08T19:19:47.044257Z","iopub.status.idle":"2026-05-08T19:19:47.050176Z","shell.execute_reply.started":"2026-05-08T19:19:47.044227Z","shell.execute_reply":"2026-05-08T19:19:47.049541Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load and Preview Data","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\ndf = pd.read_csv(f'{DATA_DIR}/stage_2_train_labels.csv')\nprint('Shape:', df.shape)\nprint(df.head(8))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:47.051189Z","iopub.execute_input":"2026-05-08T19:19:47.051502Z","iopub.status.idle":"2026-05-08T19:19:47.373073Z","shell.execute_reply.started":"2026-05-08T19:19:47.051454Z","shell.execute_reply":"2026-05-08T19:19:47.372232Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Class Distribution","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\npatients = df.groupby('patientId')['Target'].max().reset_index()\ncounts   = patients['Target'].value_counts()\n\nprint(f'Normal    (0): {counts[0]:,} ({counts[0]/len(patients)*100:.1f}%)')\nprint(f'Pneumonia (1): {counts[1]:,} ({counts[1]/len(patients)*100:.1f}%)')\nprint(f'Total        : {len(patients):,}')\n\nfig, ax = plt.subplots(figsize=(5,3))\nax.bar(['Normal','Pneumonia'], [counts[0], counts[1]],\n       color=['steelblue','tomato'], edgecolor='white', width=0.5)\nax.set_title('Class Distribution — RSNA Dataset')\nax.set_ylabel('Patients')\nfor i, v in enumerate([counts[0], counts[1]]):\n    ax.text(i, v+50, f'{v:,}', ha='center', fontweight='bold')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:47.374809Z","iopub.execute_input":"2026-05-08T19:19:47.375134Z","iopub.status.idle":"2026-05-08T19:19:47.503966Z","shell.execute_reply.started":"2026-05-08T19:19:47.375110Z","shell.execute_reply":"2026-05-08T19:19:47.503365Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualize Sample X-Rays with Bounding Boxes","metadata":{}},{"cell_type":"code","source":"import pydicom\nimport numpy as np\nimport matplotlib.patches as patches\n\ndef apply_window(img, level=-500, width=1500):\n    low  = level - width // 2\n    high = level + width // 2\n    img  = np.clip(img, low, high)\n    img  = (img - low) / (high - low)\n    return img\n\nfig, axes = plt.subplots(2, 4, figsize=(16, 8))\naxes = axes.flatten()\n\nnormal_ids    = patients[patients['Target']==0]['patientId'].sample(4, random_state=42).values\npneumonia_ids = patients[patients['Target']==1]['patientId'].sample(4, random_state=42).values\nsample_ids    = list(normal_ids) + list(pneumonia_ids)\nlabels_list   = ['Normal']*4 + ['Pneumonia']*4\n\nfor idx, (pid, label) in enumerate(zip(sample_ids, labels_list)):\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = apply_window(dcm.pixel_array.astype(np.float32))\n\n    axes[idx].imshow(img, cmap='gray')\n    axes[idx].set_title(label,\n                        color='tomato' if label=='Pneumonia' else 'steelblue',\n                        fontweight='bold')\n    axes[idx].axis('off')\n\n    if label == 'Pneumonia':\n        for _, row in df[df['patientId']==pid][['x','y','width','height']].dropna().iterrows():\n            axes[idx].add_patch(patches.Rectangle(\n                (row['x'], row['y']), row['width'], row['height'],\n                linewidth=2, edgecolor='red', facecolor='none'))\n\nplt.suptitle('Sample Chest X-Rays — red boxes = pneumonia regions', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:47.504880Z","iopub.execute_input":"2026-05-08T19:19:47.505375Z","iopub.status.idle":"2026-05-08T19:19:49.868775Z","shell.execute_reply.started":"2026-05-08T19:19:47.505318Z","shell.execute_reply":"2026-05-08T19:19:49.867804Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Save Global Config","metadata":{}},{"cell_type":"code","source":"import json\n\nDATA_DIR  = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\nTRAIN_DIR = f'{DATA_DIR}/stage_2_train_images'\nOUTPUT_DIR = '/kaggle/working'\n\nCONFIG = {\n    'data_dir'    : DATA_DIR,\n    'train_dcm'   : TRAIN_DIR,\n    'labels_csv'  : f'{DATA_DIR}/stage_2_train_labels.csv',\n    'output_dir'  : OUTPUT_DIR,\n    'img_size'    : 512,\n    'window_level': -500,\n    'window_width': 1500,\n    'train_ratio' : 0.70,\n    'val_ratio'   : 0.15,\n    'test_ratio'  : 0.15,\n    'random_seed' : 42,\n}\n\nprint('✅ Config ready.')\nprint(json.dumps(CONFIG, indent=2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:49.870143Z","iopub.execute_input":"2026-05-08T19:19:49.870473Z","iopub.status.idle":"2026-05-08T19:19:49.876886Z","shell.execute_reply.started":"2026-05-08T19:19:49.870448Z","shell.execute_reply":"2026-05-08T19:19:49.875877Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 2: EDA & Visualization","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport pydicom\nimport cv2\nimport os\nfrom tqdm import tqdm\n\nDATA_DIR  = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\nTRAIN_DIR = f'{DATA_DIR}/stage_2_train_images'\n\ndf       = pd.read_csv(f'{DATA_DIR}/stage_2_train_labels.csv')\ndf_class = pd.read_csv(f'{DATA_DIR}/stage_2_detailed_class_info.csv')\npatients = df.groupby('patientId')['Target'].max().reset_index()\n\nprint('Labels shape    :', df.shape)\nprint('Class info shape:', df_class.shape)\nprint(df_class['class'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:49.878017Z","iopub.execute_input":"2026-05-08T19:19:49.878306Z","iopub.status.idle":"2026-05-08T19:19:50.088675Z","shell.execute_reply.started":"2026-05-08T19:19:49.878274Z","shell.execute_reply":"2026-05-08T19:19:50.088012Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Missing Values","metadata":{}},{"cell_type":"code","source":"print('=== Labels CSV ===')\nprint(df.isnull().sum())\nprint(f'\\nTotal missing: {df.isnull().sum().sum()}')\n\nprint('\\n=== Class Info CSV ===')\nprint(df_class.isnull().sum())\nprint(f'\\nTotal missing: {df_class.isnull().sum().sum()}')\n\nprint('\\n=== Missing Bounding Boxes (Normal patients have NaN boxes) ===')\nprint(f'Rows with NaN boxes : {df[df[\"x\"].isnull()].shape[0]:,}')\nprint(f'Rows with boxes     : {df[df[\"x\"].notna()].shape[0]:,}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:50.089719Z","iopub.execute_input":"2026-05-08T19:19:50.090020Z","iopub.status.idle":"2026-05-08T19:19:50.111386Z","shell.execute_reply.started":"2026-05-08T19:19:50.089984Z","shell.execute_reply":"2026-05-08T19:19:50.110569Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Patient Age & Gender Distribution","metadata":{}},{"cell_type":"code","source":"detail_df = df_class.drop_duplicates('patientId').copy()\n\n# Check available columns\nprint('Available columns:', detail_df.columns.tolist())\nprint(detail_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:50.112466Z","iopub.execute_input":"2026-05-08T19:19:50.112769Z","iopub.status.idle":"2026-05-08T19:19:50.121849Z","shell.execute_reply.started":"2026-05-08T19:19:50.112744Z","shell.execute_reply":"2026-05-08T19:19:50.121077Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\n\nrecords = []\nsample_pids = patients.sample(300, random_state=42)['patientId'].values\n\nfor pid in tqdm(sample_pids):\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    target = patients[patients['patientId']==pid]['Target'].values[0]\n    \n    age    = getattr(dcm, 'PatientAge',  None)\n    gender = getattr(dcm, 'PatientSex',  None)\n    \n    records.append({\n        'patientId': pid,\n        'age'      : age,\n        'gender'   : gender,\n        'target'   : target\n    })\n\ndf_demo = pd.DataFrame(records)\nprint('Sample demographics:')\nprint(df_demo.head(10))\nprint('\\nGender counts:')\nprint(df_demo['gender'].value_counts())\nprint('\\nAge sample:')\nprint(df_demo['age'].value_counts().head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:50.125557Z","iopub.execute_input":"2026-05-08T19:19:50.126331Z","iopub.status.idle":"2026-05-08T19:19:51.602491Z","shell.execute_reply.started":"2026-05-08T19:19:50.126305Z","shell.execute_reply":"2026-05-08T19:19:51.601726Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Plot Age & Gender","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize=(16, 4))\n\n# Gender distribution\ngender_counts = df_demo['gender'].value_counts()\naxes[0].bar(gender_counts.index, gender_counts.values,\n            color=['steelblue','tomato'], edgecolor='white', width=0.4)\naxes[0].set_title('Patient Gender Distribution')\naxes[0].set_ylabel('Count')\naxes[0].set_xlabel('Gender')\nfor i, v in enumerate(gender_counts.values):\n    axes[0].text(i, v+1, f'{v:,}', ha='center', fontweight='bold')\n\n# Age distribution by class\ndf_demo['age'] = pd.to_numeric(df_demo['age'], errors='coerce')\nnormal_ages    = df_demo[df_demo['target']==0]['age'].dropna()\npneumonia_ages = df_demo[df_demo['target']==1]['age'].dropna()\n\naxes[1].hist(normal_ages,    bins=20, alpha=0.7, color='steelblue', label='Normal')\naxes[1].hist(pneumonia_ages, bins=20, alpha=0.7, color='tomato',    label='Pneumonia')\naxes[1].set_title('Age Distribution by Class')\naxes[1].set_xlabel('Age (years)')\naxes[1].set_ylabel('Count')\naxes[1].legend()\n\n# Gender by class\ngender_class = df_demo.groupby(['gender','target']).size().unstack(fill_value=0)\ngender_class.plot(kind='bar', ax=axes[2],\n                  color=['steelblue','tomato'], edgecolor='white')\naxes[2].set_title('Gender by Class')\naxes[2].set_xlabel('Gender')\naxes[2].set_ylabel('Count')\naxes[2].legend(['Normal','Pneumonia'])\naxes[2].tick_params(axis='x', rotation=0)\n\nplt.suptitle('Patient Demographics — RSNA Dataset', fontsize=13)\nplt.tight_layout()\nplt.show()\n\nprint(f'Male   — Total: {gender_counts.get(\"M\",0):,}')\nprint(f'Female — Total: {gender_counts.get(\"F\",0):,}')\nprint(f'\\nNormal    — Mean age: {normal_ages.mean():.1f} yrs')\nprint(f'Pneumonia — Mean age: {pneumonia_ages.mean():.1f} yrs')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:51.603505Z","iopub.execute_input":"2026-05-08T19:19:51.603831Z","iopub.status.idle":"2026-05-08T19:19:52.050193Z","shell.execute_reply.started":"2026-05-08T19:19:51.603805Z","shell.execute_reply":"2026-05-08T19:19:52.049272Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Class Distribution","metadata":{}},{"cell_type":"code","source":"class_counts = df_class.drop_duplicates('patientId')['class'].value_counts()\n\nfig, axes = plt.subplots(1, 2, figsize=(12, 4))\n\n# Bar chart\naxes[0].bar(class_counts.index, class_counts.values,\n            color=['steelblue', 'tomato', 'orange'], edgecolor='white')\naxes[0].set_title('Detailed Class Distribution')\naxes[0].set_ylabel('Number of Patients')\naxes[0].tick_params(axis='x', rotation=15)\nfor i, v in enumerate(class_counts.values):\n    axes[0].text(i, v+50, f'{v:,}', ha='center', fontweight='bold')\n\n# Pie chart\naxes[1].pie(class_counts.values, labels=class_counts.index,\n            colors=['steelblue', 'tomato', 'orange'],\n            autopct='%1.1f%%', startangle=90)\naxes[1].set_title('Class Proportions')\n\nplt.suptitle('RSNA Dataset — Class Distribution', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:52.051336Z","iopub.execute_input":"2026-05-08T19:19:52.052177Z","iopub.status.idle":"2026-05-08T19:19:52.239263Z","shell.execute_reply.started":"2026-05-08T19:19:52.052135Z","shell.execute_reply":"2026-05-08T19:19:52.238713Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Bounding Box Statistics","metadata":{}},{"cell_type":"code","source":"df_pos = df[df['Target'] == 1].copy()\n\nprint('Total bounding boxes      :', len(df_pos))\nprint('Patients with pneumonia   :', df_pos['patientId'].nunique())\nprint('Avg boxes per patient     :', round(len(df_pos)/df_pos['patientId'].nunique(), 2))\nprint()\nprint('Bounding Box Size Stats:')\nprint(df_pos[['width','height']].describe().round(1))\n\n# Box count per patient\nbox_counts = df_pos.groupby('patientId').size()\n\nfig, axes = plt.subplots(1, 3, figsize=(15, 4))\n\n# Box count distribution\naxes[0].hist(box_counts.values, bins=range(1, box_counts.max()+2),\n             color='tomato', edgecolor='white', align='left')\naxes[0].set_title('Number of Boxes per Pneumonia Patient')\naxes[0].set_xlabel('Number of Bounding Boxes')\naxes[0].set_ylabel('Count')\n\n# Box width distribution\naxes[1].hist(df_pos['width'].dropna(), bins=40, color='steelblue', edgecolor='white')\naxes[1].set_title('Bounding Box Width Distribution')\naxes[1].set_xlabel('Width (pixels)')\naxes[1].set_ylabel('Count')\n\n# Box height distribution\naxes[2].hist(df_pos['height'].dropna(), bins=40, color='purple', edgecolor='white')\naxes[2].set_title('Bounding Box Height Distribution')\naxes[2].set_xlabel('Height (pixels)')\naxes[2].set_ylabel('Count')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:52.240209Z","iopub.execute_input":"2026-05-08T19:19:52.240630Z","iopub.status.idle":"2026-05-08T19:19:52.704338Z","shell.execute_reply.started":"2026-05-08T19:19:52.240592Z","shell.execute_reply":"2026-05-08T19:19:52.703601Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Bounding Box Area Distribution","metadata":{}},{"cell_type":"code","source":"df_pos['area'] = df_pos['width'] * df_pos['height']\n\nfig, axes = plt.subplots(1, 2, figsize=(12, 4))\n\n# Area histogram\naxes[0].hist(df_pos['area'].dropna(), bins=40, color='steelblue', edgecolor='white')\naxes[0].set_title('Bounding Box Area Distribution')\naxes[0].set_xlabel('Area (pixels²)')\naxes[0].set_ylabel('Count')\naxes[0].axvline(df_pos['area'].mean(), color='red', linestyle='--',\n                label=f'Mean: {df_pos[\"area\"].mean():,.0f}')\naxes[0].legend()\n\n# Width vs Height scatter\naxes[1].scatter(df_pos['width'], df_pos['height'],\n                alpha=0.2, s=10, color='tomato')\naxes[1].set_title('Bounding Box Width vs Height')\naxes[1].set_xlabel('Width (pixels)')\naxes[1].set_ylabel('Height (pixels)')\n\nprint(f'Mean area  : {df_pos[\"area\"].mean():,.0f} px²')\nprint(f'Median area: {df_pos[\"area\"].median():,.0f} px²')\nprint(f'Min area   : {df_pos[\"area\"].min():,.0f} px²')\nprint(f'Max area   : {df_pos[\"area\"].max():,.0f} px²')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:52.705429Z","iopub.execute_input":"2026-05-08T19:19:52.705787Z","iopub.status.idle":"2026-05-08T19:19:53.032609Z","shell.execute_reply.started":"2026-05-08T19:19:52.705762Z","shell.execute_reply":"2026-05-08T19:19:53.031843Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Image Metadata Analysis","metadata":{}},{"cell_type":"code","source":"import pydicom\n\nsample = pydicom.dcmread(f'{TRAIN_DIR}/{patients[\"patientId\"].iloc[0]}.dcm')\nprint('Image size  :', sample.Rows, 'x', sample.Columns)\nprint('Bits        :', sample.BitsAllocated)\nprint('Modality    :', sample.Modality)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:53.033685Z","iopub.execute_input":"2026-05-08T19:19:53.034009Z","iopub.status.idle":"2026-05-08T19:19:53.042029Z","shell.execute_reply.started":"2026-05-08T19:19:53.033975Z","shell.execute_reply":"2026-05-08T19:19:53.040968Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Pixel Intensity Analysis","metadata":{}},{"cell_type":"code","source":"print('Analyzing pixel intensities from 100 samples (50 normal, 50 pneumonia)...')\n\ndef apply_window(img, level=-500, width=1500):\n    low  = level - width // 2\n    high = level + width // 2\n    img  = np.clip(img, low, high)\n    img  = (img - low) / (high - low)\n    return img\n\nnormal_pids    = patients[patients['Target']==0]['patientId'].sample(50, random_state=42).values\npneumonia_pids = patients[patients['Target']==1]['patientId'].sample(50, random_state=42).values\n\nnormal_means, pneumonia_means = [], []\nnormal_stds,  pneumonia_stds  = [], []\n\nfor pid in tqdm(normal_pids):\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = apply_window(dcm.pixel_array.astype(np.float32))\n    normal_means.append(img.mean())\n    normal_stds.append(img.std())\n\nfor pid in tqdm(pneumonia_pids):\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = apply_window(dcm.pixel_array.astype(np.float32))\n    pneumonia_means.append(img.mean())\n    pneumonia_stds.append(img.std())\n\nfig, axes = plt.subplots(1, 2, figsize=(12, 4))\n\n# Mean intensity\naxes[0].hist(normal_means,    bins=20, alpha=0.7, color='steelblue', label='Normal')\naxes[0].hist(pneumonia_means, bins=20, alpha=0.7, color='tomato',    label='Pneumonia')\naxes[0].set_title('Mean Pixel Intensity Distribution')\naxes[0].set_xlabel('Mean Intensity')\naxes[0].set_ylabel('Count')\naxes[0].legend()\n\n# Std intensity\naxes[1].hist(normal_stds,    bins=20, alpha=0.7, color='steelblue', label='Normal')\naxes[1].hist(pneumonia_stds, bins=20, alpha=0.7, color='tomato',    label='Pneumonia')\naxes[1].set_title('Pixel Intensity Std Distribution')\naxes[1].set_xlabel('Std of Intensity')\naxes[1].set_ylabel('Count')\naxes[1].legend()\n\nplt.tight_layout()\nplt.show()\n\nprint(f'\\nNormal    — Mean: {np.mean(normal_means):.3f}, Std: {np.mean(normal_stds):.3f}')\nprint(f'Pneumonia — Mean: {np.mean(pneumonia_means):.3f}, Std: {np.mean(pneumonia_stds):.3f}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:53.043384Z","iopub.execute_input":"2026-05-08T19:19:53.043799Z","iopub.status.idle":"2026-05-08T19:19:54.574588Z","shell.execute_reply.started":"2026-05-08T19:19:53.043760Z","shell.execute_reply":"2026-05-08T19:19:54.573945Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train / Val / Test","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nSEED = 42\n\n# First split: train vs temp (val+test)\ntrain_df, temp_df = train_test_split(patients, test_size=0.30,\n                                     stratify=patients['Target'],\n                                     random_state=SEED)\n\n# Second split: val vs test\nval_df, test_df = train_test_split(temp_df, test_size=0.50,\n                                   stratify=temp_df['Target'],\n                                   random_state=SEED)\n\nprint(f'Train : {len(train_df):,} ({len(train_df)/len(patients)*100:.1f}%)')\nprint(f'Val   : {len(val_df):,}  ({len(val_df)/len(patients)*100:.1f}%)')\nprint(f'Test  : {len(test_df):,}  ({len(test_df)/len(patients)*100:.1f}%)')\n\n# Verify stratification\nfor name, split in [('Train', train_df), ('Val', val_df), ('Test', test_df)]:\n    pos = split['Target'].sum()\n    print(f'{name} — Pneumonia: {pos} ({pos/len(split)*100:.1f}%)')\n\n# Save splits\ntrain_df.to_csv('/kaggle/working/train_split.csv', index=False)\nval_df.to_csv('/kaggle/working/val_split.csv',   index=False)\ntest_df.to_csv('/kaggle/working/test_split.csv',  index=False)\nprint('\\n✅ Splits saved to /kaggle/working/')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:54.575471Z","iopub.execute_input":"2026-05-08T19:19:54.576128Z","iopub.status.idle":"2026-05-08T19:19:55.294078Z","shell.execute_reply.started":"2026-05-08T19:19:54.576091Z","shell.execute_reply":"2026-05-08T19:19:55.293381Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 3: Preprocessing","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport pydicom\nimport cv2\nimport os\nfrom tqdm import tqdm\n\nDATA_DIR  = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\nTRAIN_DIR = f'{DATA_DIR}/stage_2_train_images'\nOUTPUT_DIR = '/kaggle/working'\n\ndf       = pd.read_csv(f'{DATA_DIR}/stage_2_train_labels.csv')\npatients = df.groupby('patientId')['Target'].max().reset_index()\n\ntrain_df = pd.read_csv(f'{OUTPUT_DIR}/train_split.csv')\nval_df   = pd.read_csv(f'{OUTPUT_DIR}/val_split.csv')\ntest_df  = pd.read_csv(f'{OUTPUT_DIR}/test_split.csv')\n\nprint('✅ Imports done.')\nprint(f'Train: {len(train_df):,} | Val: {len(val_df):,} | Test: {len(test_df):,}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:55.294899Z","iopub.execute_input":"2026-05-08T19:19:55.295250Z","iopub.status.idle":"2026-05-08T19:19:55.365014Z","shell.execute_reply.started":"2026-05-08T19:19:55.295213Z","shell.execute_reply":"2026-05-08T19:19:55.364149Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Functions","metadata":{}},{"cell_type":"code","source":"def load_dicom(pid):\n    \"\"\"Load raw DICOM pixel array.\"\"\"\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = dcm.pixel_array.astype(np.float32)\n    return img\n\ndef apply_window(img, level=-500, width=1500):\n    \"\"\"Apply lung windowing to enhance pulmonary structures.\"\"\"\n    low  = level - width // 2\n    high = level + width // 2\n    img  = np.clip(img, low, high)\n    img  = (img - low) / (high - low)\n    return img\n\ndef normalize(img):\n    \"\"\"Normalize image to [0, 1].\"\"\"\n    img = img - img.min()\n    if img.max() > 0:\n        img = img / img.max()\n    return img\n    \ndef enhance_contrast(img):\n\n    img_uint8 = (img * 255).astype(np.uint8)\n\n    clahe = cv2.createCLAHE(\n        clipLimit=2.0,\n        tileGridSize=(8,8)\n    )\n\n    enhanced = clahe.apply(img_uint8)\n\n    return enhanced / 255.0\n    \ndef denoise(img):\n    \"\"\"Apply Gaussian blur for noise reduction.\"\"\"\n    return cv2.GaussianBlur(img, (3, 3), 0)\n    \ndef resize_img(img, size=512):\n    \"\"\"Resize image to fixed size.\"\"\"\n    return cv2.resize(img, (size, size), interpolation=cv2.INTER_AREA)\n\ndef preprocess(pid, size=512):\n    \"\"\"Full preprocessing pipeline for one patient.\"\"\"\n    img = load_dicom(pid)       # Step 1: Load DICOM\n    img = normalize(img)        # Step 2: Normalize to [0,1] first\n    img = enhance_contrast(img)          # Step 3: enhance_contrast\n    img = resize_img(img, size) # Step 4: Resize to 512x512\n    return img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:55.366041Z","iopub.execute_input":"2026-05-08T19:19:55.366746Z","iopub.status.idle":"2026-05-08T19:19:55.458491Z","shell.execute_reply.started":"2026-05-08T19:19:55.366710Z","shell.execute_reply":"2026-05-08T19:19:55.457729Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Test Preprocessing on Single Image","metadata":{}},{"cell_type":"code","source":"pid = patients['patientId'].iloc[0]\n\nraw     = load_dicom(pid)\nnormed  = normalize(raw)\nenhanced = enhance_contrast(normed)\nresized  = resize_img(enhanced)\n\nprint(f'Raw      — shape: {raw.shape},     min: {raw.min():.1f},   max: {raw.max():.1f}')\nprint(f'Normed   — shape: {normed.shape},   min: {normed.min():.2f},  max: {normed.max():.2f}')\nprint(f'Enhanced — shape: {enhanced.shape}, min: {enhanced.min():.2f},  max: {enhanced.max():.2f}')\nprint(f'Resized  — shape: {resized.shape},  min: {resized.min():.2f},  max: {resized.max():.2f}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:55.459486Z","iopub.execute_input":"2026-05-08T19:19:55.459863Z","iopub.status.idle":"2026-05-08T19:19:55.501048Z","shell.execute_reply.started":"2026-05-08T19:19:55.459827Z","shell.execute_reply":"2026-05-08T19:19:55.500189Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Augmentation","metadata":{}},{"cell_type":"code","source":"import albumentations as A\n\naugment = A.Compose([\n    A.HorizontalFlip(p=0.5),\n    A.Rotate(limit=10, p=0.3),\n    A.Affine(translate_percent=0.1, p=0.3),\n    A.RandomBrightnessContrast(brightness_limit=0.2,\n                               contrast_limit=0.2, p=0.4),\n])\n\ndef augment_img(img):\n    \"\"\"Apply augmentation to a single image.\"\"\"\n    img_uint8 = (img * 255).astype(np.uint8)\n    augmented = augment(image=img_uint8)['image']\n    return augmented.astype(np.float32) / 255.0\n\nprint('✅ Augmentation pipeline defined.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:55.502135Z","iopub.execute_input":"2026-05-08T19:19:55.502992Z","iopub.status.idle":"2026-05-08T19:19:57.991884Z","shell.execute_reply.started":"2026-05-08T19:19:55.502948Z","shell.execute_reply":"2026-05-08T19:19:57.991188Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualize Augmentation Effects","metadata":{}},{"cell_type":"code","source":"sample_pid = patients[patients['Target']==1]['patientId'].sample(1, random_state=7).values[0]\nimg = preprocess(sample_pid)\n\nfig, axes = plt.subplots(2, 4, figsize=(16, 8))\naxes = axes.flatten()\n\n# Original\naxes[0].imshow(img, cmap='gray')\naxes[0].set_title('Original (preprocessed)')\naxes[0].axis('off')\n\n# 7 augmented versions\nfor i in range(1, 8):\n    aug = augment_img(img)\n    axes[i].imshow(aug, cmap='gray')\n    axes[i].set_title(f'Augmented #{i}')\n    axes[i].axis('off')\n\nplt.suptitle('Data Augmentation Examples', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:57.992908Z","iopub.execute_input":"2026-05-08T19:19:57.993350Z","iopub.status.idle":"2026-05-08T19:19:59.127195Z","shell.execute_reply.started":"2026-05-08T19:19:57.993323Z","shell.execute_reply":"2026-05-08T19:19:59.126332Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Before VS After Preprocessing","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 4, figsize=(16, 12))\n\nsample_pids = patients.sample(4, random_state=10)['patientId'].values\n\nfor idx, pid in enumerate(sample_pids):\n    raw      = load_dicom(pid)\n    windowed = apply_window(raw)\n    resized  = resize_img(normalize(windowed))\n\n    # Row 1: Raw\n    axes[0][idx].imshow(raw, cmap='gray')\n    axes[0][idx].set_title(f'Raw DICOM\\n{raw.shape}')\n    axes[0][idx].axis('off')\n\n    # Row 2: After windowing\n    axes[1][idx].imshow(windowed, cmap='gray')\n    axes[1][idx].set_title(f'After Windowing\\n[{windowed.min():.2f}, {windowed.max():.2f}]')\n    axes[1][idx].axis('off')\n\n    # Row 3: After resize\n    axes[2][idx].imshow(resized, cmap='gray')\n    axes[2][idx].set_title(f'Final (512x512)\\n[{resized.min():.2f}, {resized.max():.2f}]')\n    axes[2][idx].axis('off')\n\nplt.suptitle('Preprocessing Pipeline: Raw → Windowed → Final', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:19:59.128334Z","iopub.execute_input":"2026-05-08T19:19:59.129012Z","iopub.status.idle":"2026-05-08T19:20:01.772080Z","shell.execute_reply.started":"2026-05-08T19:19:59.128985Z","shell.execute_reply":"2026-05-08T19:20:01.771036Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Histogram Update","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 4, figsize=(16, 12))\n\nsample_pids = patients.sample(4, random_state=10)['patientId'].values\n\nfor idx, pid in enumerate(sample_pids):\n    raw      = load_dicom(pid)\n    normed   = normalize(raw)\n    final    = resize_img(enhance_contrast(normed))\n\n    # Row 1: Raw histogram\n    axes[0][idx].hist(raw.flatten(), bins=50, color='gray', edgecolor='none')\n    axes[0][idx].set_title(f'Raw\\nmin={raw.min():.0f} max={raw.max():.0f}')\n    axes[0][idx].set_xlabel('Pixel Value')\n    axes[0][idx].set_ylabel('Count')\n\n    # Row 2: After normalization\n    axes[1][idx].hist(normed.flatten(), bins=50, color='steelblue', edgecolor='none')\n    axes[1][idx].set_title(f'After Normalize\\nmin={normed.min():.2f} max={normed.max():.2f}')\n    axes[1][idx].set_xlabel('Pixel Value')\n    axes[1][idx].set_ylabel('Count')\n\n    # Row 3: Final (enhanced + resized)\n    axes[2][idx].hist(final.flatten(), bins=50, color='tomato', edgecolor='none')\n    axes[2][idx].set_title(f'Final (enhanced+resized)\\nmin={final.min():.2f} max={final.max():.2f}')\n    axes[2][idx].set_xlabel('Pixel Value')\n    axes[2][idx].set_ylabel('Count')\n\nplt.suptitle('Preprocessing Pipeline: Raw → Normalized → Final', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:20:01.773346Z","iopub.execute_input":"2026-05-08T19:20:01.774209Z","iopub.status.idle":"2026-05-08T19:20:03.896627Z","shell.execute_reply.started":"2026-05-08T19:20:01.774162Z","shell.execute_reply":"2026-05-08T19:20:03.895721Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 4: Lung Segmentation","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport pydicom\nimport cv2\nfrom scipy import ndimage\nfrom tqdm import tqdm\nimport os\n\nDATA_DIR  = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\nTRAIN_DIR = f'{DATA_DIR}/stage_2_train_images'\nOUTPUT_DIR = '/kaggle/working'\n\ndf       = pd.read_csv(f'{DATA_DIR}/stage_2_train_labels.csv')\npatients = df.groupby('patientId')['Target'].max().reset_index()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:20:03.897858Z","iopub.execute_input":"2026-05-08T19:20:03.898249Z","iopub.status.idle":"2026-05-08T19:20:03.950598Z","shell.execute_reply.started":"2026-05-08T19:20:03.898220Z","shell.execute_reply":"2026-05-08T19:20:03.950018Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Segmentation Function","metadata":{}},{"cell_type":"code","source":"def segment_lungs(img):\n    \"\"\"\n    Segment lung region from chest X-ray using\n    thresholding + morphological operations.\n    Returns: binary lung mask\n    \"\"\"\n    # Step 1: Convert to uint8\n    img_uint8 = (img * 255).astype(np.uint8)\n\n    # Step 2: Otsu thresholding — separates lung from background\n    _, thresh = cv2.threshold(img_uint8, 0, 255,\n                              cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n\n    # Step 3: Invert — lungs are dark in X-ray\n    thresh = cv2.bitwise_not(thresh)\n\n    # Step 4: Morphological opening — remove small noise\n    kernel  = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (5, 5))\n    opened  = cv2.morphologyEx(thresh, cv2.MORPH_OPEN, kernel)\n\n    # Step 5: Morphological closing — fill holes inside lungs\n    kernel2 = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (20, 20))\n    closed  = cv2.morphologyEx(opened, cv2.MORPH_CLOSE, kernel2)\n\n    # Step 6: Keep only the 2 largest components (left + right lung)\n    num_labels, labels, stats, _ = cv2.connectedComponentsWithStats(closed)\n    \n    # Sort by area (skip background = label 0)\n    areas = stats[1:, cv2.CC_STAT_AREA]\n    top2  = np.argsort(areas)[::-1][:2] + 1  # top 2 largest\n\n    mask = np.zeros_like(closed)\n    for label in top2:\n        mask[labels == label] = 255\n\n    return mask\n\ndef apply_mask(img, mask):\n    \"\"\"Apply lung mask to image — zero out background.\"\"\"\n    mask_norm = mask / 255.0\n    return img * mask_norm\n\nprint('✅ Segmentation functions defined.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:20:03.951422Z","iopub.execute_input":"2026-05-08T19:20:03.951847Z","iopub.status.idle":"2026-05-08T19:20:03.959774Z","shell.execute_reply.started":"2026-05-08T19:20:03.951815Z","shell.execute_reply":"2026-05-08T19:20:03.958955Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Test Segmentation on Sample Image","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 4, figsize=(16, 12))\n\nsample_pids = patients.sample(4, random_state=42)['patientId'].values\n\nfor idx, pid in enumerate(sample_pids):\n    # Preprocess\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = dcm.pixel_array.astype(np.float32)\n    img = img - img.min()\n    if img.max() > 0:\n        img = img / img.max()\n    img = enhance_contrast(img)\n    img = cv2.resize(img, (512, 512), interpolation=cv2.INTER_AREA)\n\n    # Segment\n    mask        = segment_lungs(img)\n    masked_img  = apply_mask(img, mask)\n\n    # Row 1: Original\n    axes[0][idx].imshow(img, cmap='gray')\n    axes[0][idx].set_title('Preprocessed')\n    axes[0][idx].axis('off')\n\n    # Row 2: Mask\n    axes[1][idx].imshow(mask, cmap='gray')\n    axes[1][idx].set_title('Lung Mask')\n    axes[1][idx].axis('off')\n\n    # Row 3: Masked image\n    axes[2][idx].imshow(masked_img, cmap='gray')\n    axes[2][idx].set_title('Segmented Lung')\n    axes[2][idx].axis('off')\n\nplt.suptitle('Lung Segmentation: Thresholding + Morphology', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:20:03.960890Z","iopub.execute_input":"2026-05-08T19:20:03.961437Z","iopub.status.idle":"2026-05-08T19:20:05.505441Z","shell.execute_reply.started":"2026-05-08T19:20:03.961412Z","shell.execute_reply":"2026-05-08T19:20:05.504553Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 4, figsize=(16, 12))\n\npneumonia_pids = patients[patients['Target']==1]['patientId'].sample(4, random_state=42).values\n\nfor idx, pid in enumerate(pneumonia_pids):\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = dcm.pixel_array.astype(np.float32)\n    img = img - img.min()\n    if img.max() > 0:\n        img = img / img.max()\n    img = enhance_contrast(img)\n    img = cv2.resize(img, (512, 512), interpolation=cv2.INTER_AREA)\n\n    mask       = segment_lungs(img)\n    masked_img = apply_mask(img, mask)\n\n    axes[0][idx].imshow(img, cmap='gray')\n    axes[0][idx].set_title('Pneumonia — Original')\n    axes[0][idx].axis('off')\n\n    axes[1][idx].imshow(mask, cmap='gray')\n    axes[1][idx].set_title('Lung Mask')\n    axes[1][idx].axis('off')\n\n    axes[2][idx].imshow(masked_img, cmap='gray')\n    axes[2][idx].set_title('Segmented Lung')\n    axes[2][idx].axis('off')\n\nplt.suptitle('Lung Segmentation on Pneumonia Cases', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:20:05.509142Z","iopub.execute_input":"2026-05-08T19:20:05.509448Z","iopub.status.idle":"2026-05-08T19:20:06.970283Z","shell.execute_reply.started":"2026-05-08T19:20:05.509423Z","shell.execute_reply":"2026-05-08T19:20:06.969413Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Test Segmentation on Pneumonia Cases","metadata":{}},{"cell_type":"markdown","source":"# Step 5: Feature Extraction","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"from scipy.stats import skew, kurtosis\nfrom skimage.feature import graycomatrix, graycoprops\nfrom skimage.measure import regionprops, label\nimport pandas as pd\nfrom tqdm import tqdm\nimport pydicom\nimport cv2\nimport os","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:20:06.971321Z","iopub.execute_input":"2026-05-08T19:20:06.971816Z","iopub.status.idle":"2026-05-08T19:20:06.999187Z","shell.execute_reply.started":"2026-05-08T19:20:06.971788Z","shell.execute_reply":"2026-05-08T19:20:06.998394Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Extraction Logic","metadata":{}},{"cell_type":"code","source":"def extract_features(img, mask):\n    features = {}\n    bool_mask = mask > 0\n    lung_pixels = img[bool_mask]\n\n    # 1. Intensity (global)\n    if len(lung_pixels) > 0:\n        features['mean']     = float(np.mean(lung_pixels))\n        features['std']      = float(np.std(lung_pixels))\n        features['skewness'] = float(skew(lung_pixels))\n        features['kurtosis'] = float(kurtosis(lung_pixels))\n    else:\n        features['mean'] = features['std'] = features['skewness'] = features['kurtosis'] = 0.0\n\n    # Histogram bins\n    hist, _ = np.histogram(lung_pixels if len(lung_pixels) > 0 else [0], bins=10, range=(0, 1))\n    hist = hist / (hist.sum() + 1e-6)\n    for i, val in enumerate(hist):\n        features[f'hist_bin_{i}'] = float(val)\n\n    # 2. GLCM Texture (multiple angles + distances)\n    img_uint8 = (img * 255).astype(np.uint8)\n    labeled_mask = label(bool_mask)\n    regions = regionprops(labeled_mask)\n\n    if regions:\n        min_row = min([r.bbox[0] for r in regions])\n        min_col = min([r.bbox[1] for r in regions])\n        max_row = max([r.bbox[2] for r in regions])\n        max_col = max([r.bbox[3] for r in regions])\n        \n        cropped_mask = bool_mask[min_row:max_row, min_col:max_col]\n        cropped_img = img_uint8[min_row:max_row, min_col:max_col]\n        cropped_img = cropped_img * cropped_mask\n\n        # Quantize to 64 levels for speed + stability\n        img_quantized = (cropped_img // 4).astype(np.uint8)\n        glcm = graycomatrix(img_quantized,\n                            distances=[1, 2, 3],\n                            angles=[0, np.pi/4, np.pi/2, 3*np.pi/4],\n                            levels=64, symmetric=True, normed=True)\n\n        for prop in ['contrast', 'energy', 'homogeneity', 'correlation', 'dissimilarity', 'ASM']:\n            features[prop] = float(np.mean(graycoprops(glcm, prop)))\n    else:\n        for prop in ['contrast', 'energy', 'homogeneity', 'correlation', 'dissimilarity', 'ASM']:\n            features[prop] = 0.0\n\n    # 3. Shape (global)\n    if regions:\n        total_area = sum([r.area for r in regions])\n        total_perimeter = sum([r.perimeter for r in regions])\n        features['area']        = float(total_area)\n        features['perimeter']   = float(total_perimeter)\n        features['circularity'] = (4 * np.pi * total_area) / (total_perimeter ** 2 + 1e-6)\n    else:\n        features['area'] = features['perimeter'] = features['circularity'] = 0.0\n\n    # 4. Per-lobe features (asymmetry is a key pneumonia indicator)\n    if regions:\n        regions_sorted = sorted(regions, key=lambda r: r.centroid[1])\n        if len(regions_sorted) >= 2:\n            left_lung, right_lung = regions_sorted[0], regions_sorted[1]\n        else:\n            left_lung = right_lung = regions_sorted[0]\n\n        for side, region in zip(['left', 'right'], [left_lung, right_lung]):\n            lobe_mask   = labeled_mask == region.label\n            lobe_pixels = img[lobe_mask]\n            features[f'{side}_mean'] = float(np.mean(lobe_pixels)) if len(lobe_pixels) > 0 else 0.0\n            features[f'{side}_std']  = float(np.std(lobe_pixels))  if len(lobe_pixels) > 0 else 0.0\n            features[f'{side}_area'] = float(region.area)\n            features[f'{side}_circularity'] = (4 * np.pi * region.area) / (region.perimeter ** 2 + 1e-6)\n\n        features['area_ratio'] = float(regions_sorted[0].area / (regions_sorted[-1].area + 1e-6))\n    else:\n        for side in ['left', 'right']:\n            features[f'{side}_mean'] = features[f'{side}_std'] = 0.0\n            features[f'{side}_area'] = features[f'{side}_circularity'] = 0.0\n        features['area_ratio'] = 1.0\n\n    return features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:20:07.000288Z","iopub.execute_input":"2026-05-08T19:20:07.000865Z","iopub.status.idle":"2026-05-08T19:20:07.014564Z","shell.execute_reply.started":"2026-05-08T19:20:07.000839Z","shell.execute_reply":"2026-05-08T19:20:07.013848Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Dataset Builder Pipeline","metadata":{}},{"cell_type":"code","source":"def build_feature_dataset(df_split, max_samples=None):\n    \"\"\"\n    Applies preprocessing, Chan-Vese segmentation, and feature extraction \n    to a dataset split and returns a tabular DataFrame.\n    \"\"\"\n    data = []\n    \n    if max_samples:\n        df_split = df_split.head(max_samples)\n        \n    for _, row in tqdm(df_split.iterrows(), total=len(df_split), desc=\"Extracting Features\"):\n        pid = row['patientId']\n        target = row['Target']\n        \n        try:\n            # 1. Preprocess\n            dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n            img = dcm.pixel_array.astype(np.float32)\n            img = img - img.min()\n            if img.max() > 0:\n                img = img / img.max()\n            img = enhance_contrast(img)\n            img = cv2.resize(img, (512, 512), interpolation=cv2.INTER_AREA)\n            \n            # 2. Segment (Chan-Vese)\n            mask = segment_lungs(img)\n            \n            # 3. Extract\n            feats = extract_features(img, mask)\n            \n            # Add identifiers\n            feats['patientId'] = pid\n            feats['Target'] = target\n            \n            data.append(feats)\n        except Exception as e:\n            # Skip corrupted images quietly\n            continue\n            \n    return pd.DataFrame(data)\n\nprint('✅ Pipeline function defined.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:20:07.015786Z","iopub.execute_input":"2026-05-08T19:20:07.016221Z","iopub.status.idle":"2026-05-08T19:20:07.030075Z","shell.execute_reply.started":"2026-05-08T19:20:07.016197Z","shell.execute_reply":"2026-05-08T19:20:07.029199Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Run Extraction & Save","metadata":{}},{"cell_type":"code","source":"print(\"\\nExtracting Train Features...\")\ntrain_features = build_feature_dataset(train_df)\ntrain_features.to_csv(f'{OUTPUT_DIR}/train_features.csv', index=False)\n\nprint(\"\\nExtracting Val Features...\")\nval_features = build_feature_dataset(val_df)\nval_features.to_csv(f'{OUTPUT_DIR}/val_features.csv', index=False)\n\nprint(\"\\nExtracting Test Features...\")\ntest_features = build_feature_dataset(test_df)\ntest_features.to_csv(f'{OUTPUT_DIR}/test_features.csv', index=False)\n\nprint(\"\\n✅ All features extracted and saved to CSV!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:20:07.030974Z","iopub.execute_input":"2026-05-08T19:20:07.031251Z","iopub.status.idle":"2026-05-08T19:44:31.767225Z","shell.execute_reply.started":"2026-05-08T19:20:07.031229Z","shell.execute_reply":"2026-05-08T19:44:31.766313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings('ignore')\nfrom sklearn.feature_selection import SelectKBest, mutual_info_classif\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.feature_selection import SelectKBest, mutual_info_classif\nfrom sklearn.model_selection import StratifiedKFold, cross_val_score\n \nfrom sklearn.svm import SVC\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.ensemble import RandomForestClassifier\nfrom xgboost import XGBClassifier\n \nfrom sklearn.metrics import (accuracy_score, precision_score, recall_score,\n                             f1_score, roc_auc_score, classification_report,\n                             confusion_matrix, roc_curve)\n \nfrom imblearn.over_sampling import SMOTE","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:44:31.768400Z","iopub.execute_input":"2026-05-08T19:44:31.769019Z","iopub.status.idle":"2026-05-08T19:44:32.113327Z","shell.execute_reply.started":"2026-05-08T19:44:31.768991Z","shell.execute_reply":"2026-05-08T19:44:32.112687Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load Pre-split Data ","metadata":{}},{"cell_type":"code","source":"BASE = \"/kaggle/working\"\ntrain_df = pd.read_csv(f\"{BASE}/train_features.csv\")\nval_df   = pd.read_csv(f\"{BASE}/val_features.csv\")\ntest_df  = pd.read_csv(f\"{BASE}/test_features.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:44:32.114220Z","iopub.execute_input":"2026-05-08T19:44:32.114748Z","iopub.status.idle":"2026-05-08T19:44:32.325293Z","shell.execute_reply.started":"2026-05-08T19:44:32.114723Z","shell.execute_reply":"2026-05-08T19:44:32.324714Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Train: {train_df.shape} | Val: {val_df.shape} | Test: {test_df.shape}\")\nprint(f\"Train class dist: {train_df.Target.value_counts().to_dict()}\")\nprint(f\"Val   class dist: {val_df.Target.value_counts().to_dict()}\")\nprint(f\"Test  class dist: {test_df.Target.value_counts().to_dict()}\")\n \nDROP_COLS    = ['patientId']\nTARGET_COL   = 'Target'\nFEATURE_COLS = [c for c in train_df.columns if c not in DROP_COLS + [TARGET_COL]]\n \nX_train, y_train = train_df[FEATURE_COLS].values, train_df[TARGET_COL].values\nX_val,   y_val   = val_df[FEATURE_COLS].values,   val_df[TARGET_COL].values\nX_test,  y_test  = test_df[FEATURE_COLS].values,  test_df[TARGET_COL].values\n \nprint(f\"\\nFeatures ({len(FEATURE_COLS)}): {FEATURE_COLS}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:44:32.326177Z","iopub.execute_input":"2026-05-08T19:44:32.326450Z","iopub.status.idle":"2026-05-08T19:44:32.338180Z","shell.execute_reply.started":"2026-05-08T19:44:32.326426Z","shell.execute_reply":"2026-05-08T19:44:32.337251Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Step 6: Feature Selection**","metadata":{}},{"cell_type":"code","source":"k = min(20, len(FEATURE_COLS))\nselector = SelectKBest(score_func=mutual_info_classif, k=k)\nselector.fit(X_train, y_train)\n \nselected_features = [FEATURE_COLS[i] for i, m in enumerate(selector.get_support()) if m]\nprint(f\"\\nTop-{k} selected features: {selected_features}\")\n \nX_train_sel = selector.transform(X_train)\nX_val_sel   = selector.transform(X_val)\nX_test_sel  = selector.transform(X_test)\n \n# Feature importance plot\nplt.figure(figsize=(10, 4))\nplt.bar(FEATURE_COLS, selector.scores_)\nplt.xticks(rotation=45, ha='right')\nplt.title('Mutual Information Scores')\nplt.tight_layout()\nplt.savefig('feature_importance.png', dpi=150)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:44:32.339293Z","iopub.execute_input":"2026-05-08T19:44:32.339618Z","iopub.status.idle":"2026-05-08T19:44:35.206320Z","shell.execute_reply.started":"2026-05-08T19:44:32.339596Z","shell.execute_reply":"2026-05-08T19:44:35.205689Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# SCALING","metadata":{}},{"cell_type":"code","source":"scaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train_sel)\nX_val_scaled   = scaler.transform(X_val_sel)\nX_test_scaled  = scaler.transform(X_test_sel)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:44:35.207148Z","iopub.execute_input":"2026-05-08T19:44:35.207457Z","iopub.status.idle":"2026-05-08T19:44:35.217360Z","shell.execute_reply.started":"2026-05-08T19:44:35.207433Z","shell.execute_reply":"2026-05-08T19:44:35.216459Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PCA","metadata":{}},{"cell_type":"code","source":"X_train_pca = X_train_scaled\nX_val_pca   = X_val_scaled\nX_test_pca  = X_test_scaled\nprint(f\"Skipping PCA — using all {X_train_scaled.shape[1]} scaled features directly\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:44:35.218507Z","iopub.execute_input":"2026-05-08T19:44:35.219286Z","iopub.status.idle":"2026-05-08T19:44:35.223911Z","shell.execute_reply.started":"2026-05-08T19:44:35.219252Z","shell.execute_reply":"2026-05-08T19:44:35.223131Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# SMOTE","metadata":{}},{"cell_type":"code","source":"sm = SMOTE(random_state=42)\nX_train_res, y_train_res = sm.fit_resample(X_train_pca, y_train)\nprint(f\"\\nAfter SMOTE — shape: {X_train_res.shape}, dist: {np.bincount(y_train_res)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:44:35.224880Z","iopub.execute_input":"2026-05-08T19:44:35.225232Z","iopub.status.idle":"2026-05-08T19:44:35.383847Z","shell.execute_reply.started":"2026-05-08T19:44:35.225204Z","shell.execute_reply":"2026-05-08T19:44:35.383101Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Models","metadata":{}},{"cell_type":"markdown","source":"\n\n","metadata":{},"attachments":{}},{"cell_type":"markdown","source":"# TRAIN","metadata":{}},{"cell_type":"code","source":"models = {\n    'SVM':           SVC(\n    C=10,\n    gamma='scale',\n    kernel='rbf',\n    probability=True,\n    class_weight='balanced'\n),\n    'Logistic Reg':  LogisticRegression(max_iter=1000, random_state=42),\n    'Naive Bayes':   GaussianNB(),\n    'Random Forest': RandomForestClassifier(\n    n_estimators=300,\n    max_depth=15,\n    min_samples_split=5,\n    class_weight='balanced'\n),\n    'XGBoost':       XGBClassifier(\n    n_estimators=300,\n    max_depth=6,\n    learning_rate=0.03,\n    subsample=0.8,\n    colsample_bytree=0.8,\n    eval_metric='logloss'\n),\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:44:35.384785Z","iopub.execute_input":"2026-05-08T19:44:35.385027Z","iopub.status.idle":"2026-05-08T19:44:35.390100Z","shell.execute_reply.started":"2026-05-08T19:44:35.385004Z","shell.execute_reply":"2026-05-08T19:44:35.389327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results = {}\n \nprint(\"\\n\" + \"=\"*75)\nprint(f\"{'Model':<18} {'Val-F1':>7} {'Val-AUC':>8} {'Test-Acc':>9} {'Test-Rec':>9} {'Test-F1':>8} {'Test-AUC':>9}\")\nprint(\"=\"*75)\n \nfor name, clf in models.items():\n    # Train\n    clf.fit(X_train_res, y_train_res)\n \n    # Validation\n    y_val_pred  = clf.predict(X_val_pca)\n    y_val_proba = clf.predict_proba(X_val_pca)[:, 1]\n    val_f1  = f1_score(y_val, y_val_pred)\n    val_auc = roc_auc_score(y_val, y_val_proba)\n \n    # Test\n    y_pred  = clf.predict(X_test_pca)\n    y_proba = clf.predict_proba(X_test_pca)[:, 1]\n    acc  = accuracy_score(y_test, y_pred)\n    prec = precision_score(y_test, y_pred)\n    rec  = recall_score(y_test, y_pred)\n    f1   = f1_score(y_test, y_pred)\n    auc  = roc_auc_score(y_test, y_proba)\n \n    results[name] = dict(acc=acc, precision=prec, recall=rec, f1=f1, auc=auc,\n                         val_f1=val_f1, val_auc=val_auc,\n                         y_pred=y_pred, y_proba=y_proba)\n \n    print(f\"{name:<18} {val_f1:7.3f} {val_auc:8.3f} {acc:9.3f} {rec:9.3f} {f1:8.3f} {auc:9.3f}\")\n \nprint(\"=\"*75)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:44:35.391037Z","iopub.execute_input":"2026-05-08T19:44:35.391317Z","iopub.status.idle":"2026-05-08T19:48:56.269625Z","shell.execute_reply.started":"2026-05-08T19:44:35.391295Z","shell.execute_reply":"2026-05-08T19:48:56.268743Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# DETAILED TEST REPORTS","metadata":{}},{"cell_type":"code","source":"for name, res in results.items():\n    print(f\"\\n── {name} ──\")\n    print(classification_report(y_test, res['y_pred'],\n                                 target_names=['Normal', 'Pneumonia']))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:48:56.270850Z","iopub.execute_input":"2026-05-08T19:48:56.271172Z","iopub.status.idle":"2026-05-08T19:48:56.309548Z","shell.execute_reply.started":"2026-05-08T19:48:56.271149Z","shell.execute_reply":"2026-05-08T19:48:56.308805Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# CONFUSION MATRICES","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 3, figsize=(15, 9))\naxes = axes.flatten()\nfor ax, (name, res) in zip(axes, results.items()):\n    cm = confusion_matrix(y_test, res['y_pred'])\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', ax=ax,\n                xticklabels=['Normal', 'Pneumonia'],\n                yticklabels=['Normal', 'Pneumonia'])\n    ax.set_title(name); ax.set_xlabel('Predicted'); ax.set_ylabel('Actual')\naxes[-1].axis('off')\nplt.suptitle('Confusion Matrices (Test Set)', fontsize=14)\nplt.tight_layout(); plt.savefig('confusion_matrices.png', dpi=150); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:48:56.310472Z","iopub.execute_input":"2026-05-08T19:48:56.310786Z","iopub.status.idle":"2026-05-08T19:48:57.957356Z","shell.execute_reply.started":"2026-05-08T19:48:56.310765Z","shell.execute_reply":"2026-05-08T19:48:57.956534Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ROC CURVES","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nfor name, res in results.items():\n    fpr, tpr, _ = roc_curve(y_test, res['y_proba'])\n    plt.plot(fpr, tpr, label=f\"{name} (AUC={res['auc']:.3f})\")\nplt.plot([0,1],[0,1],'k--'); plt.xlabel('FPR'); plt.ylabel('TPR')\nplt.title('ROC Curves – Test Set'); plt.legend(loc='lower right')\nplt.tight_layout(); plt.savefig('roc_curves.png', dpi=150); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:48:57.958415Z","iopub.execute_input":"2026-05-08T19:48:57.958740Z","iopub.status.idle":"2026-05-08T19:48:58.332606Z","shell.execute_reply.started":"2026-05-08T19:48:57.958708Z","shell.execute_reply":"2026-05-08T19:48:58.332002Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# SUMMARY BAR CHART","metadata":{}},{"cell_type":"code","source":"metrics = ['acc', 'precision', 'recall', 'f1', 'auc']\nsummary = pd.DataFrame({name: [res[m] for m in metrics]\n                        for name, res in results.items()}, index=metrics)\nsummary.T.plot(kind='bar', figsize=(12, 5), ylim=(0, 1.05))\nplt.title('Model Comparison (Test Set)'); plt.ylabel('Score')\nplt.xticks(rotation=20); plt.legend(loc='lower right')\nplt.tight_layout(); plt.savefig('model_comparison.png', dpi=150); plt.show()\n \nprint(\"\\n── Best Models ──\")\nprint(\"Best F1:     \", max(results, key=lambda n: results[n]['f1']))\nprint(\"Best Recall: \", max(results, key=lambda n: results[n]['recall']))\nprint(\"Best AUC:    \", max(results, key=lambda n: results[n]['auc']))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T19:48:58.333541Z","iopub.execute_input":"2026-05-08T19:48:58.334117Z","iopub.status.idle":"2026-05-08T19:48:58.690001Z","shell.execute_reply.started":"2026-05-08T19:48:58.334091Z","shell.execute_reply":"2026-05-08T19:48:58.689165Z"}},"outputs":[],"execution_count":null}]}