{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Step 1: Setup & Data Acquisition","metadata":{}},{"cell_type":"markdown","source":"## Install Needed Libraries","metadata":{}},{"cell_type":"code","source":"!pip install -q pydicom scikit-image scikit-learn \\\n               xgboost imbalanced-learn albumentations","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:24.516091Z","iopub.execute_input":"2026-05-08T16:16:24.516908Z","iopub.status.idle":"2026-05-08T16:16:28.721568Z","shell.execute_reply.started":"2026-05-08T16:16:24.516875Z","shell.execute_reply":"2026-05-08T16:16:28.720766Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Set Path","metadata":{}},{"cell_type":"code","source":"import os\n\nDATA_DIR  = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\nTRAIN_DIR = f'{DATA_DIR}/stage_2_train_images'\nOUTPUT_DIR = '/kaggle/working'\n\nprint('📁 Dataset contents:')\nfor item in sorted(os.listdir(DATA_DIR)):\n    print(f'   {item}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:28.723607Z","iopub.execute_input":"2026-05-08T16:16:28.724098Z","iopub.status.idle":"2026-05-08T16:16:28.730245Z","shell.execute_reply.started":"2026-05-08T16:16:28.724047Z","shell.execute_reply":"2026-05-08T16:16:28.729468Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load and Preview Data","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\ndf = pd.read_csv(f'{DATA_DIR}/stage_2_train_labels.csv')\nprint('Shape:', df.shape)\nprint(df.head(8))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:28.731150Z","iopub.execute_input":"2026-05-08T16:16:28.731448Z","iopub.status.idle":"2026-05-08T16:16:29.064983Z","shell.execute_reply.started":"2026-05-08T16:16:28.731415Z","shell.execute_reply":"2026-05-08T16:16:29.064333Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Class Distribution","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\npatients = df.groupby('patientId')['Target'].max().reset_index()\ncounts   = patients['Target'].value_counts()\n\nprint(f'Normal    (0): {counts[0]:,} ({counts[0]/len(patients)*100:.1f}%)')\nprint(f'Pneumonia (1): {counts[1]:,} ({counts[1]/len(patients)*100:.1f}%)')\nprint(f'Total        : {len(patients):,}')\n\nfig, ax = plt.subplots(figsize=(5,3))\nax.bar(['Normal','Pneumonia'], [counts[0], counts[1]],\n       color=['steelblue','tomato'], edgecolor='white', width=0.5)\nax.set_title('Class Distribution — RSNA Dataset')\nax.set_ylabel('Patients')\nfor i, v in enumerate([counts[0], counts[1]]):\n    ax.text(i, v+50, f'{v:,}', ha='center', fontweight='bold')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:29.065940Z","iopub.execute_input":"2026-05-08T16:16:29.066251Z","iopub.status.idle":"2026-05-08T16:16:29.262550Z","shell.execute_reply.started":"2026-05-08T16:16:29.066227Z","shell.execute_reply":"2026-05-08T16:16:29.262000Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualize Sample X-Rays with Bounding Boxes","metadata":{}},{"cell_type":"code","source":"import pydicom\nimport numpy as np\nimport matplotlib.patches as patches\n\ndef apply_window(img, level=-500, width=1500):\n    low  = level - width // 2\n    high = level + width // 2\n    img  = np.clip(img, low, high)\n    img  = (img - low) / (high - low)\n    return img\n\nfig, axes = plt.subplots(2, 4, figsize=(16, 8))\naxes = axes.flatten()\n\nnormal_ids    = patients[patients['Target']==0]['patientId'].sample(4, random_state=42).values\npneumonia_ids = patients[patients['Target']==1]['patientId'].sample(4, random_state=42).values\nsample_ids    = list(normal_ids) + list(pneumonia_ids)\nlabels_list   = ['Normal']*4 + ['Pneumonia']*4\n\nfor idx, (pid, label) in enumerate(zip(sample_ids, labels_list)):\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = apply_window(dcm.pixel_array.astype(np.float32))\n\n    axes[idx].imshow(img, cmap='gray')\n    axes[idx].set_title(label,\n                        color='tomato' if label=='Pneumonia' else 'steelblue',\n                        fontweight='bold')\n    axes[idx].axis('off')\n\n    if label == 'Pneumonia':\n        for _, row in df[df['patientId']==pid][['x','y','width','height']].dropna().iterrows():\n            axes[idx].add_patch(patches.Rectangle(\n                (row['x'], row['y']), row['width'], row['height'],\n                linewidth=2, edgecolor='red', facecolor='none'))\n\nplt.suptitle('Sample Chest X-Rays — red boxes = pneumonia regions', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:29.264173Z","iopub.execute_input":"2026-05-08T16:16:29.264722Z","iopub.status.idle":"2026-05-08T16:16:32.027166Z","shell.execute_reply.started":"2026-05-08T16:16:29.264698Z","shell.execute_reply":"2026-05-08T16:16:32.026304Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Save Global Config","metadata":{}},{"cell_type":"code","source":"import json\n\nDATA_DIR  = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\nTRAIN_DIR = f'{DATA_DIR}/stage_2_train_images'\nOUTPUT_DIR = '/kaggle/working'\n\nCONFIG = {\n    'data_dir'    : DATA_DIR,\n    'train_dcm'   : TRAIN_DIR,\n    'labels_csv'  : f'{DATA_DIR}/stage_2_train_labels.csv',\n    'output_dir'  : OUTPUT_DIR,\n    'img_size'    : 512,\n    'window_level': -500,\n    'window_width': 1500,\n    'train_ratio' : 0.70,\n    'val_ratio'   : 0.15,\n    'test_ratio'  : 0.15,\n    'random_seed' : 42,\n}\n\nprint('✅ Config ready.')\nprint(json.dumps(CONFIG, indent=2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:32.028152Z","iopub.execute_input":"2026-05-08T16:16:32.028418Z","iopub.status.idle":"2026-05-08T16:16:32.034002Z","shell.execute_reply.started":"2026-05-08T16:16:32.028393Z","shell.execute_reply":"2026-05-08T16:16:32.033103Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 2: EDA & Visualization","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport pydicom\nimport cv2\nimport os\nfrom tqdm import tqdm\n\nDATA_DIR  = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\nTRAIN_DIR = f'{DATA_DIR}/stage_2_train_images'\n\ndf       = pd.read_csv(f'{DATA_DIR}/stage_2_train_labels.csv')\ndf_class = pd.read_csv(f'{DATA_DIR}/stage_2_detailed_class_info.csv')\npatients = df.groupby('patientId')['Target'].max().reset_index()\n\nprint('Labels shape    :', df.shape)\nprint('Class info shape:', df_class.shape)\nprint(df_class['class'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:32.035023Z","iopub.execute_input":"2026-05-08T16:16:32.035314Z","iopub.status.idle":"2026-05-08T16:16:32.456869Z","shell.execute_reply.started":"2026-05-08T16:16:32.035292Z","shell.execute_reply":"2026-05-08T16:16:32.456165Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Missing Values","metadata":{}},{"cell_type":"code","source":"print('=== Labels CSV ===')\nprint(df.isnull().sum())\nprint(f'\\nTotal missing: {df.isnull().sum().sum()}')\n\nprint('\\n=== Class Info CSV ===')\nprint(df_class.isnull().sum())\nprint(f'\\nTotal missing: {df_class.isnull().sum().sum()}')\n\nprint('\\n=== Missing Bounding Boxes (Normal patients have NaN boxes) ===')\nprint(f'Rows with NaN boxes : {df[df[\"x\"].isnull()].shape[0]:,}')\nprint(f'Rows with boxes     : {df[df[\"x\"].notna()].shape[0]:,}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:32.457699Z","iopub.execute_input":"2026-05-08T16:16:32.458005Z","iopub.status.idle":"2026-05-08T16:16:32.476238Z","shell.execute_reply.started":"2026-05-08T16:16:32.457971Z","shell.execute_reply":"2026-05-08T16:16:32.475492Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Patient Age & Gender Distribution","metadata":{}},{"cell_type":"code","source":"detail_df = df_class.drop_duplicates('patientId').copy()\n\n# Check available columns\nprint('Available columns:', detail_df.columns.tolist())\nprint(detail_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:32.477089Z","iopub.execute_input":"2026-05-08T16:16:32.477515Z","iopub.status.idle":"2026-05-08T16:16:32.497415Z","shell.execute_reply.started":"2026-05-08T16:16:32.477491Z","shell.execute_reply":"2026-05-08T16:16:32.496623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\n\nrecords = []\nsample_pids = patients.sample(300, random_state=42)['patientId'].values\n\nfor pid in tqdm(sample_pids):\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    target = patients[patients['patientId']==pid]['Target'].values[0]\n    \n    age    = getattr(dcm, 'PatientAge',  None)\n    gender = getattr(dcm, 'PatientSex',  None)\n    \n    records.append({\n        'patientId': pid,\n        'age'      : age,\n        'gender'   : gender,\n        'target'   : target\n    })\n\ndf_demo = pd.DataFrame(records)\nprint('Sample demographics:')\nprint(df_demo.head(10))\nprint('\\nGender counts:')\nprint(df_demo['gender'].value_counts())\nprint('\\nAge sample:')\nprint(df_demo['age'].value_counts().head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:32.498483Z","iopub.execute_input":"2026-05-08T16:16:32.498912Z","iopub.status.idle":"2026-05-08T16:16:36.815463Z","shell.execute_reply.started":"2026-05-08T16:16:32.498880Z","shell.execute_reply":"2026-05-08T16:16:36.814812Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Plot Age & Gender","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize=(16, 4))\n\n# Gender distribution\ngender_counts = df_demo['gender'].value_counts()\naxes[0].bar(gender_counts.index, gender_counts.values,\n            color=['steelblue','tomato'], edgecolor='white', width=0.4)\naxes[0].set_title('Patient Gender Distribution')\naxes[0].set_ylabel('Count')\naxes[0].set_xlabel('Gender')\nfor i, v in enumerate(gender_counts.values):\n    axes[0].text(i, v+1, f'{v:,}', ha='center', fontweight='bold')\n\n# Age distribution by class\ndf_demo['age'] = pd.to_numeric(df_demo['age'], errors='coerce')\nnormal_ages    = df_demo[df_demo['target']==0]['age'].dropna()\npneumonia_ages = df_demo[df_demo['target']==1]['age'].dropna()\n\naxes[1].hist(normal_ages,    bins=20, alpha=0.7, color='steelblue', label='Normal')\naxes[1].hist(pneumonia_ages, bins=20, alpha=0.7, color='tomato',    label='Pneumonia')\naxes[1].set_title('Age Distribution by Class')\naxes[1].set_xlabel('Age (years)')\naxes[1].set_ylabel('Count')\naxes[1].legend()\n\n# Gender by class\ngender_class = df_demo.groupby(['gender','target']).size().unstack(fill_value=0)\ngender_class.plot(kind='bar', ax=axes[2],\n                  color=['steelblue','tomato'], edgecolor='white')\naxes[2].set_title('Gender by Class')\naxes[2].set_xlabel('Gender')\naxes[2].set_ylabel('Count')\naxes[2].legend(['Normal','Pneumonia'])\naxes[2].tick_params(axis='x', rotation=0)\n\nplt.suptitle('Patient Demographics — RSNA Dataset', fontsize=13)\nplt.tight_layout()\nplt.show()\n\nprint(f'Male   — Total: {gender_counts.get(\"M\",0):,}')\nprint(f'Female — Total: {gender_counts.get(\"F\",0):,}')\nprint(f'\\nNormal    — Mean age: {normal_ages.mean():.1f} yrs')\nprint(f'Pneumonia — Mean age: {pneumonia_ages.mean():.1f} yrs')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:36.816374Z","iopub.execute_input":"2026-05-08T16:16:36.816719Z","iopub.status.idle":"2026-05-08T16:16:37.296605Z","shell.execute_reply.started":"2026-05-08T16:16:36.816694Z","shell.execute_reply":"2026-05-08T16:16:37.295787Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Class Distribution","metadata":{}},{"cell_type":"code","source":"class_counts = df_class.drop_duplicates('patientId')['class'].value_counts()\n\nfig, axes = plt.subplots(1, 2, figsize=(12, 4))\n\n# Bar chart\naxes[0].bar(class_counts.index, class_counts.values,\n            color=['steelblue', 'tomato', 'orange'], edgecolor='white')\naxes[0].set_title('Detailed Class Distribution')\naxes[0].set_ylabel('Number of Patients')\naxes[0].tick_params(axis='x', rotation=15)\nfor i, v in enumerate(class_counts.values):\n    axes[0].text(i, v+50, f'{v:,}', ha='center', fontweight='bold')\n\n# Pie chart\naxes[1].pie(class_counts.values, labels=class_counts.index,\n            colors=['steelblue', 'tomato', 'orange'],\n            autopct='%1.1f%%', startangle=90)\naxes[1].set_title('Class Proportions')\n\nplt.suptitle('RSNA Dataset — Class Distribution', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:37.297649Z","iopub.execute_input":"2026-05-08T16:16:37.297969Z","iopub.status.idle":"2026-05-08T16:16:37.494388Z","shell.execute_reply.started":"2026-05-08T16:16:37.297930Z","shell.execute_reply":"2026-05-08T16:16:37.493846Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Bounding Box Statistics","metadata":{}},{"cell_type":"code","source":"df_pos = df[df['Target'] == 1].copy()\n\nprint('Total bounding boxes      :', len(df_pos))\nprint('Patients with pneumonia   :', df_pos['patientId'].nunique())\nprint('Avg boxes per patient     :', round(len(df_pos)/df_pos['patientId'].nunique(), 2))\nprint()\nprint('Bounding Box Size Stats:')\nprint(df_pos[['width','height']].describe().round(1))\n\n# Box count per patient\nbox_counts = df_pos.groupby('patientId').size()\n\nfig, axes = plt.subplots(1, 3, figsize=(15, 4))\n\n# Box count distribution\naxes[0].hist(box_counts.values, bins=range(1, box_counts.max()+2),\n             color='tomato', edgecolor='white', align='left')\naxes[0].set_title('Number of Boxes per Pneumonia Patient')\naxes[0].set_xlabel('Number of Bounding Boxes')\naxes[0].set_ylabel('Count')\n\n# Box width distribution\naxes[1].hist(df_pos['width'].dropna(), bins=40, color='steelblue', edgecolor='white')\naxes[1].set_title('Bounding Box Width Distribution')\naxes[1].set_xlabel('Width (pixels)')\naxes[1].set_ylabel('Count')\n\n# Box height distribution\naxes[2].hist(df_pos['height'].dropna(), bins=40, color='purple', edgecolor='white')\naxes[2].set_title('Bounding Box Height Distribution')\naxes[2].set_xlabel('Height (pixels)')\naxes[2].set_ylabel('Count')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:37.495315Z","iopub.execute_input":"2026-05-08T16:16:37.495673Z","iopub.status.idle":"2026-05-08T16:16:37.950859Z","shell.execute_reply.started":"2026-05-08T16:16:37.495638Z","shell.execute_reply":"2026-05-08T16:16:37.949943Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Bounding Box Area Distribution","metadata":{}},{"cell_type":"code","source":"df_pos['area'] = df_pos['width'] * df_pos['height']\n\nfig, axes = plt.subplots(1, 2, figsize=(12, 4))\n\n# Area histogram\naxes[0].hist(df_pos['area'].dropna(), bins=40, color='steelblue', edgecolor='white')\naxes[0].set_title('Bounding Box Area Distribution')\naxes[0].set_xlabel('Area (pixels²)')\naxes[0].set_ylabel('Count')\naxes[0].axvline(df_pos['area'].mean(), color='red', linestyle='--',\n                label=f'Mean: {df_pos[\"area\"].mean():,.0f}')\naxes[0].legend()\n\n# Width vs Height scatter\naxes[1].scatter(df_pos['width'], df_pos['height'],\n                alpha=0.2, s=10, color='tomato')\naxes[1].set_title('Bounding Box Width vs Height')\naxes[1].set_xlabel('Width (pixels)')\naxes[1].set_ylabel('Height (pixels)')\n\nprint(f'Mean area  : {df_pos[\"area\"].mean():,.0f} px²')\nprint(f'Median area: {df_pos[\"area\"].median():,.0f} px²')\nprint(f'Min area   : {df_pos[\"area\"].min():,.0f} px²')\nprint(f'Max area   : {df_pos[\"area\"].max():,.0f} px²')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:37.953590Z","iopub.execute_input":"2026-05-08T16:16:37.953923Z","iopub.status.idle":"2026-05-08T16:16:38.270696Z","shell.execute_reply.started":"2026-05-08T16:16:37.953896Z","shell.execute_reply":"2026-05-08T16:16:38.269995Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Image Metadata Analysis","metadata":{}},{"cell_type":"code","source":"import pydicom\n\nsample = pydicom.dcmread(f'{TRAIN_DIR}/{patients[\"patientId\"].iloc[0]}.dcm')\nprint('Image size  :', sample.Rows, 'x', sample.Columns)\nprint('Bits        :', sample.BitsAllocated)\nprint('Modality    :', sample.Modality)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:38.271408Z","iopub.execute_input":"2026-05-08T16:16:38.271705Z","iopub.status.idle":"2026-05-08T16:16:38.290850Z","shell.execute_reply.started":"2026-05-08T16:16:38.271682Z","shell.execute_reply":"2026-05-08T16:16:38.290063Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Pixel Intensity Analysis","metadata":{}},{"cell_type":"code","source":"print('Analyzing pixel intensities from 100 samples (50 normal, 50 pneumonia)...')\n\ndef apply_window(img, level=-500, width=1500):\n    low  = level - width // 2\n    high = level + width // 2\n    img  = np.clip(img, low, high)\n    img  = (img - low) / (high - low)\n    return img\n\nnormal_pids    = patients[patients['Target']==0]['patientId'].sample(50, random_state=42).values\npneumonia_pids = patients[patients['Target']==1]['patientId'].sample(50, random_state=42).values\n\nnormal_means, pneumonia_means = [], []\nnormal_stds,  pneumonia_stds  = [], []\n\nfor pid in tqdm(normal_pids):\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = apply_window(dcm.pixel_array.astype(np.float32))\n    normal_means.append(img.mean())\n    normal_stds.append(img.std())\n\nfor pid in tqdm(pneumonia_pids):\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = apply_window(dcm.pixel_array.astype(np.float32))\n    pneumonia_means.append(img.mean())\n    pneumonia_stds.append(img.std())\n\nfig, axes = plt.subplots(1, 2, figsize=(12, 4))\n\n# Mean intensity\naxes[0].hist(normal_means,    bins=20, alpha=0.7, color='steelblue', label='Normal')\naxes[0].hist(pneumonia_means, bins=20, alpha=0.7, color='tomato',    label='Pneumonia')\naxes[0].set_title('Mean Pixel Intensity Distribution')\naxes[0].set_xlabel('Mean Intensity')\naxes[0].set_ylabel('Count')\naxes[0].legend()\n\n# Std intensity\naxes[1].hist(normal_stds,    bins=20, alpha=0.7, color='steelblue', label='Normal')\naxes[1].hist(pneumonia_stds, bins=20, alpha=0.7, color='tomato',    label='Pneumonia')\naxes[1].set_title('Pixel Intensity Std Distribution')\naxes[1].set_xlabel('Std of Intensity')\naxes[1].set_ylabel('Count')\naxes[1].legend()\n\nplt.tight_layout()\nplt.show()\n\nprint(f'\\nNormal    — Mean: {np.mean(normal_means):.3f}, Std: {np.mean(normal_stds):.3f}')\nprint(f'Pneumonia — Mean: {np.mean(pneumonia_means):.3f}, Std: {np.mean(pneumonia_stds):.3f}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:38.291776Z","iopub.execute_input":"2026-05-08T16:16:38.292062Z","iopub.status.idle":"2026-05-08T16:16:40.521718Z","shell.execute_reply.started":"2026-05-08T16:16:38.292032Z","shell.execute_reply":"2026-05-08T16:16:40.521030Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train / Val / Test","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nSEED = 42\n\n# First split: train vs temp (val+test)\ntrain_df, temp_df = train_test_split(patients, test_size=0.30,\n                                     stratify=patients['Target'],\n                                     random_state=SEED)\n\n# Second split: val vs test\nval_df, test_df = train_test_split(temp_df, test_size=0.50,\n                                   stratify=temp_df['Target'],\n                                   random_state=SEED)\n\nprint(f'Train : {len(train_df):,} ({len(train_df)/len(patients)*100:.1f}%)')\nprint(f'Val   : {len(val_df):,}  ({len(val_df)/len(patients)*100:.1f}%)')\nprint(f'Test  : {len(test_df):,}  ({len(test_df)/len(patients)*100:.1f}%)')\n\n# Verify stratification\nfor name, split in [('Train', train_df), ('Val', val_df), ('Test', test_df)]:\n    pos = split['Target'].sum()\n    print(f'{name} — Pneumonia: {pos} ({pos/len(split)*100:.1f}%)')\n\n# Save splits\ntrain_df.to_csv('/kaggle/working/train_split.csv', index=False)\nval_df.to_csv('/kaggle/working/val_split.csv',   index=False)\ntest_df.to_csv('/kaggle/working/test_split.csv',  index=False)\nprint('\\n✅ Splits saved to /kaggle/working/')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:40.522717Z","iopub.execute_input":"2026-05-08T16:16:40.523050Z","iopub.status.idle":"2026-05-08T16:16:41.286834Z","shell.execute_reply.started":"2026-05-08T16:16:40.523026Z","shell.execute_reply":"2026-05-08T16:16:41.286073Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 3: Preprocessing","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport pydicom\nimport cv2\nimport os\nfrom tqdm import tqdm\n\nDATA_DIR  = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\nTRAIN_DIR = f'{DATA_DIR}/stage_2_train_images'\nOUTPUT_DIR = '/kaggle/working'\n\ndf       = pd.read_csv(f'{DATA_DIR}/stage_2_train_labels.csv')\npatients = df.groupby('patientId')['Target'].max().reset_index()\n\ntrain_df = pd.read_csv(f'{OUTPUT_DIR}/train_split.csv')\nval_df   = pd.read_csv(f'{OUTPUT_DIR}/val_split.csv')\ntest_df  = pd.read_csv(f'{OUTPUT_DIR}/test_split.csv')\n\nprint('✅ Imports done.')\nprint(f'Train: {len(train_df):,} | Val: {len(val_df):,} | Test: {len(test_df):,}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:41.287855Z","iopub.execute_input":"2026-05-08T16:16:41.288173Z","iopub.status.idle":"2026-05-08T16:16:41.359519Z","shell.execute_reply.started":"2026-05-08T16:16:41.288150Z","shell.execute_reply":"2026-05-08T16:16:41.358964Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Functions","metadata":{}},{"cell_type":"code","source":"def load_dicom(pid):\n    \"\"\"Load raw DICOM pixel array.\"\"\"\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = dcm.pixel_array.astype(np.float32)\n    return img\n\ndef apply_window(img, level=-500, width=1500):\n    \"\"\"Apply lung windowing to enhance pulmonary structures.\"\"\"\n    low  = level - width // 2\n    high = level + width // 2\n    img  = np.clip(img, low, high)\n    img  = (img - low) / (high - low)\n    return img\n\ndef normalize(img):\n    \"\"\"Normalize image to [0, 1].\"\"\"\n    img = img - img.min()\n    if img.max() > 0:\n        img = img / img.max()\n    return img\n\ndef denoise(img):\n    \"\"\"Apply Gaussian blur for noise reduction.\"\"\"\n    return cv2.GaussianBlur(img, (3, 3), 0)\n    \ndef resize_img(img, size=512):\n    \"\"\"Resize image to fixed size.\"\"\"\n    return cv2.resize(img, (size, size), interpolation=cv2.INTER_AREA)\n\ndef preprocess(pid, size=512):\n    \"\"\"Full preprocessing pipeline for one patient.\"\"\"\n    img = load_dicom(pid)       # Step 1: Load DICOM\n    img = normalize(img)        # Step 2: Normalize to [0,1] first\n    img = denoise(img)          # Step 3: Noise reduction\n    img = resize_img(img, size) # Step 4: Resize to 512x512\n    return img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:41.360347Z","iopub.execute_input":"2026-05-08T16:16:41.360678Z","iopub.status.idle":"2026-05-08T16:16:41.367807Z","shell.execute_reply.started":"2026-05-08T16:16:41.360654Z","shell.execute_reply":"2026-05-08T16:16:41.367079Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Test Preprocessing on Single Image","metadata":{}},{"cell_type":"code","source":"pid = patients['patientId'].iloc[0]\n\nraw     = load_dicom(pid)\nnormed  = normalize(raw)\ndenoised = denoise(normed)\nresized  = resize_img(denoised)\n\nprint(f'Raw      — shape: {raw.shape},     min: {raw.min():.1f},   max: {raw.max():.1f}')\nprint(f'Normed   — shape: {normed.shape},   min: {normed.min():.2f},  max: {normed.max():.2f}')\nprint(f'Denoised — shape: {denoised.shape}, min: {denoised.min():.2f},  max: {denoised.max():.2f}')\nprint(f'Resized  — shape: {resized.shape},  min: {resized.min():.2f},  max: {resized.max():.2f}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:41.368680Z","iopub.execute_input":"2026-05-08T16:16:41.368954Z","iopub.status.idle":"2026-05-08T16:16:41.419804Z","shell.execute_reply.started":"2026-05-08T16:16:41.368920Z","shell.execute_reply":"2026-05-08T16:16:41.418823Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Augmentation","metadata":{}},{"cell_type":"code","source":"import albumentations as A\n\naugment = A.Compose([\n    A.HorizontalFlip(p=0.5),\n    A.Rotate(limit=10, p=0.3),\n    A.Affine(translate_percent=0.1, p=0.3),\n    A.RandomBrightnessContrast(brightness_limit=0.2,\n                               contrast_limit=0.2, p=0.4),\n])\n\ndef augment_img(img):\n    \"\"\"Apply augmentation to a single image.\"\"\"\n    img_uint8 = (img * 255).astype(np.uint8)\n    augmented = augment(image=img_uint8)['image']\n    return augmented.astype(np.float32) / 255.0\n\nprint('✅ Augmentation pipeline defined.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:41.420935Z","iopub.execute_input":"2026-05-08T16:16:41.421240Z","iopub.status.idle":"2026-05-08T16:16:47.275708Z","shell.execute_reply.started":"2026-05-08T16:16:41.421205Z","shell.execute_reply":"2026-05-08T16:16:47.274834Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualize Augmentation Effects","metadata":{}},{"cell_type":"code","source":"sample_pid = patients[patients['Target']==1]['patientId'].sample(1, random_state=7).values[0]\nimg = preprocess(sample_pid)\n\nfig, axes = plt.subplots(2, 4, figsize=(16, 8))\naxes = axes.flatten()\n\n# Original\naxes[0].imshow(img, cmap='gray')\naxes[0].set_title('Original (preprocessed)')\naxes[0].axis('off')\n\n# 7 augmented versions\nfor i in range(1, 8):\n    aug = augment_img(img)\n    axes[i].imshow(aug, cmap='gray')\n    axes[i].set_title(f'Augmented #{i}')\n    axes[i].axis('off')\n\nplt.suptitle('Data Augmentation Examples', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:47.276644Z","iopub.execute_input":"2026-05-08T16:16:47.277077Z","iopub.status.idle":"2026-05-08T16:16:48.322025Z","shell.execute_reply.started":"2026-05-08T16:16:47.277051Z","shell.execute_reply":"2026-05-08T16:16:48.321268Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Before VS After Preprocessing","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 4, figsize=(16, 12))\n\nsample_pids = patients.sample(4, random_state=10)['patientId'].values\n\nfor idx, pid in enumerate(sample_pids):\n    raw      = load_dicom(pid)\n    windowed = apply_window(raw)\n    resized  = resize_img(normalize(windowed))\n\n    # Row 1: Raw\n    axes[0][idx].imshow(raw, cmap='gray')\n    axes[0][idx].set_title(f'Raw DICOM\\n{raw.shape}')\n    axes[0][idx].axis('off')\n\n    # Row 2: After windowing\n    axes[1][idx].imshow(windowed, cmap='gray')\n    axes[1][idx].set_title(f'After Windowing\\n[{windowed.min():.2f}, {windowed.max():.2f}]')\n    axes[1][idx].axis('off')\n\n    # Row 3: After resize\n    axes[2][idx].imshow(resized, cmap='gray')\n    axes[2][idx].set_title(f'Final (512x512)\\n[{resized.min():.2f}, {resized.max():.2f}]')\n    axes[2][idx].axis('off')\n\nplt.suptitle('Preprocessing Pipeline: Raw → Windowed → Final', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:48.323406Z","iopub.execute_input":"2026-05-08T16:16:48.323760Z","iopub.status.idle":"2026-05-08T16:16:50.903237Z","shell.execute_reply.started":"2026-05-08T16:16:48.323727Z","shell.execute_reply":"2026-05-08T16:16:50.902439Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Histogram Update","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 4, figsize=(16, 12))\n\nsample_pids = patients.sample(4, random_state=10)['patientId'].values\n\nfor idx, pid in enumerate(sample_pids):\n    raw      = load_dicom(pid)\n    normed   = normalize(raw)\n    final    = resize_img(denoise(normed))\n\n    # Row 1: Raw histogram\n    axes[0][idx].hist(raw.flatten(), bins=50, color='gray', edgecolor='none')\n    axes[0][idx].set_title(f'Raw\\nmin={raw.min():.0f} max={raw.max():.0f}')\n    axes[0][idx].set_xlabel('Pixel Value')\n    axes[0][idx].set_ylabel('Count')\n\n    # Row 2: After normalization\n    axes[1][idx].hist(normed.flatten(), bins=50, color='steelblue', edgecolor='none')\n    axes[1][idx].set_title(f'After Normalize\\nmin={normed.min():.2f} max={normed.max():.2f}')\n    axes[1][idx].set_xlabel('Pixel Value')\n    axes[1][idx].set_ylabel('Count')\n\n    # Row 3: Final (denoised + resized)\n    axes[2][idx].hist(final.flatten(), bins=50, color='tomato', edgecolor='none')\n    axes[2][idx].set_title(f'Final (denoised+resized)\\nmin={final.min():.2f} max={final.max():.2f}')\n    axes[2][idx].set_xlabel('Pixel Value')\n    axes[2][idx].set_ylabel('Count')\n\nplt.suptitle('Preprocessing Pipeline: Raw → Normalized → Final', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:50.904536Z","iopub.execute_input":"2026-05-08T16:16:50.904881Z","iopub.status.idle":"2026-05-08T16:16:53.193211Z","shell.execute_reply.started":"2026-05-08T16:16:50.904856Z","shell.execute_reply":"2026-05-08T16:16:53.192379Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 4: Lung Segmentation","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport pydicom\nimport cv2\nfrom scipy import ndimage\nfrom tqdm import tqdm\nimport os\n\nDATA_DIR  = '/kaggle/input/competitions/rsna-pneumonia-detection-challenge'\nTRAIN_DIR = f'{DATA_DIR}/stage_2_train_images'\nOUTPUT_DIR = '/kaggle/working'\n\ndf       = pd.read_csv(f'{DATA_DIR}/stage_2_train_labels.csv')\npatients = df.groupby('patientId')['Target'].max().reset_index()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:53.194302Z","iopub.execute_input":"2026-05-08T16:16:53.194692Z","iopub.status.idle":"2026-05-08T16:16:53.246149Z","shell.execute_reply.started":"2026-05-08T16:16:53.194655Z","shell.execute_reply":"2026-05-08T16:16:53.245598Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Segmentation Function","metadata":{}},{"cell_type":"code","source":"def segment_lungs(img):\n    \"\"\"\n    Segment lung region from chest X-ray using\n    thresholding + morphological operations.\n    Returns: binary lung mask\n    \"\"\"\n    # Step 1: Convert to uint8\n    img_uint8 = (img * 255).astype(np.uint8)\n\n    # Step 2: Otsu thresholding — separates lung from background\n    _, thresh = cv2.threshold(img_uint8, 0, 255,\n                              cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n\n    # Step 3: Invert — lungs are dark in X-ray\n    thresh = cv2.bitwise_not(thresh)\n\n    # Step 4: Morphological opening — remove small noise\n    kernel  = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (5, 5))\n    opened  = cv2.morphologyEx(thresh, cv2.MORPH_OPEN, kernel)\n\n    # Step 5: Morphological closing — fill holes inside lungs\n    kernel2 = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (20, 20))\n    closed  = cv2.morphologyEx(opened, cv2.MORPH_CLOSE, kernel2)\n\n    # Step 6: Keep only the 2 largest components (left + right lung)\n    num_labels, labels, stats, _ = cv2.connectedComponentsWithStats(closed)\n    \n    # Sort by area (skip background = label 0)\n    areas = stats[1:, cv2.CC_STAT_AREA]\n    top2  = np.argsort(areas)[::-1][:2] + 1  # top 2 largest\n\n    mask = np.zeros_like(closed)\n    for label in top2:\n        mask[labels == label] = 255\n\n    return mask\n\ndef apply_mask(img, mask):\n    \"\"\"Apply lung mask to image — zero out background.\"\"\"\n    mask_norm = mask / 255.0\n    return img * mask_norm\n\nprint('✅ Segmentation functions defined.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:53.247224Z","iopub.execute_input":"2026-05-08T16:16:53.247718Z","iopub.status.idle":"2026-05-08T16:16:53.256467Z","shell.execute_reply.started":"2026-05-08T16:16:53.247679Z","shell.execute_reply":"2026-05-08T16:16:53.255473Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Test Segmentation on Sample Image","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 4, figsize=(16, 12))\n\nsample_pids = patients.sample(4, random_state=42)['patientId'].values\n\nfor idx, pid in enumerate(sample_pids):\n    # Preprocess\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = dcm.pixel_array.astype(np.float32)\n    img = img - img.min()\n    if img.max() > 0:\n        img = img / img.max()\n    img = cv2.GaussianBlur(img, (3,3), 0)\n    img = cv2.resize(img, (512, 512), interpolation=cv2.INTER_AREA)\n\n    # Segment\n    mask        = segment_lungs(img)\n    masked_img  = apply_mask(img, mask)\n\n    # Row 1: Original\n    axes[0][idx].imshow(img, cmap='gray')\n    axes[0][idx].set_title('Preprocessed')\n    axes[0][idx].axis('off')\n\n    # Row 2: Mask\n    axes[1][idx].imshow(mask, cmap='gray')\n    axes[1][idx].set_title('Lung Mask')\n    axes[1][idx].axis('off')\n\n    # Row 3: Masked image\n    axes[2][idx].imshow(masked_img, cmap='gray')\n    axes[2][idx].set_title('Segmented Lung')\n    axes[2][idx].axis('off')\n\nplt.suptitle('Lung Segmentation: Thresholding + Morphology', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:53.257748Z","iopub.execute_input":"2026-05-08T16:16:53.258165Z","iopub.status.idle":"2026-05-08T16:16:54.736071Z","shell.execute_reply.started":"2026-05-08T16:16:53.258136Z","shell.execute_reply":"2026-05-08T16:16:54.734828Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Test Segmentation on Pneumonia Cases","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 4, figsize=(16, 12))\n\npneumonia_pids = patients[patients['Target']==1]['patientId'].sample(4, random_state=42).values\n\nfor idx, pid in enumerate(pneumonia_pids):\n    dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n    img = dcm.pixel_array.astype(np.float32)\n    img = img - img.min()\n    if img.max() > 0:\n        img = img / img.max()\n    img = cv2.GaussianBlur(img, (3,3), 0)\n    img = cv2.resize(img, (512, 512), interpolation=cv2.INTER_AREA)\n\n    mask       = segment_lungs(img)\n    masked_img = apply_mask(img, mask)\n\n    axes[0][idx].imshow(img, cmap='gray')\n    axes[0][idx].set_title('Pneumonia — Original')\n    axes[0][idx].axis('off')\n\n    axes[1][idx].imshow(mask, cmap='gray')\n    axes[1][idx].set_title('Lung Mask')\n    axes[1][idx].axis('off')\n\n    axes[2][idx].imshow(masked_img, cmap='gray')\n    axes[2][idx].set_title('Segmented Lung')\n    axes[2][idx].axis('off')\n\nplt.suptitle('Lung Segmentation on Pneumonia Cases', fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:54.737364Z","iopub.execute_input":"2026-05-08T16:16:54.737795Z","iopub.status.idle":"2026-05-08T16:16:56.154347Z","shell.execute_reply.started":"2026-05-08T16:16:54.737761Z","shell.execute_reply":"2026-05-08T16:16:56.153465Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 5: Feature Extraction","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"from scipy.stats import skew, kurtosis\nfrom skimage.feature import graycomatrix, graycoprops\nfrom skimage.measure import regionprops, label\nimport pandas as pd\nfrom tqdm import tqdm\nimport pydicom\nimport cv2\nimport os","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:56.155432Z","iopub.execute_input":"2026-05-08T16:16:56.155765Z","iopub.status.idle":"2026-05-08T16:16:56.261147Z","shell.execute_reply.started":"2026-05-08T16:16:56.155730Z","shell.execute_reply":"2026-05-08T16:16:56.260612Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Extraction Logic","metadata":{}},{"cell_type":"code","source":"def extract_features(img, mask):\n    \"\"\"\n    Extracts Intensity, Texture (GLCM), and Shape features \n    from the segmented lung region.\n    \"\"\"\n    features = {}\n    \n    # Get only the pixels inside the lung mask\n    bool_mask = mask > 0\n    lung_pixels = img[bool_mask]\n    \n    # ---------------------------------\n    # 1. Intensity Features\n    # ---------------------------------\n    if len(lung_pixels) > 0:\n        features['mean']     = float(np.mean(lung_pixels))\n        features['std']      = float(np.std(lung_pixels))\n        features['skewness'] = float(skew(lung_pixels))\n        features['kurtosis'] = float(kurtosis(lung_pixels))\n    else:\n        features['mean'] = features['std'] = features['skewness'] = features['kurtosis'] = 0.0\n\n    # ---------------------------------\n    # 2. Texture Features (GLCM)\n    # ---------------------------------\n    # Convert image to uint8 [0-255] for GLCM calculations\n    img_uint8 = (img * 255).astype(np.uint8)\n    \n    labeled_mask = label(bool_mask)\n    regions = regionprops(labeled_mask)\n    \n    if regions:\n        # Crop the image to the lung bounds to prevent the black background from ruining the texture score\n        min_row = min([r.bbox[0] for r in regions])\n        min_col = min([r.bbox[1] for r in regions])\n        max_row = max([r.bbox[2] for r in regions])\n        max_col = max([r.bbox[3] for r in regions])\n        \n        cropped_img = img_uint8[min_row:max_row, min_col:max_col]\n        \n        # Calculate GLCM\n        glcm = graycomatrix(cropped_img, distances=[1], angles=[0], levels=256, symmetric=True, normed=True)\n        \n        features['contrast']    = float(graycoprops(glcm, 'contrast')[0, 0])\n        features['energy']      = float(graycoprops(glcm, 'energy')[0, 0])\n        features['homogeneity'] = float(graycoprops(glcm, 'homogeneity')[0, 0])\n        features['correlation'] = float(graycoprops(glcm, 'correlation')[0, 0])\n    else:\n        features['contrast'] = features['energy'] = features['homogeneity'] = features['correlation'] = 0.0\n\n    # ---------------------------------\n    # 3. Shape Features\n    # ---------------------------------\n    if regions:\n        total_area = sum([r.area for r in regions])\n        total_perimeter = sum([r.perimeter for r in regions])\n        \n        features['area']      = total_area\n        features['perimeter'] = total_perimeter\n        \n        # Circularity formula\n        if total_perimeter > 0:\n            features['circularity'] = (4 * np.pi * total_area) / (total_perimeter ** 2)\n        else:\n            features['circularity'] = 0.0\n    else:\n        features['area'] = features['perimeter'] = features['circularity'] = 0.0\n        \n    return features\n\nprint('✅ Extraction function defined.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:56.262071Z","iopub.execute_input":"2026-05-08T16:16:56.262538Z","iopub.status.idle":"2026-05-08T16:16:56.274220Z","shell.execute_reply.started":"2026-05-08T16:16:56.262515Z","shell.execute_reply":"2026-05-08T16:16:56.273415Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Dataset Builder Pipeline","metadata":{}},{"cell_type":"code","source":"def build_feature_dataset(df_split, max_samples=None):\n    \"\"\"\n    Applies preprocessing, Chan-Vese segmentation, and feature extraction \n    to a dataset split and returns a tabular DataFrame.\n    \"\"\"\n    data = []\n    \n    if max_samples:\n        df_split = df_split.head(max_samples)\n        \n    for _, row in tqdm(df_split.iterrows(), total=len(df_split), desc=\"Extracting Features\"):\n        pid = row['patientId']\n        target = row['Target']\n        \n        try:\n            # 1. Preprocess\n            dcm = pydicom.dcmread(f'{TRAIN_DIR}/{pid}.dcm')\n            img = dcm.pixel_array.astype(np.float32)\n            img = img - img.min()\n            if img.max() > 0:\n                img = img / img.max()\n            img = cv2.GaussianBlur(img, (3,3), 0)\n            img = cv2.resize(img, (512, 512), interpolation=cv2.INTER_AREA)\n            \n            # 2. Segment (Chan-Vese)\n            mask = segment_lungs(img)\n            \n            # 3. Extract\n            feats = extract_features(img, mask)\n            \n            # Add identifiers\n            feats['patientId'] = pid\n            feats['Target'] = target\n            \n            data.append(feats)\n        except Exception as e:\n            # Skip corrupted images quietly\n            continue\n            \n    return pd.DataFrame(data)\n\nprint('✅ Pipeline function defined.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:16:56.275301Z","iopub.execute_input":"2026-05-08T16:16:56.275701Z","iopub.status.idle":"2026-05-08T16:16:56.290426Z","shell.execute_reply.started":"2026-05-08T16:16:56.275657Z","shell.execute_reply":"2026-05-08T16:16:56.289629Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Run Extraction & Save","metadata":{}},{"cell_type":"code","source":"print(\"\\nExtracting Train Features...\")\ntrain_features = build_feature_dataset(train_df)\ntrain_features.to_csv(f'{OUTPUT_DIR}/train_features.csv', index=False)\n\nprint(\"\\nExtracting Val Features...\")\nval_features = build_feature_dataset(val_df)\nval_features.to_csv(f'{OUTPUT_DIR}/val_features.csv', index=False)\n\nprint(\"\\nExtracting Test Features...\")\ntest_features = build_feature_dataset(test_df)\ntest_features.to_csv(f'{OUTPUT_DIR}/test_features.csv', index=False)\n\nprint(\"\\n✅ All features extracted and saved to CSV!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:17:57.020407Z","iopub.execute_input":"2026-05-08T16:17:57.021533Z","iopub.status.idle":"2026-05-08T16:35:04.282231Z","shell.execute_reply.started":"2026-05-08T16:17:57.021497Z","shell.execute_reply":"2026-05-08T16:35:04.281624Z"}},"outputs":[],"execution_count":null}]}