{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"02fb6ba0","cell_type":"markdown","source":"# EfficientNetB0 + Random Forest — RSNA Pneumonia Detection Challenge\n\nThis notebook adapts the original pipeline to the Kaggle competition:\n\n`rsna-pneumonia-detection-challenge`\n\nIt computes:\n\n- **Accuracy**\n- **Precision**\n- **F1-score**\n\n## Task used in this notebook\nThe original RSNA competition is an **object detection** task with bounding boxes.\nTo stay consistent with the original **EfficientNetB0 + Random Forest** classification pipeline, this notebook converts it into a **binary image classification** task:\n\n- **PNEUMONIA**\n- **NON_PNEUMONIA**\n","metadata":{}},{"id":"9ef189ef","cell_type":"markdown","source":"## 1 · Find the RSNA dataset and build binary labels","metadata":{}},{"id":"8d3c4798","cell_type":"code","source":"import os\nimport pathlib\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\n\ndef find_rsna_root():\n    candidates = [\n        pathlib.Path('/kaggle/input/rsna-pneumonia-detection-challenge'),\n        pathlib.Path('/kaggle/input/rsna-pneumonia-detection-2'),\n        pathlib.Path('/kaggle/input/competitions/rsna-pneumonia-detection-challenge'),\n    ]\n\n    for p in candidates:\n        if p.exists():\n            return p\n\n    search_roots = [pathlib.Path('/kaggle/input'), pathlib.Path('/kaggle/input/competitions')]\n    matched = []\n    for root in search_roots:\n        if not root.exists():\n            continue\n        for p in [root] + [x for x in root.rglob('*') if x.is_dir()]:\n            try:\n                names = {x.name.lower() for x in p.iterdir()}\n            except Exception:\n                continue\n            if 'stage_2_train_images' in names and 'stage_2_train_labels.csv' in names:\n                matched.append(p)\n\n    if not matched:\n        raise FileNotFoundError(\n            'Could not locate the RSNA Pneumonia Detection Challenge dataset. '\n            'Attach the competition data in Kaggle, then rerun.'\n        )\n\n    matched = sorted(set(matched), key=lambda p: len(str(p)))\n    return matched[0]\n\nDATA_ROOT = find_rsna_root()\nIMAGES_DIR = DATA_ROOT / 'stage_2_train_images'\nLABELS_CSV = DATA_ROOT / 'stage_2_train_labels.csv'\n\nprint('DATA_ROOT :', DATA_ROOT)\nprint('IMAGES_DIR exists:', IMAGES_DIR.exists())\nprint('LABELS_CSV exists:', LABELS_CSV.exists())\n\nlabels_raw = pd.read_csv(LABELS_CSV)\nprint('\\nRaw labels preview:')\nprint(labels_raw.head())\n\nlabels_df = (\n    labels_raw.groupby('patientId', as_index=False)['Target']\n    .max()\n    .rename(columns={'Target': 'target'})\n)\nlabels_df['target_name'] = labels_df['target'].map({0: 'NON_PNEUMONIA', 1: 'PNEUMONIA'})\n\ntrain_df, test_df = train_test_split(\n    labels_df,\n    test_size=0.20,\n    random_state=42,\n    stratify=labels_df['target']\n)\n\ntrain_df = train_df.copy().reset_index(drop=True)\ntest_df = test_df.copy().reset_index(drop=True)\n\nprint('\\nTrain counts:')\nprint(train_df['target_name'].value_counts())\nprint('\\nTest counts:')\nprint(test_df['target_name'].value_counts())\n","metadata":{"execution":{"iopub.status.busy":"2026-03-29T19:35:08.442344Z","iopub.execute_input":"2026-03-29T19:35:08.443043Z","iopub.status.idle":"2026-03-29T19:35:13.004534Z","shell.execute_reply.started":"2026-03-29T19:35:08.442997Z","shell.execute_reply":"2026-03-29T19:35:13.003737Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"4eb53b26","cell_type":"markdown","source":"## 2 · Convert DICOM images to PNG and stage train/test folders","metadata":{}},{"id":"4d2887ae","cell_type":"code","source":"import shutil\nimport numpy as np\nimport pydicom\nfrom PIL import Image\nfrom tqdm.auto import tqdm\n\nSTAGED = pathlib.Path('/kaggle/working/staged_rsna_pneumonia')\nIMG_SIZE = (224, 224)\n\nif STAGED.exists():\n    shutil.rmtree(STAGED)\n\nfor split in ('train', 'test'):\n    for cls in ('NON_PNEUMONIA', 'PNEUMONIA'):\n        (STAGED / split / cls).mkdir(parents=True, exist_ok=True)\n\ndef dicom_to_uint8_rgb(dcm_path):\n    ds = pydicom.dcmread(str(dcm_path))\n    arr = ds.pixel_array.astype(np.float32)\n\n    if getattr(ds, 'PhotometricInterpretation', '') == 'MONOCHROME1':\n        arr = arr.max() - arr\n\n    arr -= arr.min()\n    if arr.max() > 0:\n        arr = arr / arr.max()\n\n    arr = (arr * 255).clip(0, 255).astype(np.uint8)\n    img = Image.fromarray(arr).convert('RGB')\n    img = img.resize(IMG_SIZE)\n    return img\n\ndef stage_split(df, split_name):\n    missing = []\n    written = 0\n\n    for row in tqdm(df.itertuples(index=False), total=len(df), desc=f'Staging {split_name}'):\n        patient_id = row.patientId\n        target_name = row.target_name\n\n        dcm_path = IMAGES_DIR / f'{patient_id}.dcm'\n        out_path = STAGED / split_name / target_name / f'{patient_id}.png'\n\n        if not dcm_path.exists():\n            missing.append(patient_id)\n            continue\n\n        try:\n            img = dicom_to_uint8_rgb(dcm_path)\n            img.save(out_path)\n            written += 1\n        except Exception:\n            missing.append(patient_id)\n\n    return written, missing\n\ntrain_written, train_missing = stage_split(train_df, 'train')\ntest_written, test_missing = stage_split(test_df, 'test')\n\nprint(f'Train written : {train_written}')\nprint(f'Test written  : {test_written}')\nprint(f'Train missing : {len(train_missing)}')\nprint(f'Test missing  : {len(test_missing)}')\n\nif train_missing:\n    print('Sample missing/corrupt train IDs:', train_missing[:10])\nif test_missing:\n    print('Sample missing/corrupt test IDs :', test_missing[:10])\n\nfor split in ('train', 'test'):\n    for cls in ('NON_PNEUMONIA', 'PNEUMONIA'):\n        n = len(list((STAGED / split / cls).glob('*.png')))\n        print(f'{split}/{cls}: {n} images')\n","metadata":{"execution":{"iopub.status.busy":"2026-03-29T19:35:13.006061Z","iopub.execute_input":"2026-03-29T19:35:13.006324Z","iopub.status.idle":"2026-03-29T19:58:24.651522Z","shell.execute_reply.started":"2026-03-29T19:35:13.006300Z","shell.execute_reply":"2026-03-29T19:58:24.650642Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"255d642a","cell_type":"markdown","source":"## 3 · Imports","metadata":{}},{"id":"b2e1a88a","cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import EfficientNetB0\n\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import (\n    accuracy_score, precision_score,\n    f1_score, confusion_matrix,\n    classification_report\n)\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nprint('TensorFlow:', tf.__version__)\n","metadata":{"execution":{"iopub.status.busy":"2026-03-29T19:58:24.652913Z","iopub.execute_input":"2026-03-29T19:58:24.653540Z","iopub.status.idle":"2026-03-29T19:59:06.475267Z","shell.execute_reply.started":"2026-03-29T19:58:24.653509Z","shell.execute_reply":"2026-03-29T19:59:06.474437Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"1b44898a","cell_type":"markdown","source":"## 4 · Load staged images","metadata":{}},{"id":"d26dc4e6","cell_type":"code","source":"TRAIN_DIR  = str(STAGED / 'train')\nTEST_DIR   = str(STAGED / 'test')\nBATCH_SIZE = 32\n\ndatagen = ImageDataGenerator(rescale=1.0 / 255)\n\ntrain_gen = datagen.flow_from_directory(\n    TRAIN_DIR,\n    target_size=IMG_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode='sparse',\n    shuffle=False\n)\n\ntest_gen = datagen.flow_from_directory(\n    TEST_DIR,\n    target_size=IMG_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode='sparse',\n    shuffle=False\n)\n\nCLASS_NAMES = list(train_gen.class_indices.keys())\nprint('Class indices:', train_gen.class_indices)\n","metadata":{"execution":{"iopub.status.busy":"2026-03-29T19:59:06.476990Z","iopub.execute_input":"2026-03-29T19:59:06.477460Z","iopub.status.idle":"2026-03-29T19:59:06.745764Z","shell.execute_reply.started":"2026-03-29T19:59:06.477432Z","shell.execute_reply":"2026-03-29T19:59:06.745031Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"956654f7","cell_type":"markdown","source":"## 5 · Load EfficientNetB0","metadata":{}},{"id":"79413311","cell_type":"code","source":"base_model = EfficientNetB0(\n    weights='imagenet',\n    include_top=False,\n    input_shape=(224, 224, 3)\n)\nbase_model.trainable = False\nprint('EfficientNetB0 loaded (frozen).')\n","metadata":{"execution":{"iopub.status.busy":"2026-03-29T19:59:06.746687Z","iopub.execute_input":"2026-03-29T19:59:06.746956Z","iopub.status.idle":"2026-03-29T19:59:12.684727Z","shell.execute_reply.started":"2026-03-29T19:59:06.746929Z","shell.execute_reply":"2026-03-29T19:59:12.683944Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"7d19ac9c","cell_type":"code","source":"def extract_features(generator, model):\n    generator.reset()\n    feats = model.predict(generator, verbose=1)\n    return feats.reshape(feats.shape[0], -1)\n\nprint('Extracting training features...')\nX_train = extract_features(train_gen, base_model)\ny_train = train_gen.classes\n\nprint('Extracting test features...')\nX_test = extract_features(test_gen, base_model)\ny_test = test_gen.classes\n\nprint(f'Train features: {X_train.shape}')\nprint(f'Test features : {X_test.shape}')\n","metadata":{"execution":{"iopub.status.busy":"2026-03-29T19:59:12.686031Z","iopub.execute_input":"2026-03-29T19:59:12.686299Z","iopub.status.idle":"2026-03-29T20:01:01.117715Z","shell.execute_reply.started":"2026-03-29T19:59:12.686275Z","shell.execute_reply":"2026-03-29T20:01:01.117076Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"0233e0e1","cell_type":"markdown","source":"## 6 · Train Random Forest","metadata":{}},{"id":"66033c49","cell_type":"code","source":"rf = RandomForestClassifier(\n    n_estimators=100,\n    random_state=42,\n    n_jobs=-1,\n    class_weight='balanced'\n)\nrf.fit(X_train, y_train)\nprint('Random Forest trained.')\n","metadata":{"execution":{"iopub.status.busy":"2026-03-29T20:01:01.118688Z","iopub.execute_input":"2026-03-29T20:01:01.118949Z","iopub.status.idle":"2026-03-29T20:08:28.242528Z","shell.execute_reply.started":"2026-03-29T20:01:01.118925Z","shell.execute_reply":"2026-03-29T20:08:28.241853Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"1c7d4e4e","cell_type":"markdown","source":"## 7 · Evaluate","metadata":{}},{"id":"35c111cd","cell_type":"code","source":"y_pred = rf.predict(X_test)\n\naccuracy = accuracy_score(y_test, y_pred)\nprecision = precision_score(y_test, y_pred, average='weighted', zero_division=0)\nf1 = f1_score(y_test, y_pred, average='weighted', zero_division=0)\n\nprint(f'Accuracy  : {accuracy * 100:.2f}%')\nprint(f'Precision : {precision * 100:.2f}%')\nprint(f'F1-Score  : {f1 * 100:.2f}%')\nprint()\nprint(classification_report(y_test, y_pred, target_names=CLASS_NAMES, zero_division=0))\n","metadata":{"execution":{"iopub.status.busy":"2026-03-29T20:08:28.243731Z","iopub.execute_input":"2026-03-29T20:08:28.244426Z","iopub.status.idle":"2026-03-29T20:08:28.544713Z","shell.execute_reply.started":"2026-03-29T20:08:28.244398Z","shell.execute_reply":"2026-03-29T20:08:28.544012Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"763bf5f1","cell_type":"markdown","source":"## 8 · Confusion matrix","metadata":{}},{"id":"beb9d19d","cell_type":"code","source":"cm = confusion_matrix(y_test, y_pred)\n\nplt.figure(figsize=(6, 5))\nsns.heatmap(\n    cm, annot=True, fmt='d', cmap='Blues',\n    xticklabels=CLASS_NAMES,\n    yticklabels=CLASS_NAMES\n)\nplt.xlabel('Predicted Labels')\nplt.ylabel('True Labels')\nplt.title('Confusion Matrix — EfficientNetB0 + Random Forest')\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2026-03-29T20:08:28.545560Z","iopub.execute_input":"2026-03-29T20:08:28.545816Z","iopub.status.idle":"2026-03-29T20:08:28.772916Z","shell.execute_reply.started":"2026-03-29T20:08:28.545792Z","shell.execute_reply":"2026-03-29T20:08:28.772244Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"bd3d9545","cell_type":"markdown","source":"## 9 · Metrics chart","metadata":{}},{"id":"53b8e8cf","cell_type":"code","source":"metrics = {'Accuracy': accuracy, 'Precision': precision, 'F1-Score': f1}\n\nfig, ax = plt.subplots(figsize=(6, 4))\nbars = ax.bar(metrics.keys(), [v * 100 for v in metrics.values()])\nax.bar_label(bars, fmt='%.1f%%', padding=3)\nax.set_ylim(0, 115)\nax.set_ylabel('Score (%)')\nax.set_title('EfficientNetB0 + RF — RSNA Pneumonia Detection Challenge')\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2026-03-29T20:08:28.774755Z","iopub.execute_input":"2026-03-29T20:08:28.775016Z","iopub.status.idle":"2026-03-29T20:08:28.905068Z","shell.execute_reply.started":"2026-03-29T20:08:28.774993Z","shell.execute_reply":"2026-03-29T20:08:28.904416Z"},"trusted":true},"outputs":[],"execution_count":null}]}