{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042,"isSourceIdPinned":false}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# EfficientNetB0 + Random Forest on the RSNA Dataset using split 80/10/10","metadata":{}},{"cell_type":"code","source":"# ─── RSNA ADAPTER (remplace find_dataset_root) ────────────────────────────────\n\nimport os\nimport pathlib\nimport numpy as np\nimport pandas as pd\nfrom collections import defaultdict\nfrom PIL import Image\n\nIMAGE_EXTS = {'.png', '.jpg', '.jpeg', '.bmp', '.webp'}\n\ndef is_image_file(p):\n    return p.is_file() and p.suffix.lower() in IMAGE_EXTS\n\ndef count_images_in_dir(d):\n    try:\n        return sum(1 for x in d.iterdir() if is_image_file(x))\n    except Exception:\n        return 0\n\ndef class_dirs_with_images(root):\n    out = []\n    try:\n        for child in root.iterdir():\n            if child.is_dir():\n                direct_images = count_images_in_dir(child)\n                nested_images = count_images_in_dir(child / 'images')\n                if direct_images > 0 or nested_images > 0:\n                    out.append(child)\n    except Exception:\n        pass\n    return out\n\ndef looks_like_split_root(root):\n    split_names = {'train', 'test', 'val', 'valid', 'validation'}\n    try:\n        split_dirs = [p for p in root.iterdir() if p.is_dir() and p.name.lower() in split_names]\n    except Exception:\n        return False\n    if not split_dirs:\n        return False\n    for s in split_dirs:\n        if len(class_dirs_with_images(s)) >= 2:\n            return True\n    return False\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-05T13:56:12.602672Z","iopub.execute_input":"2026-05-05T13:56:12.603140Z","iopub.status.idle":"2026-05-05T13:56:12.617025Z","shell.execute_reply.started":"2026-05-05T13:56:12.603098Z","shell.execute_reply":"2026-05-05T13:56:12.616300Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pathlib\n\nrsna = pathlib.Path('/kaggle/input/competitions/rsna-pneumonia-detection-challenge')\nprint(\"=== Contenu du dataset RSNA ===\")\nfor p in sorted(rsna.iterdir()):\n    marker = '[DIR]' if p.is_dir() else '[FILE]'\n    print(f\"{marker} {p.name}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-05T13:56:12.618917Z","iopub.execute_input":"2026-05-05T13:56:12.619267Z","iopub.status.idle":"2026-05-05T13:56:12.642612Z","shell.execute_reply.started":"2026-05-05T13:56:12.619231Z","shell.execute_reply":"2026-05-05T13:56:12.641845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 1. Localiser les fichiers RSNA ────────────────────────────────────────────\nRSNA_ROOT  = pathlib.Path('/kaggle/input/competitions/rsna-pneumonia-detection-challenge')\nLABELS_CSV = RSNA_ROOT / 'stage_2_train_labels.csv'\nDICOM_DIR  = RSNA_ROOT / 'stage_2_train_images'\n\nassert LABELS_CSV.exists(), f'Labels CSV non trouvé : {LABELS_CSV}'\nassert DICOM_DIR.exists(),  f'Dossier DICOM non trouvé : {DICOM_DIR}'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-05T13:56:12.643397Z","iopub.execute_input":"2026-05-05T13:56:12.643665Z","iopub.status.idle":"2026-05-05T13:56:12.647812Z","shell.execute_reply.started":"2026-05-05T13:56:12.643645Z","shell.execute_reply":"2026-05-05T13:56:12.647080Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 2. Mapping patientId → label (0=NORMAL, 1=PNEUMONIA) ─────────────────────\ndf = pd.read_csv(LABELS_CSV)\npatient_labels = (\n    df.groupby('patientId')['Target']\n    .max()\n    .reset_index()\n)\nprint(f'Total patients : {len(patient_labels)}')\nprint(patient_labels['Target'].value_counts().rename({0: 'NORMAL', 1: 'PNEUMONIA'}))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-05T13:56:12.648588Z","iopub.execute_input":"2026-05-05T13:56:12.648816Z","iopub.status.idle":"2026-05-05T13:56:12.716799Z","shell.execute_reply.started":"2026-05-05T13:56:12.648797Z","shell.execute_reply":"2026-05-05T13:56:12.716001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 3. Conversion DICOM → PNG dans des dossiers par classe ────────────────────\nORGANISED = pathlib.Path('/kaggle/working/rsna_organised')\nCLASS_MAP  = {0: 'NORMAL', 1: 'PNEUMONIA'}\n\nif not ORGANISED.exists():\n    try:\n        import pydicom\n    except ImportError:\n        os.system('pip install pydicom -q')\n        import pydicom\n\n    for cls_name in CLASS_MAP.values():\n        (ORGANISED / cls_name).mkdir(parents=True, exist_ok=True)\n\n    for _, row in patient_labels.iterrows():\n        pid      = row['patientId']\n        target   = int(row['Target'])\n        dcm_path = DICOM_DIR / f'{pid}.dcm'\n\n        if not dcm_path.exists():\n            continue\n\n        ds  = pydicom.dcmread(str(dcm_path))\n        arr = ds.pixel_array.astype(np.float32)\n        arr -= arr.min()\n        if arr.max() > 0:\n            arr /= arr.max()\n        arr = (arr * 255).astype(np.uint8)\n\n        img      = Image.fromarray(arr).convert('RGB')   # 3 canaux pour EfficientNet\n        out_path = ORGANISED / CLASS_MAP[target] / f'{pid}.png'\n        img.save(str(out_path))\n\n    print('Conversion DICOM → PNG terminée.')\nelse:\n    print('Dossier organisé déjà existant, conversion ignorée.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-05T13:56:12.718689Z","iopub.execute_input":"2026-05-05T13:56:12.718982Z","iopub.status.idle":"2026-05-05T13:56:12.726815Z","shell.execute_reply.started":"2026-05-05T13:56:12.718959Z","shell.execute_reply":"2026-05-05T13:56:12.725981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 4. DATA_ROOT pointe vers les dossiers par classe ─────────────────────────\nDATA_ROOT = ORGANISED\n\nprint('Detected DATA_ROOT:', DATA_ROOT)\nprint('\\nImmediate contents:')\nprint(sorted([p.name for p in DATA_ROOT.iterdir()]))\n\nclass_dirs = class_dirs_with_images(DATA_ROOT)\nprint('\\nDetected structure: unsplit class folders.')\nprint('Class folders:', [c.name for c in class_dirs])\n\n# ─────────────────────────────────────────────────────────────────────────────\n# ## Prepare train/test folders\n# ─────────────────────────────────────────────────────────────────────────────\n\nimport shutil\nimport random\n\nSTAGED      = pathlib.Path('/kaggle/working/staged_kermany')\nTRAIN_RATIO = 0.80\nVAL_RATIO   = 0.10\nTEST_RATIO  = 0.10\nrandom.seed(42)\n\nif STAGED.exists():\n    shutil.rmtree(STAGED)\n\ndef image_files_in_class_dir(class_dir):\n    images_subdir = class_dir / 'images'\n    search_dir = images_subdir if images_subdir.exists() else class_dir\n    return sorted([p for p in search_dir.iterdir() if is_image_file(p)])\n\ndef stage_from_existing_splits(root):\n    found = defaultdict(list)\n\n    for split_dir in [p for p in root.iterdir() if p.is_dir()]:\n        if split_dir.name.lower() not in {'train', 'test', 'val', 'valid', 'validation'}:\n            continue\n        for class_dir in class_dirs_with_images(split_dir):\n            found[class_dir.name].extend(image_files_in_class_dir(class_dir))\n\n    class_names = sorted(found.keys())\n    if len(class_names) < 2:\n        raise FileNotFoundError('Could not find at least two class folders inside dataset splits.')\n\n    for split in ('train', 'val', 'test'):\n        for cls in class_names:\n            (STAGED / split / cls).mkdir(parents=True, exist_ok=True)\n\n    for cls, files in found.items():\n        random.shuffle(files)\n        n       = len(files)\n        n_train = int(n * TRAIN_RATIO)\n        n_val   = int(n * VAL_RATIO)\n\n        train_files = files[:n_train]\n        val_files   = files[n_train:n_train + n_val]\n        test_files  = files[n_train + n_val:]\n\n        for img in train_files:\n            shutil.copy(img, STAGED / 'train' / cls / img.name)\n        for img in val_files:\n            shutil.copy(img, STAGED / 'val' / cls / img.name)\n        for img in test_files:\n            shutil.copy(img, STAGED / 'test' / cls / img.name)\n\n    return class_names\n\ndef stage_from_unsplit_classes(root):\n    class_dirs = class_dirs_with_images(root)\n    if len(class_dirs) < 2:\n        raise FileNotFoundError(\n            f'Expected at least two class folders under {root}, found {[p.name for p in class_dirs]}.'\n        )\n\n    class_names = sorted([p.name for p in class_dirs])\n\n    for split in ('train', 'val', 'test'):\n        for cls in class_names:\n            (STAGED / split / cls).mkdir(parents=True, exist_ok=True)\n\n    for class_dir in class_dirs:\n        files = image_files_in_class_dir(class_dir)\n        if not files:\n            continue\n\n        random.shuffle(files)\n        n       = len(files)\n        n_train = int(n * TRAIN_RATIO)\n        n_val   = int(n * VAL_RATIO)\n\n        train_files = files[:n_train]\n        val_files   = files[n_train:n_train + n_val]\n        test_files  = files[n_train + n_val:]\n\n        for img in train_files:\n            shutil.copy(img, STAGED / 'train' / class_dir.name / img.name)\n        for img in val_files:\n            shutil.copy(img, STAGED / 'val' / class_dir.name / img.name)\n        for img in test_files:\n            shutil.copy(img, STAGED / 'test' / class_dir.name / img.name)\n\n    return class_names\n\nif looks_like_split_root(DATA_ROOT):\n    CLASS_NAMES_RAW = stage_from_existing_splits(DATA_ROOT)\nelse:\n    CLASS_NAMES_RAW = stage_from_unsplit_classes(DATA_ROOT)\n\nprint('Using classes:', CLASS_NAMES_RAW)\nfor split in ('train', 'val', 'test'):\n    for cls in CLASS_NAMES_RAW:\n        n = len(list((STAGED / split / cls).glob('*')))\n        print(f'{split}/{cls}: {n} images')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-05T13:56:12.727909Z","iopub.execute_input":"2026-05-05T13:56:12.728186Z","iopub.status.idle":"2026-05-05T13:56:14.389060Z","shell.execute_reply.started":"2026-05-05T13:56:12.728162Z","shell.execute_reply":"2026-05-05T13:56:14.388262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ## 3 · Imports\n\n\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import EfficientNetB0\n\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import (\n    accuracy_score, precision_score, recall_score,\n    f1_score, roc_auc_score, confusion_matrix,\n    classification_report\n)\n\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nprint('TensorFlow:', tf.__version__)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-05T13:56:14.389883Z","iopub.execute_input":"2026-05-05T13:56:14.390279Z","iopub.status.idle":"2026-05-05T13:56:14.395279Z","shell.execute_reply.started":"2026-05-05T13:56:14.390245Z","shell.execute_reply":"2026-05-05T13:56:14.394613Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ## 4 · Data Generators\n# Tolérance aux images tronquées\nfrom PIL import ImageFile\nImageFile.LOAD_TRUNCATED_IMAGES = True\n\nTRAIN_DIR  = str(STAGED / 'train')\nVAL_DIR    = str(STAGED / 'val')\nTEST_DIR   = str(STAGED / 'test')\nBATCH_SIZE = 32\nIMG_SIZE   = (224, 224)\n\ndatagen = ImageDataGenerator(rescale=1.0 / 255)\n\ntrain_gen = datagen.flow_from_directory(\n    TRAIN_DIR,\n    target_size=IMG_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode='sparse',\n    shuffle=False\n)\n\nval_gen = datagen.flow_from_directory(\n    VAL_DIR,\n    target_size=IMG_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode='sparse',\n    shuffle=False\n)\n\ntest_gen = datagen.flow_from_directory(\n    TEST_DIR,\n    target_size=IMG_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode='sparse',\n    shuffle=False\n)\n\nCLASS_NAMES = list(train_gen.class_indices.keys())\nprint('Class indices:', train_gen.class_indices)\nBATCH_SIZE = 32\nIMG_SIZE   = (224, 224)\n\ndatagen = ImageDataGenerator(rescale=1.0 / 255)\n\ntrain_gen = datagen.flow_from_directory(\n    TRAIN_DIR,\n    target_size=IMG_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode='sparse',\n    shuffle=False\n)\n\ntest_gen = datagen.flow_from_directory(\n    TEST_DIR,\n    target_size=IMG_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode='sparse',\n    shuffle=False\n)\n\nCLASS_NAMES = list(train_gen.class_indices.keys())\nprint('Class indices:', train_gen.class_indices)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-05T13:56:14.396211Z","iopub.execute_input":"2026-05-05T13:56:14.396464Z","iopub.status.idle":"2026-05-05T13:56:14.473474Z","shell.execute_reply.started":"2026-05-05T13:56:14.396442Z","shell.execute_reply":"2026-05-05T13:56:14.472883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ## 5 · EfficientNetB0 Feature Extractor\n\n\nbase_model = EfficientNetB0(\n    weights='imagenet',\n    include_top=False,\n    input_shape=(224, 224, 3)\n)\nbase_model.trainable = False\nprint('EfficientNetB0 loaded (frozen).')\n\ndef extract_features(generator, model):\n    generator.reset()\n    all_feats = []\n    all_labels = []\n\n    for i in range(len(generator)):\n        try:\n            batch_x, batch_y = generator[i]\n            feats = model.predict(batch_x, verbose=0)\n            all_feats.append(feats.reshape(feats.shape[0], -1))\n            all_labels.extend(batch_y)\n        except Exception as e:\n            print(f'  Batch {i} ignoré : {e}')\n            continue\n\n    return np.vstack(all_feats), np.array(all_labels)\n\nprint('Extracting training features...')\nX_train, y_train = extract_features(train_gen, base_model)\n\nprint('Extracting validation features...')\nX_val, y_val = extract_features(val_gen, base_model)\n\nprint('Extracting test features...')\nX_test, y_test = extract_features(test_gen, base_model)\n\nprint(f'Train features: {X_train.shape}')\nprint(f'Val features  : {X_val.shape}')\nprint(f'Test features : {X_test.shape}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-05T13:56:14.474417Z","iopub.execute_input":"2026-05-05T13:56:14.474754Z","iopub.status.idle":"2026-05-05T13:58:20.950936Z","shell.execute_reply.started":"2026-05-05T13:56:14.474731Z","shell.execute_reply":"2026-05-05T13:58:20.950218Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ## 6 · Train Random Forest\n\n\nrf = RandomForestClassifier(\n    n_estimators=100,\n    random_state=42,\n    n_jobs=-1\n)\nrf.fit(X_train, y_train)\nprint('Random Forest trained.')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-05T13:58:20.951802Z","iopub.execute_input":"2026-05-05T13:58:20.952004Z","iopub.status.idle":"2026-05-05T13:58:52.987766Z","shell.execute_reply.started":"2026-05-05T13:58:20.951984Z","shell.execute_reply":"2026-05-05T13:58:52.986971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ## 7 · Metrics — Accuracy · Precision · Recall · F1-Score · AUC\n\n\ny_pred = rf.predict(X_test)\n\n# Probabilites pour la classe positive (PNEUMONIA = 1)\npos_idx = list(rf.classes_).index(1)\ny_prob  = rf.predict_proba(X_test)[:, pos_idx]\n\naccuracy  = accuracy_score(y_test, y_pred)\nprecision = precision_score(y_test, y_pred, zero_division=0)\nrecall    = recall_score(y_test, y_pred, zero_division=0)\nf1        = f1_score(y_test, y_pred, zero_division=0)\nauc       = roc_auc_score(y_test, y_prob)\n\nprint(f'Accuracy  : {accuracy * 100:.2f}%')\nprint(f'Precision : {precision * 100:.2f}%')\nprint(f'Recall    : {recall * 100:.2f}%')\nprint(f'F1-Score  : {f1 * 100:.2f}%')\nprint(f'AUC       : {auc * 100:.2f}%')\n\nprint()\nprint(classification_report(y_test, y_pred, target_names=CLASS_NAMES, zero_division=0))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-05T13:58:52.988860Z","iopub.execute_input":"2026-05-05T13:58:52.989147Z","iopub.status.idle":"2026-05-05T13:58:53.108786Z","shell.execute_reply.started":"2026-05-05T13:58:52.989124Z","shell.execute_reply":"2026-05-05T13:58:53.108182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ## 8 · Confusion Matrix\n\n\ncm = confusion_matrix(y_test, y_pred)\n\nplt.figure(figsize=(8, 6))\nsns.heatmap(\n    cm, annot=True, fmt='d', cmap='Blues',\n    xticklabels=CLASS_NAMES,\n    yticklabels=CLASS_NAMES\n)\nplt.xlabel('Predicted Labels')\nplt.ylabel('True Labels')\nplt.title('Confusion Matrix — EfficientNetB0 + Random Forest')\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-05T13:58:53.109791Z","iopub.execute_input":"2026-05-05T13:58:53.110089Z","iopub.status.idle":"2026-05-05T13:58:53.328225Z","shell.execute_reply.started":"2026-05-05T13:58:53.110054Z","shell.execute_reply":"2026-05-05T13:58:53.327630Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ## 9 · Summary Bar Chart\n\n\nmetrics = {\n    'Accuracy' : accuracy,\n    'Precision': precision,\n    'Recall'   : recall,\n    'F1-Score' : f1,\n    'AUC'      : auc\n}\n\nfig, ax = plt.subplots(figsize=(6, 4))\nbars = ax.bar(metrics.keys(), [v * 100 for v in metrics.values()])\nax.bar_label(bars, fmt='%.1f%%', padding=3)\nax.set_ylim(0, 115)\nax.set_ylabel('Score (%)')\nax.set_title('EfficientNetB0 + RF — RSNA Pneumonia Detection Dataset')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-05T13:58:53.329111Z","iopub.execute_input":"2026-05-05T13:58:53.329442Z","iopub.status.idle":"2026-05-05T13:58:53.460286Z","shell.execute_reply.started":"2026-05-05T13:58:53.329419Z","shell.execute_reply":"2026-05-05T13:58:53.459654Z"}},"outputs":[],"execution_count":null}]}