{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"colab":{"provenance":[],"authorship_tag":"ABX9TyOz2+QHgzIGhgKgM6t/fTGP"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport cv2\nimport pydicom\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import (\n    accuracy_score, precision_score, recall_score, f1_score,\n    confusion_matrix, classification_report, roc_curve, auc\n)\n\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.applications.efficientnet import preprocess_input\n\n","metadata":{"id":"uSPTdx6ME6U0","executionInfo":{"status":"ok","timestamp":1729833005881,"user_tz":240,"elapsed":37543,"user":{"displayName":"Ojo","userId":"15914584124912270723"}},"outputId":"1d30e469-8281-4112-94ef-e88517584c37","trusted":true,"execution":{"iopub.status.busy":"2026-03-28T00:53:19.714149Z","iopub.execute_input":"2026-03-28T00:53:19.714583Z","iopub.status.idle":"2026-03-28T00:53:19.719630Z","shell.execute_reply.started":"2026-03-28T00:53:19.714543Z","shell.execute_reply":"2026-03-28T00:53:19.718848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# 2) PATHS KAGGLE\n# =========================\nDATASET_PATH = \"/kaggle/input/competitions/rsna-pneumonia-detection-challenge\"\n\nTRAIN_IMG_DIR = os.path.join(DATASET_PATH, \"stage_2_train_images\")\nTEST_IMG_DIR  = os.path.join(DATASET_PATH, \"stage_2_test_images\")\n\nTRAIN_LABELS_CSV = os.path.join(DATASET_PATH, \"stage_2_train_labels.csv\")\nCLASS_INFO_CSV   = os.path.join(DATASET_PATH, \"stage_2_detailed_class_info.csv\")\n\n# =========================\n# 3) LECTURE DES CSV\n# =========================\nlabels_df = pd.read_csv(TRAIN_LABELS_CSV)\nclass_info_df = pd.read_csv(CLASS_INFO_CSV)\n\nprint(\"labels_df shape:\", labels_df.shape)\nprint(\"class_info_df shape:\", class_info_df.shape)\nprint(labels_df.head())\nprint(class_info_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-28T00:53:28.226876Z","iopub.execute_input":"2026-03-28T00:53:28.227190Z","iopub.status.idle":"2026-03-28T00:53:28.296255Z","shell.execute_reply.started":"2026-03-28T00:53:28.227160Z","shell.execute_reply":"2026-03-28T00:53:28.295587Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"***label binaire par image***","metadata":{}},{"cell_type":"code","source":"\n# =========================\n# 4) CONSTRUIRE UN DATAFRAME BINAIRE PAR IMAGE\n# =========================\n# Pour chaque patientId, si au moins une ligne a Target=1 => pneumonie\nimage_labels = labels_df.groupby(\"patientId\")[\"Target\"].max().reset_index()\n\n# Construire le chemin vers le fichier DICOM\nimage_labels[\"filepath\"] = image_labels[\"patientId\"].apply(\n    lambda x: os.path.join(TRAIN_IMG_DIR, f\"{x}.dcm\")\n)\n\n# Garder seulement les fichiers existants\nimage_labels = image_labels[image_labels[\"filepath\"].apply(os.path.exists)].reset_index(drop=True)\n\nprint(\"Nombre total d'images:\", len(image_labels))\nprint(image_labels[\"Target\"].value_counts())\nprint(image_labels.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-28T00:53:37.121415Z","iopub.execute_input":"2026-03-28T00:53:37.121991Z","iopub.status.idle":"2026-03-28T00:54:10.998859Z","shell.execute_reply.started":"2026-03-28T00:53:37.121959Z","shell.execute_reply":"2026-03-28T00:54:10.997967Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Split train / validation**","metadata":{}},{"cell_type":"code","source":"# =========================\n# 5) TRAIN / VALIDATION SPLIT\n# =========================\ntrain_df, val_df = train_test_split(\n    image_labels,\n    test_size=0.2,\n    random_state=42,\n    stratify=image_labels[\"Target\"]\n)\n\ntrain_df = train_df.reset_index(drop=True)\nval_df = val_df.reset_index(drop=True)\n\nprint(\"Train:\", train_df.shape)\nprint(\"Validation:\", val_df.shape)\nprint(\"Train labels:\\n\", train_df[\"Target\"].value_counts())\nprint(\"Validation labels:\\n\", val_df[\"Target\"].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-28T00:54:44.939832Z","iopub.execute_input":"2026-03-28T00:54:44.940173Z","iopub.status.idle":"2026-03-28T00:54:44.969278Z","shell.execute_reply.started":"2026-03-28T00:54:44.940144Z","shell.execute_reply":"2026-03-28T00:54:44.968393Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Loader DICOM compatible avec EfficientNet**","metadata":{}},{"cell_type":"code","source":"# =========================\n# 6) LECTURE DICOM\n# =========================\ndef load_dicom_image(path, img_size=(224, 224)):\n    ds = pydicom.dcmread(path)\n    img = ds.pixel_array.astype(np.float32)\n\n    # Normalisation min-max\n    img = img - np.min(img)\n    if np.max(img) > 0:\n        img = img / np.max(img)\n\n    # Convertir en uint8 pour resize OpenCV\n    img = (img * 255).astype(np.uint8)\n\n    # Resize\n    img = cv2.resize(img, img_size)\n\n    # Convertir grayscale -> RGB\n    img = np.stack([img, img, img], axis=-1)\n\n    # preprocess_input EfficientNet\n    img = preprocess_input(img.astype(np.float32))\n\n    return img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-28T00:55:57.420146Z","iopub.execute_input":"2026-03-28T00:55:57.420668Z","iopub.status.idle":"2026-03-28T00:55:57.426465Z","shell.execute_reply.started":"2026-03-28T00:55:57.420635Z","shell.execute_reply":"2026-03-28T00:55:57.425645Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Générateur Keras personnalisé**","metadata":{}},{"cell_type":"code","source":"# =========================\n# 7) GENERATOR PERSONNALISÉ\n# =========================\nclass DicomDataGenerator(Sequence):\n    def __init__(self, dataframe, x_col, y_col=None, batch_size=32, img_size=(224, 224), shuffle=False):\n        self.dataframe = dataframe.copy()\n        self.x_col = x_col\n        self.y_col = y_col\n        self.batch_size = batch_size\n        self.img_size = img_size\n        self.shuffle = shuffle\n        self.indices = np.arange(len(self.dataframe))\n        self.on_epoch_end()\n\n    def __len__(self):\n        return int(np.ceil(len(self.dataframe) / self.batch_size))\n\n    def __getitem__(self, index):\n        batch_indices = self.indices[index * self.batch_size:(index + 1) * self.batch_size]\n        batch_df = self.dataframe.iloc[batch_indices]\n\n        images = np.array([load_dicom_image(p, self.img_size) for p in batch_df[self.x_col]])\n\n        if self.y_col is not None:\n            labels = batch_df[self.y_col].values.astype(np.int32)\n            return images, labels\n        else:\n            return images\n\n    def on_epoch_end(self):\n        if self.shuffle:\n            np.random.shuffle(self.indices)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-28T00:56:05.168149Z","iopub.execute_input":"2026-03-28T00:56:05.168481Z","iopub.status.idle":"2026-03-28T00:56:05.175595Z","shell.execute_reply.started":"2026-03-28T00:56:05.168454Z","shell.execute_reply":"2026-03-28T00:56:05.174932Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Créer les générateurs**","metadata":{}},{"cell_type":"code","source":"# =========================\n# 8) GENERATORS\n# =========================\ntrain_generator = DicomDataGenerator(\n    dataframe=train_df,\n    x_col=\"filepath\",\n    y_col=\"Target\",\n    batch_size=32,\n    img_size=(224, 224),\n    shuffle=False\n)\n\nval_generator = DicomDataGenerator(\n    dataframe=val_df,\n    x_col=\"filepath\",\n    y_col=\"Target\",\n    batch_size=32,\n    img_size=(224, 224),\n    shuffle=False\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-28T00:56:15.238033Z","iopub.execute_input":"2026-03-28T00:56:15.238748Z","iopub.status.idle":"2026-03-28T00:56:15.245148Z","shell.execute_reply.started":"2026-03-28T00:56:15.238719Z","shell.execute_reply":"2026-03-28T00:56:15.244369Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Extraire les features avec EfficientNetB0**","metadata":{}},{"cell_type":"code","source":"# =========================\n# 9) MODELE EFFICIENTNET\n# =========================\nbase_model = EfficientNetB0(\n    weights=\"imagenet\",\n    include_top=False,\n    pooling=\"avg\",\n    input_shape=(224, 224, 3)\n)\n\nbase_model.trainable = False\n\ndef extract_features(generator, model):\n    features = model.predict(generator, verbose=1)\n    return features\n\ntrain_features = extract_features(train_generator, base_model)\nval_features = extract_features(val_generator, base_model)\n\ntrain_labels = train_df[\"Target\"].values\nval_labels = val_df[\"Target\"].values\n\nprint(\"train_features shape:\", train_features.shape)\nprint(\"val_features shape:\", val_features.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-28T01:02:42.179407Z","iopub.execute_input":"2026-03-28T01:02:42.179801Z","iopub.status.idle":"2026-03-28T01:08:16.487608Z","shell.execute_reply.started":"2026-03-28T01:02:42.179761Z","shell.execute_reply":"2026-03-28T01:08:16.486823Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Random Forest + métriques**","metadata":{}},{"cell_type":"code","source":"# =========================\n# 10) RANDOM FOREST\n# =========================\nrf_classifier = RandomForestClassifier(\n    n_estimators=100,\n    random_state=42,\n    n_jobs=-1\n)\n\nrf_classifier.fit(train_features, train_labels)\n\n# Prédictions\ny_pred = rf_classifier.predict(val_features)\ny_score = rf_classifier.predict_proba(val_features)[:, 1]\n\n# Accuracy\naccuracy = accuracy_score(val_labels, y_pred)\nprint(f\"Accuracy: {accuracy * 100:.2f}%\")\n\n# Confusion Matrix\nconf_matrix = confusion_matrix(val_labels, y_pred)\nplt.figure(figsize=(8, 6))\nsns.heatmap(\n    conf_matrix,\n    annot=True,\n    fmt='d',\n    cmap='Blues',\n    xticklabels=[\"Normal\", \"Pneumonia\"],\n    yticklabels=[\"Normal\", \"Pneumonia\"]\n)\nplt.xlabel('Predicted Labels')\nplt.ylabel('True Labels')\nplt.title('Confusion Matrix')\nplt.show()\n\n# Precision, Recall, F1\nprecision = precision_score(val_labels, y_pred)\nrecall = recall_score(val_labels, y_pred)\nf1 = f1_score(val_labels, y_pred)\n\nprint(f\"Precision: {precision * 100:.2f}%\")\nprint(f\"Recall: {recall * 100:.2f}%\")\nprint(f\"F1 Score: {f1 * 100:.2f}%\")\n\n# Classification Report\nprint(classification_report(val_labels, y_pred, target_names=[\"Normal\", \"Pneumonia\"]))\n\n# ROC Curve\nfpr, tpr, _ = roc_curve(val_labels, y_score)\nroc_auc = auc(fpr, tpr)\n\nplt.figure(figsize=(7, 5))\nplt.plot(fpr, tpr, lw=2, label=f'ROC curve (AUC = {roc_auc:.2f})')\nplt.plot([0, 1], [0, 1], 'k--', lw=2)\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('ROC Curve - RSNA Pneumonia')\nplt.legend(loc=\"lower right\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-28T01:08:29.966702Z","iopub.execute_input":"2026-03-28T01:08:29.967426Z","iopub.status.idle":"2026-03-28T01:09:46.750458Z","shell.execute_reply.started":"2026-03-28T01:08:29.967389Z","shell.execute_reply":"2026-03-28T01:09:46.749576Z"}},"outputs":[],"execution_count":null}]}