{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nimport tensorflow as tf\nimport random\nfrom tensorflow.keras import layers, Model\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.applications.densenet import preprocess_input\nfrom tensorflow.keras.applications import DenseNet121\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, confusion_matrix, roc_auc_score, roc_curve\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, EarlyStopping, ModelCheckpoint\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom tensorflow.keras.metrics import Precision, Recall, AUC\nfrom tensorflow.keras import regularizers\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T19:39:29.409837Z","iopub.execute_input":"2026-03-24T19:39:29.410769Z","iopub.status.idle":"2026-03-24T19:39:29.416067Z","shell.execute_reply.started":"2026-03-24T19:39:29.410737Z","shell.execute_reply":"2026-03-24T19:39:29.415292Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Installer pydicom pour lire les fichiers DICOM\n!pip install pydicom\nimport pydicom\n\nprint(\"✅ pydicom installé\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T19:39:36.287626Z","iopub.execute_input":"2026-03-24T19:39:36.288273Z","iopub.status.idle":"2026-03-24T19:39:39.717320Z","shell.execute_reply.started":"2026-03-24T19:39:36.288239Z","shell.execute_reply":"2026-03-24T19:39:39.716521Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Chargement des données RSNA**","metadata":{}},{"cell_type":"code","source":"# Chemins\nbase_path = \"/kaggle/input/competitions/rsna-pneumonia-detection-challenge\"\ntrain_images_path = os.path.join(base_path, \"stage_2_train_images\")\ntrain_labels_path = os.path.join(base_path, \"stage_2_train_labels.csv\")\n\n# Paramètres\nimg_size = 224\nbatch_size = 32\nrandom.seed(42)\nnp.random.seed(42)\ntf.random.set_seed(42)\n\n# Charger les labels\ndf_labels = pd.read_csv(train_labels_path)\nprint(f\"Total annotations: {len(df_labels)}\")\n\n# Grouper par patient (une image = 1 label)\npatient_labels = df_labels.groupby('patientId')['Target'].max().reset_index()\nlabel_dict = dict(zip(patient_labels['patientId'], patient_labels['Target']))\n\n# Compter les classes\nn_normal = sum(1 for v in label_dict.values() if v == 0)\nn_pneumonia = sum(1 for v in label_dict.values() if v == 1)\nprint(f\"NORMAL: {n_normal}, PNEUMONIA: {n_pneumonia}\")\nprint(f\"Ratio: {n_normal/n_pneumonia:.2f}:1\")\n\n# Visualisation\nplt.figure(figsize=(6,4))\nsns.barplot(x=['NORMAL', 'PNEUMONIA'], y=[n_normal, n_pneumonia])\nplt.title(\"Distribution des classes RSNA\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T19:39:42.358688Z","iopub.execute_input":"2026-03-24T19:39:42.359457Z","iopub.status.idle":"2026-03-24T19:39:42.602303Z","shell.execute_reply.started":"2026-03-24T19:39:42.359418Z","shell.execute_reply":"2026-03-24T19:39:42.601619Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**CRÉATION D'UN DATASET ÉQUILIBRÉ**","metadata":{}},{"cell_type":"code","source":"# ============================================\n# CHARGEMENT AVEC 10000 IMAGES\n# ============================================\n\nimport random\nrandom.seed(42)\nnp.random.seed(42)\ntf.random.set_seed(42)\n\nmax_images = 10000  # ← 10000 images\n\nprint(f\"Chargement de {max_images} images...\")\npatient_ids = list(label_dict.keys())\n# Récupérer les IDs\npositive_ids = [pid for pid in patient_ids if label_dict[pid] == 1]\nnegative_ids = [pid for pid in patient_ids if label_dict[pid] == 0]\n\nprint(f\"Total disponibles - Positifs: {len(positive_ids)}, Négatifs: {len(negative_ids)}\")\n\n# Prendre 5000 de chaque classe (total 10000)\nn_pos = min(len(positive_ids), max_images // 2)  # 5000\nn_neg = min(len(negative_ids), max_images - n_pos)  # 5000\n\npositive_sample = random.sample(positive_ids, n_pos)\nnegative_sample = random.sample(negative_ids, n_neg)\npatient_ids_balanced = positive_sample + negative_sample\nrandom.shuffle(patient_ids_balanced)\n\nprint(f\"Positifs: {len(positive_sample)}, Négatifs: {len(negative_sample)}\")\nprint(f\"Total: {len(patient_ids_balanced)} images\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T19:39:52.465241Z","iopub.execute_input":"2026-03-24T19:39:52.466154Z","iopub.status.idle":"2026-03-24T19:39:52.489917Z","shell.execute_reply.started":"2026-03-24T19:39:52.466109Z","shell.execute_reply":"2026-03-24T19:39:52.489274Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Chargement des images**","metadata":{}},{"cell_type":"code","source":"IMG_SIZE = 224\n\ndef load_rsna_image(patient_id, img_size=IMG_SIZE):\n    dcm_path = os.path.join(train_images_path, f\"{patient_id}.dcm\")\n    if not os.path.exists(dcm_path): return None\n    try:\n        dicom = pydicom.dcmread(dcm_path)\n        img = dicom.pixel_array\n        # 1. Normalisation 8-bit (0-255)\n        img = img.astype(np.float32)\n        img = (img - np.min(img)) / (np.max(img) - np.min(img) + 1e-7) * 255.0\n        img = img.astype(np.uint8)\n        # 2. Resize et CLAHE\n        img = cv2.resize(img, (img_size, img_size))\n        clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8,8))\n        img = clahe.apply(img)\n        # 3. Conversion RGB (3 canaux pour DenseNet)\n        img_rgb = cv2.cvtColor(img, cv2.COLOR_GRAY2RGB)\n        return img_rgb\n    except:\n        return None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T19:40:03.973631Z","iopub.execute_input":"2026-03-24T19:40:03.973931Z","iopub.status.idle":"2026-03-24T19:40:03.980581Z","shell.execute_reply.started":"2026-03-24T19:40:03.973905Z","shell.execute_reply":"2026-03-24T19:40:03.979701Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Chargement des images equilibré**","metadata":{}},{"cell_type":"code","source":"# ============================================\n# CHARGEMENT DES IMAGES (10000 images)\n# ============================================\n\nprint(\"Chargement des images...\")\nX_data = []\ny_data = []\n\nfor i, pid in enumerate(patient_ids_balanced):\n    if i % 500 == 0:\n        print(f\"  {i}/{len(patient_ids_balanced)}\")\n    \n    img = load_rsna_image(pid, 224)\n    if img is not None:\n        X_data.append(img)\n        y_data.append(label_dict[pid])\n\nX_data = np.array(X_data, dtype=np.float32)\ny_data = np.array(y_data)\n\nprint(f\"\\n✅ Chargé: X={X_data.shape}, y={y_data.shape}\")\nprint(f\"Distribution: {np.bincount(y_data)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T19:40:10.690379Z","iopub.execute_input":"2026-03-24T19:40:10.690891Z","iopub.status.idle":"2026-03-24T19:41:49.231565Z","shell.execute_reply.started":"2026-03-24T19:40:10.690843Z","shell.execute_reply":"2026-03-24T19:41:49.230847Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Split et préparation des données**","metadata":{}},{"cell_type":"code","source":"SEED = 42\n\nx_train, x_temp, y_train, y_temp = train_test_split(X_data, y_data, test_size=0.3, stratify=y_data, random_state=SEED)\nx_val, x_test, y_val, y_test = train_test_split(x_temp, y_temp, test_size=0.5, stratify=y_temp, random_state=SEED)\n\n# Normalisation spécifique DenseNet (On ne le fait qu'UNE FOIS ici)\nx_train = preprocess_input(x_train)\nx_val = preprocess_input(x_val)\nx_test = preprocess_input(x_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T19:44:20.464101Z","iopub.execute_input":"2026-03-24T19:44:20.464436Z","iopub.status.idle":"2026-03-24T19:44:27.790977Z","shell.execute_reply.started":"2026-03-24T19:44:20.464409Z","shell.execute_reply":"2026-03-24T19:44:27.790378Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Modèle DenseNet (identique à avant)**","metadata":{}},{"cell_type":"code","source":"base_model = DenseNet121(weights='imagenet', include_top=False, input_shape=(IMG_SIZE, IMG_SIZE, 3))\nbase_model.trainable = False # On gèle le moteur\n\ninputs = layers.Input(shape=(IMG_SIZE, IMG_SIZE, 3))\nx = base_model(inputs, training=False)\nx = layers.GlobalAveragePooling2D()(x)\nx = layers.BatchNormalization()(x)\nx = layers.Dense(512, activation='relu')(x)\nx = layers.Dropout(0.5)(x)\noutputs = layers.Dense(1, activation='sigmoid')(x)\n\nmodel = Model(inputs, outputs)\nmodel.compile(optimizer=Adam(1e-4), loss='binary_crossentropy', metrics=['accuracy'])\n\n# Data Augmentation (sans rescale !)\ndatagen = ImageDataGenerator(rotation_range=15, width_shift_range=0.1, height_shift_range=0.1, horizontal_flip=True)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T19:44:36.111473Z","iopub.execute_input":"2026-03-24T19:44:36.112197Z","iopub.status.idle":"2026-03-24T19:44:38.396983Z","shell.execute_reply.started":"2026-03-24T19:44:36.112165Z","shell.execute_reply":"2026-03-24T19:44:38.396222Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Callbacks et entraînement**","metadata":{}},{"cell_type":"code","source":"# ============================================\n# CALLBACKS OPTIMISÉS\n# ============================================\n\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, EarlyStopping, ModelCheckpoint\n\nreduce_lr = ReduceLROnPlateau(\n    monitor='val_loss',\n    factor=0.5,\n    patience=4,\n    verbose=1,\n    min_lr=1e-7\n)\n\nearly_stop = EarlyStopping(\n    monitor='val_loss',\n    patience=20,\n    restore_best_weights=True,\n    verbose=1\n)\n\ncheckpoint = ModelCheckpoint(\n    'best_cnn_improved.keras',\n    monitor='val_accuracy',\n    save_best_only=True,\n    mode='max',\n    verbose=1\n)\n\nprint(\"✅ Callbacks configurés\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T19:44:47.349858Z","iopub.execute_input":"2026-03-24T19:44:47.350203Z","iopub.status.idle":"2026-03-24T19:44:47.355918Z","shell.execute_reply.started":"2026-03-24T19:44:47.350176Z","shell.execute_reply":"2026-03-24T19:44:47.355119Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**FEATURE EXTRACTION**","metadata":{}},{"cell_type":"code","source":"\nprint(\"=\" * 60)\nprint(\"PHASE 1: Feature Extraction\")\nprint(\"=\" * 60)\n\nhistory_phase1 = model.fit(\n    datagen.flow(x_train, y_train, batch_size=32),\n    validation_data=(x_val, y_val),\n    epochs=30,\n    callbacks=[reduce_lr, early_stop, checkpoint],\n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T19:44:53.244609Z","iopub.execute_input":"2026-03-24T19:44:53.244924Z","iopub.status.idle":"2026-03-24T20:20:23.896967Z","shell.execute_reply.started":"2026-03-24T19:44:53.244893Z","shell.execute_reply":"2026-03-24T20:20:23.896341Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**fine-tuning**","metadata":{}},{"cell_type":"code","source":"print(\"Phase 2 : Fine-Tuning profond...\")\nbase_model.trainable = True\n# On ne dégèle que les 50 dernières couches\nfor layer in base_model.layers[:-50]:\n    layer.trainable = False\n\n# Learning rate très faible obligatoire\nmodel.compile(optimizer=Adam(1e-5), loss='binary_crossentropy', metrics=['accuracy'])\n\ncheckpoint = ModelCheckpoint('best_densenet.keras', save_best_only=True, monitor='val_accuracy')\nearly_stop = EarlyStopping(monitor='val_accuracy', patience=5, restore_best_weights=True)\n\nhistory = model.fit(datagen.flow(x_train, y_train, batch_size=32), \n                    validation_data=(x_val, y_val), \n                    epochs=10, \n                    callbacks=[checkpoint, early_stop])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T20:21:08.484872Z","iopub.execute_input":"2026-03-24T20:21:08.485683Z","iopub.status.idle":"2026-03-24T20:32:42.941949Z","shell.execute_reply.started":"2026-03-24T20:21:08.485651Z","shell.execute_reply":"2026-03-24T20:32:42.941335Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**ÉVALUATION FINALE**","metadata":{}},{"cell_type":"code","source":"# ============================================\n# ÉVALUATION FINALE\n# ============================================\n\nfrom tensorflow.keras.models import load_model\nfrom sklearn.metrics import classification_report, confusion_matrix, roc_curve, auc\n\n# Charger le meilleur modèle (Phase 2)\nbest_model = load_model('best_cnn_improved.keras')\n\n# Évaluation\neval_result = best_model.evaluate(x_test, y_test)\nprint(\"\\n\" + \"=\" * 60)\nprint(\"RÉSULTATS FINAUX (DENSENET + FINE-TUNING)\")\nprint(\"=\" * 60)\nprint(f\"Loss: {eval_result[0]:.4f}\")\nprint(f\"Accuracy: {eval_result[1]*100:.2f}%\")\n\n# Prédictions\ny_pred_proba = best_model.predict(x_test)\ny_pred = (y_pred_proba > 0.5).astype(\"int32\").flatten()\n\n# Rapport\nprint(\"\\n\" + \"=\" * 60)\nprint(\"CLASSIFICATION REPORT\")\nprint(\"=\" * 60)\nprint(classification_report(y_test, y_pred, target_names=['NORMAL', 'PNEUMONIA']))\n\n# Matrice de confusion\ncm = confusion_matrix(y_test, y_pred)\nplt.figure(figsize=(8, 6))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues',\n            xticklabels=['NORMAL', 'PNEUMONIA'],\n            yticklabels=['NORMAL', 'PNEUMONIA'])\nplt.title(f'Matrice de confusion\\nAccuracy: {eval_result[1]*100:.2f}%')\nplt.show()\n\n# Courbe ROC\nfpr, tpr, _ = roc_curve(y_test, y_pred_proba)\nroc_auc = auc(fpr, tpr)\n\nplt.figure(figsize=(8, 6))\nplt.plot(fpr, tpr, 'b-', linewidth=2, label=f'AUC = {roc_auc:.4f}')\nplt.plot([0, 1], [0, 1], 'r--')\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('ROC Curve')\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T20:37:07.144942Z","iopub.execute_input":"2026-03-24T20:37:07.145615Z","iopub.status.idle":"2026-03-24T20:38:04.515335Z","shell.execute_reply.started":"2026-03-24T20:37:07.145583Z","shell.execute_reply":"2026-03-24T20:38:04.514585Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**VISUALISATION DES COURBES**","metadata":{}},{"cell_type":"code","source":"# ============================================\n# VISUALISATION DES COURBES D'APPRENTISSAGE\n# ============================================\n\nfig, axes = plt.subplots(1, 2, figsize=(14, 5))\n\n# Accuracy\nepochs1 = range(len(history_phase1.history['accuracy']))\naxes[0].plot(epochs1, history_phase1.history['accuracy'], 'b-o', label='Train Phase 1')\naxes[0].plot(epochs1, history_phase1.history['val_accuracy'], 'r-s', label='Val Phase 1')\n\nif 'history_phase2' in locals():\n    epochs2 = range(len(history_phase2.history['accuracy']))\n    axes[0].plot([e + epochs1[-1] + 1 for e in epochs2], \n                 history_phase2.history['accuracy'], 'b--o', label='Train Phase 2')\n    axes[0].plot([e + epochs1[-1] + 1 for e in epochs2], \n                 history_phase2.history['val_accuracy'], 'r--s', label='Val Phase 2')\n\naxes[0].set_title('Accuracy')\naxes[0].set_xlabel('Epochs')\naxes[0].set_ylabel('Accuracy')\naxes[0].legend()\naxes[0].grid(True)\n\n# Loss\naxes[1].plot(epochs1, history_phase1.history['loss'], 'b-o', label='Train Phase 1')\naxes[1].plot(epochs1, history_phase1.history['val_loss'], 'r-s', label='Val Phase 1')\n\nif 'history_phase2' in locals():\n    axes[1].plot([e + epochs1[-1] + 1 for e in epochs2], \n                 history_phase2.history['loss'], 'b--o', label='Train Phase 2')\n    axes[1].plot([e + epochs1[-1] + 1 for e in epochs2], \n                 history_phase2.history['val_loss'], 'r--s', label='Val Phase 2')\n\naxes[1].set_title('Loss')\naxes[1].set_xlabel('Epochs')\naxes[1].set_ylabel('Loss')\naxes[1].legend()\naxes[1].grid(True)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T20:34:19.707902Z","iopub.execute_input":"2026-03-24T20:34:19.708642Z","iopub.status.idle":"2026-03-24T20:34:20.000300Z","shell.execute_reply.started":"2026-03-24T20:34:19.708608Z","shell.execute_reply":"2026-03-24T20:34:19.999724Z"}},"outputs":[],"execution_count":null}]}