{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":29653,"databundleVersionId":2420395,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport torch\nimport torch.nn as nn\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib\nimport pydicom as dicom\nimport cv2\nimport ast\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:22:09.760173Z","iopub.execute_input":"2026-07-11T11:22:09.760596Z","iopub.status.idle":"2026-07-11T11:22:14.424077Z","shell.execute_reply.started":"2026-07-11T11:22:09.760571Z","shell.execute_reply":"2026-07-11T11:22:14.423209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/'\nos.listdir(path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:22:14.425402Z","iopub.execute_input":"2026-07-11T11:22:14.425829Z","iopub.status.idle":"2026-07-11T11:22:14.431580Z","shell.execute_reply.started":"2026-07-11T11:22:14.425810Z","shell.execute_reply":"2026-07-11T11:22:14.430721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/'\nos.listdir(path)\ntrain_data = pd.read_csv(path+'train_labels.csv')\nsamp_subm = pd.read_csv(path+'sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:22:14.432449Z","iopub.execute_input":"2026-07-11T11:22:14.432725Z","iopub.status.idle":"2026-07-11T11:22:14.462465Z","shell.execute_reply.started":"2026-07-11T11:22:14.432704Z","shell.execute_reply":"2026-07-11T11:22:14.461926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('Samples train:', len(train_data))\nprint('Samples test:', len(samp_subm))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:22:14.463118Z","iopub.execute_input":"2026-07-11T11:22:14.463346Z","iopub.status.idle":"2026-07-11T11:22:14.467745Z","shell.execute_reply.started":"2026-07-11T11:22:14.463324Z","shell.execute_reply":"2026-07-11T11:22:14.467248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:22:14.469562Z","iopub.execute_input":"2026-07-11T11:22:14.469787Z","iopub.status.idle":"2026-07-11T11:22:14.492957Z","shell.execute_reply.started":"2026-07-11T11:22:14.469771Z","shell.execute_reply":"2026-07-11T11:22:14.492414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data[\"MGMT_value\"].value_counts().head(2).plot(kind = 'pie', autopct='%1.1f%%', figsize=(8, 8)).legend()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:22:14.493650Z","iopub.execute_input":"2026-07-11T11:22:14.493918Z","iopub.status.idle":"2026-07-11T11:22:14.846249Z","shell.execute_reply.started":"2026-07-11T11:22:14.493895Z","shell.execute_reply":"2026-07-11T11:22:14.845384Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data[\"MGMT_value\"].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:22:14.847116Z","iopub.execute_input":"2026-07-11T11:22:14.847389Z","iopub.status.idle":"2026-07-11T11:22:14.854617Z","shell.execute_reply.started":"2026-07-11T11:22:14.847371Z","shell.execute_reply":"2026-07-11T11:22:14.853898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"samp_subm.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:22:14.855544Z","iopub.execute_input":"2026-07-11T11:22:14.855969Z","iopub.status.idle":"2026-07-11T11:22:14.869805Z","shell.execute_reply.started":"2026-07-11T11:22:14.855948Z","shell.execute_reply":"2026-07-11T11:22:14.868926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"folder = str(train_data.loc[0, 'BraTS21ID']).zfill(5)\nfolder","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:22:14.870417Z","iopub.execute_input":"2026-07-11T11:22:14.870635Z","iopub.status.idle":"2026-07-11T11:22:14.881427Z","shell.execute_reply.started":"2026-07-11T11:22:14.870621Z","shell.execute_reply":"2026-07-11T11:22:14.880596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.listdir(path+'train/'+folder)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:22:14.882286Z","iopub.execute_input":"2026-07-11T11:22:14.882528Z","iopub.status.idle":"2026-07-11T11:22:14.897161Z","shell.execute_reply.started":"2026-07-11T11:22:14.882505Z","shell.execute_reply":"2026-07-11T11:22:14.896337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('Number of FLAIR images:', len(os.listdir(path+'train/'+folder+'/'+'FLAIR')))\nprint('Number of T1w images:', len(os.listdir(path+'train/'+folder+'/'+'T1w')))\nprint('Number of T1wCE images:', len(os.listdir(path+'train/'+folder+'/'+'T1wCE')))\nprint('Number of T2w images:', len(os.listdir(path+'train/'+folder+'/'+'T2w')))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:22:14.898075Z","iopub.execute_input":"2026-07-11T11:22:14.898334Z","iopub.status.idle":"2026-07-11T11:22:14.941997Z","shell.execute_reply.started":"2026-07-11T11:22:14.898314Z","shell.execute_reply":"2026-07-11T11:22:14.941452Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path_file = ''.join([path, 'train/', folder, '/', 'FLAIR/'])\nimage = os.listdir(path_file)[0]\ndata_file = dicom.dcmread(path_file+image)\nimg = data_file.pixel_array\nprint('Image shape:', img.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:22:14.942666Z","iopub.execute_input":"2026-07-11T11:22:14.942955Z","iopub.status.idle":"2026-07-11T11:22:14.959620Z","shell.execute_reply.started":"2026-07-11T11:22:14.942934Z","shell.execute_reply":"2026-07-11T11:22:14.959059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Flair Image\ndef plot_examples(row = 0, cat = 'FLAIR'): \n    folder = str(train_data.loc[row, 'BraTS21ID']).zfill(5)\n    path_file = ''.join([path, 'train/', folder, '/', cat, '/'])\n    images = os.listdir(path_file)\n    \n    fig, axs = plt.subplots(1, 5, figsize=(30, 30))\n    fig.subplots_adjust(hspace = .2, wspace=.2)\n    axs = axs.ravel()\n    \n    for num in range(5):\n        data_file = dicom.dcmread(path_file+images[num])\n        img = data_file.pixel_array\n        axs[num].imshow(img, cmap='gray')\n        axs[num].set_title(cat+' '+images[num])\n        axs[num].set_xticklabels([])\n        axs[num].set_yticklabels([])\n        \nrow = 0\nplot_examples(row = row, cat = 'FLAIR')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:22:14.960365Z","iopub.execute_input":"2026-07-11T11:22:14.960599Z","iopub.status.idle":"2026-07-11T11:22:15.691229Z","shell.execute_reply.started":"2026-07-11T11:22:14.960579Z","shell.execute_reply":"2026-07-11T11:22:15.690465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#T1w Images\nplot_examples(row = row, cat = 'T1w')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:22:15.694923Z","iopub.execute_input":"2026-07-11T11:22:15.695243Z","iopub.status.idle":"2026-07-11T11:22:16.388038Z","shell.execute_reply.started":"2026-07-11T11:22:15.695224Z","shell.execute_reply":"2026-07-11T11:22:16.387074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#T1wCE Images\nplot_examples(row = row, cat = 'T1wCE')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:22:16.388896Z","iopub.execute_input":"2026-07-11T11:22:16.389187Z","iopub.status.idle":"2026-07-11T11:22:17.100505Z","shell.execute_reply.started":"2026-07-11T11:22:16.389163Z","shell.execute_reply":"2026-07-11T11:22:17.099585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#T2w Images\nplot_examples(row = row, cat = 'T2w')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:22:17.101387Z","iopub.execute_input":"2026-07-11T11:22:17.101676Z","iopub.status.idle":"2026-07-11T11:22:17.816472Z","shell.execute_reply.started":"2026-07-11T11:22:17.101659Z","shell.execute_reply":"2026-07-11T11:22:17.815833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom sklearn.model_selection import train_test_split\nimport cv2\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import classification_report, confusion_matrix, roc_curve, auc, accuracy_score\n\n# =====================================================\n# Path dataset\n# =====================================================\npath = '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/'\ntrain_labels = pd.read_csv(path + 'train_labels.csv')\n\n# =====================================================\n# Configuration\n# =====================================================\nIMG_SIZE = 128\nBATCH_SIZE = 16\nEPOCHS = 40\nMODALITY = 'FLAIR'\n\n\n# =====================================================\n# Lecture et prétraitement d'une image DICOM\n# =====================================================\ndef load_dicom_image(filepath, img_size=IMG_SIZE):\n\n    dicom = pydicom.dcmread(filepath)\n\n    img = dicom.pixel_array.astype(np.float32)\n\n    # -------------------------------------------------\n    # Normalisation robuste (percentiles)\n    # -------------------------------------------------\n    p1 = np.percentile(img, 1)\n    p99 = np.percentile(img, 99)\n\n    img = np.clip(img, p1, p99)\n    img = (img - p1) / (p99 - p1 + 1e-8)\n\n    # -------------------------------------------------\n    # CLAHE\n    # -------------------------------------------------\n    img = (img * 255).astype(np.uint8)\n\n    clahe = cv2.createCLAHE(\n        clipLimit=2.0,\n        tileGridSize=(8, 8)\n    )\n\n    img = clahe.apply(img)\n\n    # -------------------------------------------------\n    # Resize\n    # -------------------------------------------------\n    img = cv2.resize(img, (img_size, img_size))\n\n    # -------------------------------------------------\n    # Retour float\n    # -------------------------------------------------\n    img = img.astype(np.float32) / 255.0\n\n    # -------------------------------------------------\n    # RGB\n    # -------------------------------------------------\n    img = np.stack([img] * 3, axis=-1)\n\n    return img\n\n\n# =====================================================\n# Data Augmentation\n# =====================================================\ndef augment_image(img):\n\n    # Flip horizontal\n    if np.random.rand() < 0.5:\n        img = cv2.flip(img, 1)\n\n    # Rotation\n    if np.random.rand() < 0.3:\n\n        angle = np.random.uniform(-10, 10)\n\n        M = cv2.getRotationMatrix2D(\n            (IMG_SIZE // 2, IMG_SIZE // 2),\n            angle,\n            1\n        )\n\n        img = cv2.warpAffine(\n            img,\n            M,\n            (IMG_SIZE, IMG_SIZE)\n        )\n\n    # -------------------------------------------------\n    # Zoom aléatoire\n    # -------------------------------------------------\n    if np.random.rand() < 0.3:\n\n        scale = np.random.uniform(0.9, 1.1)\n\n        h, w = img.shape[:2]\n\n        M = cv2.getRotationMatrix2D(\n            (w // 2, h // 2),\n            0,\n            scale\n        )\n\n        img = cv2.warpAffine(\n            img,\n            M,\n            (w, h)\n        )\n\n    # Luminosité / Contraste\n    if np.random.rand() < 0.3:\n\n        alpha = np.random.uniform(0.9, 1.1)\n        beta = np.random.uniform(-10, 10)\n\n        img = cv2.convertScaleAbs(\n            img,\n            alpha=alpha,\n            beta=beta\n        )\n\n    return img\n\n\n# =====================================================\n# Chargement des données d'un patient\n# =====================================================\ndef load_patient_data(patient_id, num_slices=16):\n\n    patient_path = os.path.join(\n        path,\n        \"train\",\n        str(patient_id).zfill(5),\n        MODALITY\n    )\n\n    if not os.path.exists(patient_path):\n        print(f\"Patient {patient_id} introuvable\")\n        return None\n\n    # -------------------------------------------------\n    # Liste des DICOM\n    # -------------------------------------------------\n    dicom_files = sorted([\n        f for f in os.listdir(patient_path)\n        if f.endswith(\".dcm\")\n    ])\n\n    if len(dicom_files) == 0:\n        print(f\"Aucun fichier DICOM pour le patient {patient_id}\")\n        return None\n\n    # =====================================================\n    # Sélection des slices centrales\n    # =====================================================\n    center = len(dicom_files) // 2\n    half = num_slices // 2\n\n    start = max(0, center - half)\n    end = start + num_slices\n\n    selected_files = dicom_files[start:end]\n\n    while len(selected_files) < num_slices:\n        selected_files.append(selected_files[-1])\n\n    # =====================================================\n    # Chargement des images\n    # =====================================================\n    slices = []\n\n    for filename in selected_files:\n\n        img_path = os.path.join(patient_path, filename)\n\n        img = load_dicom_image(img_path)\n\n        # Data augmentation\n        if np.random.rand() < 0.5:\n\n            gray = img[:, :, 0]\n\n            gray = augment_image(gray)\n\n            img = np.stack([gray] * 3, axis=-1)\n\n        slices.append(img)\n\n    return np.array(slices, dtype=np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:22:17.817224Z","iopub.execute_input":"2026-07-11T11:22:17.817403Z","iopub.status.idle":"2026-07-11T11:22:30.107341Z","shell.execute_reply.started":"2026-07-11T11:22:17.817389Z","shell.execute_reply":"2026-07-11T11:22:30.106696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Membuat dataset\nX = []\ny = []\n\nprint(\"Memuat data training...\")\nfor idx, row in train_labels.iterrows():\n    patient_id = row['BraTS21ID']\n    label = row['MGMT_value']\n    \n    patient_data = load_patient_data(patient_id)\n    if patient_data is not None:\n        X.append(patient_data)\n        y.append(label)\n\n# Konversi ke numpy array\nX = np.array(X, dtype=np.float32)\ny = np.array(y, dtype=np.float32)\n\nprint(f\"Total data yang dimuat: {len(X)} sampel\")\nprint(f\"Distribusi kelas: {np.sum(y == 1)} positif, {np.sum(y == 0)} negatif\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:22:30.108117Z","iopub.execute_input":"2026-07-11T11:22:30.108431Z","iopub.status.idle":"2026-07-11T11:24:24.691553Z","shell.execute_reply.started":"2026-07-11T11:22:30.108415Z","shell.execute_reply":"2026-07-11T11:24:24.690813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split data: training (60%), validation (20%), test (20%)\nX_train, X_temp, y_train, y_temp = train_test_split(\n    X, y, test_size=0.4, random_state=42, stratify=y\n)\nX_val, X_test, y_val, y_test = train_test_split(\n    X_temp, y_temp, test_size=0.5, random_state=42, stratify=y_temp\n)\n# Calcul des poids des classes                                                        33\nfrom sklearn.utils.class_weight import compute_class_weight\n\nweights = compute_class_weight(\n    class_weight='balanced',\n    classes=np.unique(y_train),\n    y=y_train\n)\n\nclass_weight = {\n    0: weights[0],\n    1: weights[1]\n}\n\nprint(\"Class Weights :\", class_weight)\n\nprint(\"\\nDistribusi dataset:\")\nprint(f\"Training:   {len(X_train)} sampel\")\nprint(f\"Validation: {len(X_val)} sampel\")\nprint(f\"Test:       {len(X_test)} sampel\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:24:24.692365Z","iopub.execute_input":"2026-07-11T11:24:24.692666Z","iopub.status.idle":"2026-07-11T11:24:25.285380Z","shell.execute_reply.started":"2026-07-11T11:24:24.692642Z","shell.execute_reply":"2026-07-11T11:24:25.284488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras import regularizers\n\n# =====================================================\n# Architecture CNN 3D améliorée\n# =====================================================\n\ndef build_3d_cnn(input_shape, num_classes=1):\n\n    model = models.Sequential([\n\n        # =====================================================\n        # Bloc 1\n        # =====================================================\n        layers.Conv3D(\n            filters=32,\n            kernel_size=(3,3,3),\n            activation='relu',\n            padding='same',\n            kernel_initializer='he_normal',\n            kernel_regularizer=regularizers.l2(1e-4),\n            input_shape=input_shape\n        ),\n\n        layers.BatchNormalization(),\n\n        layers.MaxPooling3D(\n            pool_size=(2,2,2),\n            padding='same'\n        ),\n\n        layers.Dropout(0.20),\n\n\n        # =====================================================\n        # Bloc 2\n        # =====================================================\n        layers.Conv3D(\n            filters=64,\n            kernel_size=(3,3,3),\n            activation='relu',\n            padding='same',\n            kernel_initializer='he_normal',\n            kernel_regularizer=regularizers.l2(1e-4)\n        ),\n\n        layers.BatchNormalization(),\n\n        layers.MaxPooling3D(\n            pool_size=(2,2,2),\n            padding='same'\n        ),\n\n        layers.Dropout(0.30),\n\n\n        # =====================================================\n        # Bloc 3\n        # =====================================================\n        layers.Conv3D(\n            filters=128,\n            kernel_size=(3,3,3),\n            activation='relu',\n            padding='same',\n            kernel_initializer='he_normal',\n            kernel_regularizer=regularizers.l2(1e-4)\n        ),\n\n        layers.BatchNormalization(),\n\n        layers.MaxPooling3D(\n            pool_size=(2,2,2),\n            padding='same'\n        ),\n\n        layers.Dropout(0.40),\n\n\n        # =====================================================\n        # Bloc 4\n        # =====================================================\n        layers.Conv3D(\n            filters=256,\n            kernel_size=(3,3,3),\n            activation='relu',\n            padding='same',\n            kernel_initializer='he_normal',\n            kernel_regularizer=regularizers.l2(1e-4)\n        ),\n\n        layers.BatchNormalization(),\n\n        layers.MaxPooling3D(\n            pool_size=(2,2,2),\n            padding='same'\n        ),\n\n        layers.Dropout(0.50),\n\n\n        # =====================================================\n        # Classification\n        # =====================================================\n        layers.GlobalAveragePooling3D(),\n\n        layers.Dense(\n            128,\n            activation='relu',\n            kernel_initializer='he_normal',\n            kernel_regularizer=regularizers.l2(1e-4)\n        ),\n\n        layers.Dropout(0.50),\n\n        layers.Dense(\n            64,\n            activation='relu',\n            kernel_initializer='he_normal',\n            kernel_regularizer=regularizers.l2(1e-4)\n        ),\n\n        layers.Dropout(0.40),\n\n        layers.Dense(\n            num_classes,\n            activation='sigmoid'\n        )\n\n    ])\n\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:24:25.286321Z","iopub.execute_input":"2026-07-11T11:24:25.286640Z","iopub.status.idle":"2026-07-11T11:24:25.297060Z","shell.execute_reply.started":"2026-07-11T11:24:25.286613Z","shell.execute_reply":"2026-07-11T11:24:25.296267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Bangun model\ninput_shape = (X_train.shape[1], X_train.shape[2], X_train.shape[3], X_train.shape[4])\nprint(f\"\\nInput shape: {input_shape}\")\nmodel = build_3d_cnn(input_shape, num_classes=1)\n\n# Ringkasan model\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:24:25.297974Z","iopub.execute_input":"2026-07-11T11:24:25.298309Z","iopub.status.idle":"2026-07-11T11:24:27.583805Z","shell.execute_reply.started":"2026-07-11T11:24:25.298290Z","shell.execute_reply":"2026-07-11T11:24:27.583198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================================================\n# Compilation du modèle\n# =====================================================\n\nfrom tensorflow.keras.optimizers import AdamW\n\nmodel.compile(\n\n    optimizer=AdamW(\n        learning_rate=2e-4,\n        weight_decay=1e-5\n    ),\n\n        loss=tf.keras.losses.BinaryFocalCrossentropy(\n        gamma=2\n    ),\n\n        metrics=[\n        tf.keras.metrics.BinaryAccuracy(name='accuracy'),\n        tf.keras.metrics.AUC(name='auc'),\n        tf.keras.metrics.Precision(name='precision'),\n        tf.keras.metrics.Recall(name='recall')\n    ]\n)\n\n# =====================================================\n# Learning Rate Scheduler\n# =====================================================\n\nlr_scheduler = tf.keras.callbacks.LearningRateScheduler(\n\n    lambda epoch: 2e-4 * (0.95 ** epoch),\n\n    verbose=0\n)\n\n# =====================================================\n# Callbacks\n# =====================================================\n\ncallbacks = [\n\n    tf.keras.callbacks.EarlyStopping(\n        monitor='val_auc',\n        mode='max',\n        patience=12,\n        restore_best_weights=True,\n        verbose=1\n    ),\n\n    tf.keras.callbacks.ReduceLROnPlateau(\n        monitor='val_loss',\n        factor=0.5,\n        patience=2,\n        min_lr=1e-6,\n        verbose=1\n    ),\n\n    tf.keras.callbacks.ModelCheckpoint(\n        filepath='best_model.h5',\n        monitor='val_auc',\n        mode='max',\n        save_best_only=True,\n        verbose=1\n    ),\n\n    lr_scheduler\n\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:24:27.584522Z","iopub.execute_input":"2026-07-11T11:24:27.585566Z","iopub.status.idle":"2026-07-11T11:24:27.620356Z","shell.execute_reply.started":"2026-07-11T11:24:27.585549Z","shell.execute_reply":"2026-07-11T11:24:27.619536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Training\nprint(\"\\nMemulai training...\")\nhistory = model.fit(\n    X_train, y_train,\n    validation_data=(X_val, y_val),\n    batch_size=BATCH_SIZE,\n    epochs=EPOCHS,\n    callbacks=callbacks,\n    #on ajoute class_weight                                                          44\n    class_weight=class_weight\n)\n\n# Plot history training\ndef plot_history(history):\n    plt.figure(figsize=(12, 5))\n    \n    # Plot loss\n    plt.subplot(1, 2, 1)\n    plt.plot(history.history['loss'], label='Training Loss')\n    plt.plot(history.history['val_loss'], label='Validation Loss')\n    plt.title('Training and Validation Loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.legend()\n    \n    # Plot AUC\n    plt.subplot(1, 2, 2)\n    plt.plot(history.history['auc'], label='Training AUC')\n    plt.plot(history.history['val_auc'], label='Validation AUC')\n    plt.title('Training and Validation AUC')\n    plt.xlabel('Epoch')\n    plt.ylabel('AUC')\n    plt.legend()\n    \n    plt.tight_layout()\n    plt.savefig('training_history.png')\n    plt.show()\n\nplot_history(history)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:24:27.621166Z","iopub.execute_input":"2026-07-11T11:24:27.621727Z","iopub.status.idle":"2026-07-11T11:27:10.075514Z","shell.execute_reply.started":"2026-07-11T11:24:27.621702Z","shell.execute_reply":"2026-07-11T11:27:10.074892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================================================\n# Évaluation sur le jeu de validation\n# =====================================================\n\nprint(\"\\nÉvaluation sur le jeu de validation :\")\n\nval_loss, val_acc, val_auc, val_precision, val_recall = model.evaluate(\n    X_val,\n    y_val,\n    verbose=1\n)\n\nprint(f\"Validation Loss      : {val_loss:.4f}\")\nprint(f\"Validation Accuracy  : {val_acc:.4f}\")\nprint(f\"Validation AUC       : {val_auc:.4f}\")\nprint(f\"Validation Precision : {val_precision:.4f}\")\nprint(f\"Validation Recall    : {val_recall:.4f}\")\n\n\n# =====================================================\n# Évaluation sur le jeu de test\n# =====================================================\n\nprint(\"\\nÉvaluation sur le jeu de test :\")\n\ntest_loss, test_acc, test_auc, test_precision, test_recall = model.evaluate(\n    X_test,\n    y_test,\n    verbose=1\n)\n\nprint(f\"Test Loss      : {test_loss:.4f}\")\nprint(f\"Test Accuracy  : {test_acc:.4f}\")\nprint(f\"Test AUC       : {test_auc:.4f}\")\nprint(f\"Test Precision : {test_precision:.4f}\")\nprint(f\"Test Recall    : {test_recall:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:29:03.260900Z","iopub.execute_input":"2026-07-11T11:29:03.261687Z","iopub.status.idle":"2026-07-11T11:29:06.574193Z","shell.execute_reply.started":"2026-07-11T11:29:03.261663Z","shell.execute_reply":"2026-07-11T11:29:06.573283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================================================\n# Prédictions sur le jeu de test\n# =====================================================\n\n# Probabilités prédites\ny_pred_prob = model.predict(X_test).flatten()\n\n# =====================================================\n# Calcul du meilleur seuil (Indice de Youden)\n# =====================================================\n\nfpr, tpr, thresholds = roc_curve(y_test, y_pred_prob)\n\nyouden_index = tpr - fpr\n\nbest_threshold = thresholds[np.argmax(youden_index)]\n\nprint(f\"Meilleur seuil (Youden) : {best_threshold:.3f}\")\n\n# =====================================================\n# Sauvegarde du meilleur seuil\n# =====================================================\n\nnp.save(\"best_threshold.npy\", best_threshold)\n\nprint(\"Le meilleur seuil a été sauvegardé dans : best_threshold.npy\")\n\n# =====================================================\n# Classification avec le meilleur seuil\n# =====================================================\n\ny_pred = (y_pred_prob > best_threshold).astype(int)\n\n# =====================================================\n# Accuracy\n# =====================================================\n\naccuracy = accuracy_score(y_test, y_pred)\n\nprint(f\"Accuracy : {accuracy:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:29:39.094183Z","iopub.execute_input":"2026-07-11T11:29:39.094851Z","iopub.status.idle":"2026-07-11T11:29:42.297982Z","shell.execute_reply.started":"2026-07-11T11:29:39.094825Z","shell.execute_reply":"2026-07-11T11:29:42.297403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Classification report\nprint(\"\\nClassification Report:\")\nprint(classification_report(y_test, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:29:56.182779Z","iopub.execute_input":"2026-07-11T11:29:56.183529Z","iopub.status.idle":"2026-07-11T11:29:56.199470Z","shell.execute_reply.started":"2026-07-11T11:29:56.183504Z","shell.execute_reply":"2026-07-11T11:29:56.198119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Confusion matrix\nconf_matrix = confusion_matrix(y_test, y_pred)\nprint(\"\\nConfusion Matrix:\")\nprint(conf_matrix)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:30:02.048086Z","iopub.execute_input":"2026-07-11T11:30:02.048803Z","iopub.status.idle":"2026-07-11T11:30:02.059390Z","shell.execute_reply.started":"2026-07-11T11:30:02.048777Z","shell.execute_reply":"2026-07-11T11:30:02.058410Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import precision_recall_curve","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:33:23.884956Z","iopub.execute_input":"2026-07-11T11:33:23.885652Z","iopub.status.idle":"2026-07-11T11:33:23.889227Z","shell.execute_reply.started":"2026-07-11T11:33:23.885629Z","shell.execute_reply":"2026-07-11T11:33:23.888553Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"precision, recall, _ = precision_recall_curve(\n    y_test,\n    y_pred_prob\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:33:51.383715Z","iopub.execute_input":"2026-07-11T11:33:51.384547Z","iopub.status.idle":"2026-07-11T11:33:51.390360Z","shell.execute_reply.started":"2026-07-11T11:33:51.384523Z","shell.execute_reply":"2026-07-11T11:33:51.389331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================================================\n# Courbe Precision-Recall\n# =====================================================\n\nprecision, recall, _ = precision_recall_curve(\n    y_test,\n    y_pred_prob\n)\n\nplt.figure(figsize=(6,5))\nplt.plot(recall, precision, color='green', linewidth=2)\n\nplt.xlabel(\"Recall\")\nplt.ylabel(\"Precision\")\nplt.title(\"Precision-Recall Curve\")\n\nplt.grid(True)\n\nplt.savefig(\"precision_recall_curve.png\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:33:53.673435Z","iopub.execute_input":"2026-07-11T11:33:53.673923Z","iopub.status.idle":"2026-07-11T11:33:53.902883Z","shell.execute_reply.started":"2026-07-11T11:33:53.673902Z","shell.execute_reply":"2026-07-11T11:33:53.902164Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot ROC curve\nfpr, tpr, thresholds = roc_curve(y_test, y_pred_prob)\nroc_auc = auc(fpr, tpr)\n\nplt.figure()\nplt.plot(fpr, tpr, color='darkorange', lw=2, label=f'ROC curve (area = {roc_auc:.2f})')\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic')\nplt.legend(loc=\"lower right\")\nplt.savefig('roc_curve.png')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:34:08.857414Z","iopub.execute_input":"2026-07-11T11:34:08.857947Z","iopub.status.idle":"2026-07-11T11:34:09.108316Z","shell.execute_reply.started":"2026-07-11T11:34:08.857924Z","shell.execute_reply":"2026-07-11T11:34:09.107423Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Simpan model akhir\nmodel.save('final_model.h5')\nprint(\"\\nModel akhir disimpan sebagai 'final_model.h5'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:34:15.562983Z","iopub.execute_input":"2026-07-11T11:34:15.563582Z","iopub.status.idle":"2026-07-11T11:34:15.653409Z","shell.execute_reply.started":"2026-07-11T11:34:15.563561Z","shell.execute_reply":"2026-07-11T11:34:15.652511Z"}},"outputs":[],"execution_count":null}]}