{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":29653,"databundleVersionId":2420395,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport torch\nimport torch.nn as nn\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib\nimport pydicom as dicom\nimport cv2\nimport ast\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:15:27.407112Z","iopub.execute_input":"2026-07-11T10:15:27.407696Z","iopub.status.idle":"2026-07-11T10:15:27.411813Z","shell.execute_reply.started":"2026-07-11T10:15:27.407672Z","shell.execute_reply":"2026-07-11T10:15:27.411083Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/'\nos.listdir(path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:15:27.417590Z","iopub.execute_input":"2026-07-11T10:15:27.418110Z","iopub.status.idle":"2026-07-11T10:15:27.431110Z","shell.execute_reply.started":"2026-07-11T10:15:27.418091Z","shell.execute_reply":"2026-07-11T10:15:27.430376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/'\nos.listdir(path)\ntrain_data = pd.read_csv(path+'train_labels.csv')\nsamp_subm = pd.read_csv(path+'sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:15:27.458845Z","iopub.execute_input":"2026-07-11T10:15:27.459317Z","iopub.status.idle":"2026-07-11T10:15:27.468625Z","shell.execute_reply.started":"2026-07-11T10:15:27.459301Z","shell.execute_reply":"2026-07-11T10:15:27.467993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('Samples train:', len(train_data))\nprint('Samples test:', len(samp_subm))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:15:27.470131Z","iopub.execute_input":"2026-07-11T10:15:27.470349Z","iopub.status.idle":"2026-07-11T10:15:27.474600Z","shell.execute_reply.started":"2026-07-11T10:15:27.470334Z","shell.execute_reply":"2026-07-11T10:15:27.473990Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:15:27.475320Z","iopub.execute_input":"2026-07-11T10:15:27.475561Z","iopub.status.idle":"2026-07-11T10:15:27.488551Z","shell.execute_reply.started":"2026-07-11T10:15:27.475537Z","shell.execute_reply":"2026-07-11T10:15:27.487734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data[\"MGMT_value\"].value_counts().head(2).plot(kind = 'pie', autopct='%1.1f%%', figsize=(8, 8)).legend()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:15:27.490144Z","iopub.execute_input":"2026-07-11T10:15:27.490330Z","iopub.status.idle":"2026-07-11T10:15:27.602707Z","shell.execute_reply.started":"2026-07-11T10:15:27.490317Z","shell.execute_reply":"2026-07-11T10:15:27.601889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data[\"MGMT_value\"].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:15:27.603543Z","iopub.execute_input":"2026-07-11T10:15:27.603803Z","iopub.status.idle":"2026-07-11T10:15:27.609966Z","shell.execute_reply.started":"2026-07-11T10:15:27.603781Z","shell.execute_reply":"2026-07-11T10:15:27.609295Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"samp_subm.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:15:27.610844Z","iopub.execute_input":"2026-07-11T10:15:27.611202Z","iopub.status.idle":"2026-07-11T10:15:27.624755Z","shell.execute_reply.started":"2026-07-11T10:15:27.611169Z","shell.execute_reply":"2026-07-11T10:15:27.624037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"folder = str(train_data.loc[0, 'BraTS21ID']).zfill(5)\nfolder","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:15:27.625317Z","iopub.execute_input":"2026-07-11T10:15:27.625530Z","iopub.status.idle":"2026-07-11T10:15:27.635502Z","shell.execute_reply.started":"2026-07-11T10:15:27.625508Z","shell.execute_reply":"2026-07-11T10:15:27.634929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.listdir(path+'train/'+folder)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:15:27.636377Z","iopub.execute_input":"2026-07-11T10:15:27.636811Z","iopub.status.idle":"2026-07-11T10:15:27.649119Z","shell.execute_reply.started":"2026-07-11T10:15:27.636795Z","shell.execute_reply":"2026-07-11T10:15:27.648486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('Number of FLAIR images:', len(os.listdir(path+'train/'+folder+'/'+'FLAIR')))\nprint('Number of T1w images:', len(os.listdir(path+'train/'+folder+'/'+'T1w')))\nprint('Number of T1wCE images:', len(os.listdir(path+'train/'+folder+'/'+'T1wCE')))\nprint('Number of T2w images:', len(os.listdir(path+'train/'+folder+'/'+'T2w')))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:15:27.650031Z","iopub.execute_input":"2026-07-11T10:15:27.650294Z","iopub.status.idle":"2026-07-11T10:15:27.657682Z","shell.execute_reply.started":"2026-07-11T10:15:27.650279Z","shell.execute_reply":"2026-07-11T10:15:27.657039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path_file = ''.join([path, 'train/', folder, '/', 'FLAIR/'])\nimage = os.listdir(path_file)[0]\ndata_file = dicom.dcmread(path_file+image)\nimg = data_file.pixel_array\nprint('Image shape:', img.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:15:27.659698Z","iopub.execute_input":"2026-07-11T10:15:27.659869Z","iopub.status.idle":"2026-07-11T10:15:27.669436Z","shell.execute_reply.started":"2026-07-11T10:15:27.659856Z","shell.execute_reply":"2026-07-11T10:15:27.668623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Flair Image\ndef plot_examples(row = 0, cat = 'FLAIR'): \n    folder = str(train_data.loc[row, 'BraTS21ID']).zfill(5)\n    path_file = ''.join([path, 'train/', folder, '/', cat, '/'])\n    images = os.listdir(path_file)\n    \n    fig, axs = plt.subplots(1, 5, figsize=(30, 30))\n    fig.subplots_adjust(hspace = .2, wspace=.2)\n    axs = axs.ravel()\n    \n    for num in range(5):\n        data_file = dicom.dcmread(path_file+images[num])\n        img = data_file.pixel_array\n        axs[num].imshow(img, cmap='gray')\n        axs[num].set_title(cat+' '+images[num])\n        axs[num].set_xticklabels([])\n        axs[num].set_yticklabels([])\n        \nrow = 0\nplot_examples(row = row, cat = 'FLAIR')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:15:27.670041Z","iopub.execute_input":"2026-07-11T10:15:27.670248Z","iopub.status.idle":"2026-07-11T10:15:28.280819Z","shell.execute_reply.started":"2026-07-11T10:15:27.670235Z","shell.execute_reply":"2026-07-11T10:15:28.279907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#T1w Images\nplot_examples(row = row, cat = 'T1w')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:15:28.281634Z","iopub.execute_input":"2026-07-11T10:15:28.281895Z","iopub.status.idle":"2026-07-11T10:15:28.873439Z","shell.execute_reply.started":"2026-07-11T10:15:28.281879Z","shell.execute_reply":"2026-07-11T10:15:28.872663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#T1wCE Images\nplot_examples(row = row, cat = 'T1wCE')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:15:28.874205Z","iopub.execute_input":"2026-07-11T10:15:28.874460Z","iopub.status.idle":"2026-07-11T10:15:29.485541Z","shell.execute_reply.started":"2026-07-11T10:15:28.874437Z","shell.execute_reply":"2026-07-11T10:15:29.484674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#T2w Images\nplot_examples(row = row, cat = 'T2w')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:15:29.486484Z","iopub.execute_input":"2026-07-11T10:15:29.486731Z","iopub.status.idle":"2026-07-11T10:15:30.093521Z","shell.execute_reply.started":"2026-07-11T10:15:29.486706Z","shell.execute_reply":"2026-07-11T10:15:30.092873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom sklearn.model_selection import train_test_split\nimport cv2\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import classification_report, confusion_matrix, roc_curve, auc\n\n# Path dataset\npath = '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/'\ntrain_labels = pd.read_csv(path + 'train_labels.csv')\n\n# Konfigurasi\nIMG_SIZE = 128\nBATCH_SIZE = 32\nEPOCHS = 15\nMODALITY = 'FLAIR' \n\n# Fungsi untuk membaca dan memproses gambar DICOM\ndef load_dicom_image(filepath, img_size=IMG_SIZE):\n    dicom = pydicom.dcmread(filepath)\n    img = dicom.pixel_array.astype(float)\n    \n    # Normalisasi\n    img = (img - img.min()) / (img.max() - img.min())\n    \n    # Konversi ke uint8\n    img = (img * 255).astype(np.uint8)\n    \n    # Resize\n    img = cv2.resize(img, (img_size, img_size))\n    \n    # Stack ke 3 channel\n    img = np.stack([img]*3, axis=-1)\n    return img\n\n# Fungsi untuk memuat data pasien\ndef load_patient_data(patient_id, num_slices=16):\n    patient_path = os.path.join(path, 'train', str(patient_id).zfill(5), MODALITY)\n    slices = []\n    \n    if not os.path.exists(patient_path):\n        print(f\"Data tidak ditemukan untuk pasien {patient_id}\")\n        return None\n    \n    # Dapatkan semua file DICOM\n    dicom_files = sorted([f for f in os.listdir(patient_path) if f.endswith('.dcm')])\n    \n    if not dicom_files:\n        print(f\"Tidak ada file DICOM untuk pasien {patient_id}\")\n        return None\n    \n    # Pilih slice secara merata\n    step = max(1, len(dicom_files) // num_slices)\n    selected_files = dicom_files[::step][:num_slices]\n    \n    # Muat slice yang dipilih\n    for filename in selected_files:\n        img_path = os.path.join(patient_path, filename)\n        img = load_dicom_image(img_path)\n        slices.append(img)\n    \n    # Jika tidak cukup slice, duplikat yang terakhir\n    while len(slices) < num_slices:\n        slices.append(slices[-1].copy())  # Gunakan copy untuk menghindari reference yang sama\n    \n    return np.array(slices)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:15:30.096592Z","iopub.execute_input":"2026-07-11T10:15:30.097298Z","iopub.status.idle":"2026-07-11T10:15:30.108837Z","shell.execute_reply.started":"2026-07-11T10:15:30.097280Z","shell.execute_reply":"2026-07-11T10:15:30.108028Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Membuat dataset\nX = []\ny = []\n\nprint(\"Memuat data training...\")\nfor idx, row in train_labels.iterrows():\n    patient_id = row['BraTS21ID']\n    label = row['MGMT_value']\n    \n    patient_data = load_patient_data(patient_id)\n    if patient_data is not None:\n        X.append(patient_data)\n        y.append(label)\n\n# Konversi ke numpy array\nX = np.array(X, dtype=np.float32)\ny = np.array(y, dtype=np.float32)\n\nprint(f\"Total data yang dimuat: {len(X)} sampel\")\nprint(f\"Distribusi kelas: {np.sum(y == 1)} positif, {np.sum(y == 0)} negatif\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:15:30.109607Z","iopub.execute_input":"2026-07-11T10:15:30.109835Z","iopub.status.idle":"2026-07-11T10:16:01.917531Z","shell.execute_reply.started":"2026-07-11T10:15:30.109814Z","shell.execute_reply":"2026-07-11T10:16:01.916756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split data: training (60%), validation (20%), test (20%)\nX_train, X_temp, y_train, y_temp = train_test_split(\n    X, y, test_size=0.4, random_state=42, stratify=y\n)\nX_val, X_test, y_val, y_test = train_test_split(\n    X_temp, y_temp, test_size=0.5, random_state=42, stratify=y_temp\n)\n\nprint(\"\\nDistribusi dataset:\")\nprint(f\"Training:   {len(X_train)} sampel\")\nprint(f\"Validation: {len(X_val)} sampel\")\nprint(f\"Test:       {len(X_test)} sampel\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:16:01.918303Z","iopub.execute_input":"2026-07-11T10:16:01.918639Z","iopub.status.idle":"2026-07-11T10:16:02.659758Z","shell.execute_reply.started":"2026-07-11T10:16:01.918620Z","shell.execute_reply":"2026-07-11T10:16:02.659024Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Arsitektur model CNN 3D\ndef build_3d_cnn(input_shape, num_classes):\n    model = models.Sequential([\n        # Blok konvolusi 1\n        layers.Conv3D(16, (3, 3, 3), activation='relu', padding='same', input_shape=input_shape),\n        layers.BatchNormalization(),\n        layers.MaxPooling3D((2, 2, 2)),\n        layers.Dropout(0.2),\n        \n        # Blok konvolusi 2\n        layers.Conv3D(32, (3, 3, 3), activation='relu', padding='same'),\n        layers.BatchNormalization(),\n        layers.MaxPooling3D((2, 2, 2)),\n        layers.Dropout(0.3),\n        \n        # Blok konvolusi 3\n        layers.Conv3D(64, (3, 3, 3), activation='relu', padding='same'),\n        layers.BatchNormalization(),\n        layers.MaxPooling3D((2, 2, 2)),\n        layers.Dropout(0.4),\n        \n        layers.GlobalAveragePooling3D(),\n        layers.Dense(128, activation='relu'),\n        layers.Dropout(0.5),\n        layers.Dense(num_classes, activation='sigmoid')\n    ])\n    \n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:16:02.660790Z","iopub.execute_input":"2026-07-11T10:16:02.661309Z","iopub.status.idle":"2026-07-11T10:16:02.667089Z","shell.execute_reply.started":"2026-07-11T10:16:02.661288Z","shell.execute_reply":"2026-07-11T10:16:02.666221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Bangun model\ninput_shape = (X_train.shape[1], X_train.shape[2], X_train.shape[3], X_train.shape[4])\nprint(f\"\\nInput shape: {input_shape}\")\n#model = build_3d_cnn(input_shape, num_classes=1)\nmodel = build_3d_cnn(input_shape, 1)\n\n# Ringkasan model\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:16:02.668073Z","iopub.execute_input":"2026-07-11T10:16:02.668807Z","iopub.status.idle":"2026-07-11T10:16:02.790447Z","shell.execute_reply.started":"2026-07-11T10:16:02.668779Z","shell.execute_reply":"2026-07-11T10:16:02.789857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Kompilasi model\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n    loss='binary_crossentropy',\n    metrics=['accuracy', tf.keras.metrics.AUC(name='auc')]\n)\n\n# Callback\ncallbacks = [\n    tf.keras.callbacks.EarlyStopping(\n        patience=5, \n        monitor='val_auc', \n        mode='max', \n        restore_best_weights=True,\n        verbose=1\n    ),\n    tf.keras.callbacks.ReduceLROnPlateau(\n        monitor='val_loss', \n        factor=0.2, \n        patience=3, \n        min_lr=1e-6,\n        verbose=1\n    ),\n    tf.keras.callbacks.ModelCheckpoint(\n        filepath='best_model.h5',\n        save_best_only=True,\n        monitor='val_auc',\n        mode='max',\n        verbose=1\n    )\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:16:02.791147Z","iopub.execute_input":"2026-07-11T10:16:02.791359Z","iopub.status.idle":"2026-07-11T10:16:02.813244Z","shell.execute_reply.started":"2026-07-11T10:16:02.791344Z","shell.execute_reply":"2026-07-11T10:16:02.812400Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Training\nprint(\"\\nMemulai training...\")\nhistory = model.fit(\n    X_train, y_train,\n    validation_data=(X_val, y_val),\n    batch_size=BATCH_SIZE,\n    epochs=EPOCHS,\n    callbacks=callbacks\n)\n\n# Plot history training\ndef plot_history(history):\n    plt.figure(figsize=(12, 5))\n    \n    # Plot loss\n    plt.subplot(1, 2, 1)\n    plt.plot(history.history['loss'], label='Training Loss')\n    plt.plot(history.history['val_loss'], label='Validation Loss')\n    plt.title('Training and Validation Loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.legend()\n    \n    # Plot AUC\n    plt.subplot(1, 2, 2)\n    plt.plot(history.history['auc'], label='Training AUC')\n    plt.plot(history.history['val_auc'], label='Validation AUC')\n    plt.title('Training and Validation AUC')\n    plt.xlabel('Epoch')\n    plt.ylabel('AUC')\n    plt.legend()\n    \n    plt.tight_layout()\n    plt.savefig('training_history.png')\n    plt.show()\n\nplot_history(history)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:16:02.814037Z","iopub.execute_input":"2026-07-11T10:16:02.814361Z","iopub.status.idle":"2026-07-11T10:16:41.202487Z","shell.execute_reply.started":"2026-07-11T10:16:02.814344Z","shell.execute_reply":"2026-07-11T10:16:41.201900Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluasi pada validation set\nprint(\"\\nEvaluasi pada validation set:\")\nval_loss, val_acc, val_auc = model.evaluate(X_val, y_val)\nprint(f\"Validation Loss: {val_loss:.4f}\")\nprint(f\"Validation Accuracy: {val_acc:.4f}\")\nprint(f\"Validation AUC: {val_auc:.4f}\")\n\n# Evaluasi pada test set\nprint(\"\\nEvaluasi pada test set:\")\ntest_loss, test_acc, test_auc = model.evaluate(X_test, y_test)\nprint(f\"Test Loss: {test_loss:.4f}\")\nprint(f\"Test Accuracy: {test_acc:.4f}\")\nprint(f\"Test AUC: {test_auc:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:16:41.203580Z","iopub.execute_input":"2026-07-11T10:16:41.203860Z","iopub.status.idle":"2026-07-11T10:16:43.677736Z","shell.execute_reply.started":"2026-07-11T10:16:41.203836Z","shell.execute_reply":"2026-07-11T10:16:43.677038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prediksi pada test set\ny_pred_prob = model.predict(X_test).flatten()\ny_pred = (y_pred_prob > 0.5).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:16:43.678539Z","iopub.execute_input":"2026-07-11T10:16:43.678778Z","iopub.status.idle":"2026-07-11T10:16:45.781801Z","shell.execute_reply.started":"2026-07-11T10:16:43.678762Z","shell.execute_reply":"2026-07-11T10:16:45.781155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Classification report\nprint(\"\\nClassification Report:\")\nprint(classification_report(y_test, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:16:45.782624Z","iopub.execute_input":"2026-07-11T10:16:45.782849Z","iopub.status.idle":"2026-07-11T10:16:45.796230Z","shell.execute_reply.started":"2026-07-11T10:16:45.782832Z","shell.execute_reply":"2026-07-11T10:16:45.795283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Confusion matrix\nconf_matrix = confusion_matrix(y_test, y_pred)\nprint(\"\\nConfusion Matrix:\")\nprint(conf_matrix)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:16:45.796933Z","iopub.execute_input":"2026-07-11T10:16:45.797898Z","iopub.status.idle":"2026-07-11T10:16:45.803985Z","shell.execute_reply.started":"2026-07-11T10:16:45.797880Z","shell.execute_reply":"2026-07-11T10:16:45.803152Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot ROC curve\nfpr, tpr, thresholds = roc_curve(y_test, y_pred_prob)\nroc_auc = auc(fpr, tpr)\n\nplt.figure()\nplt.plot(fpr, tpr, color='darkorange', lw=2, label=f'ROC curve (area = {roc_auc:.2f})')\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic')\nplt.legend(loc=\"lower right\")\nplt.savefig('roc_curve.png')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:16:45.804717Z","iopub.execute_input":"2026-07-11T10:16:45.805008Z","iopub.status.idle":"2026-07-11T10:16:46.037075Z","shell.execute_reply.started":"2026-07-11T10:16:45.804966Z","shell.execute_reply":"2026-07-11T10:16:46.036377Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Simpan model akhir\nmodel.save('final_model.h5')\nprint(\"\\nModel akhir disimpan sebagai 'final_model.h5'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:16:46.037979Z","iopub.execute_input":"2026-07-11T10:16:46.038284Z","iopub.status.idle":"2026-07-11T10:16:46.096759Z","shell.execute_reply.started":"2026-07-11T10:16:46.038259Z","shell.execute_reply":"2026-07-11T10:16:46.096183Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Fixer toutes les graines AVANT tout\nimport random, os\nimport numpy as np\nimport tensorflow as tf\n\nSEED = 42\nrandom.seed(SEED)\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)\nos.environ['PYTHONHASHSEED'] = str(SEED)\nos.environ['TF_DETERMINISTIC_OPS'] = '1'\n\nprint(f\" Toutes les graines fixées à {SEED}\")\n\n# ── Réentraînement stable ──────────────────────────────────────\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom sklearn.metrics import (confusion_matrix, classification_report,\n                              balanced_accuracy_score)\n\ncw = compute_class_weight('balanced', classes=np.unique(y_train), y=y_train)\ncw_dict = {0: float(cw[0]), 1: float(cw[1])}\n\ninput_shape = (X_train.shape[1], X_train.shape[2],\n               X_train.shape[3], X_train.shape[4])\n\n# Même architecture originale\ntf.random.set_seed(SEED)\nmodel_stable = build_3d_cnn(input_shape, num_classes=1)\nmodel_stable.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=0.0006),\n    loss='binary_crossentropy',\n    metrics=['accuracy', tf.keras.metrics.AUC(name='auc')]\n)\n\ncallbacks_stable = [\n    tf.keras.callbacks.EarlyStopping(\n        patience=8, monitor='val_auc', mode='max',\n        restore_best_weights=True, verbose=1),\n    tf.keras.callbacks.ReduceLROnPlateau(\n        monitor='val_auc', factor=0.5, patience=4,\n        min_lr=1e-7, mode='max', verbose=1),\n]\n\nprint(\"\\n=== Entraînement stable (SEED=42) ===\")\nhistory_stable = model_stable.fit(\n    X_train, y_train,\n    validation_data=(X_val, y_val),\n    batch_size=8,\n    epochs=30,\n    callbacks=callbacks_stable,\n    class_weight=cw_dict\n)\n\n# Évaluation\ny_prob_stable = model_stable.predict(X_test, verbose=0).flatten()\n\n# Recherche seuil optimal\nprint(f\"\\n{'Seuil':<8} {'Acc':>6} {'Rec0':>6} {'Rec1':>6} {'BalAcc':>8} {'CM'}\")\nprint(\"-\"*65)\nmeilleur_seuil = 0.5\nmeilleur_bal   = 0\n\nfor seuil in np.arange(0.30, 0.71, 0.05):\n    y_pred_s = (y_prob_stable > seuil).astype(int)\n    cm = confusion_matrix(y_test, y_pred_s)\n    if cm.shape != (2,2): continue\n    acc  = (cm[0,0]+cm[1,1])/cm.sum()\n    rec0 = cm[0,0]/cm[0].sum()\n    rec1 = cm[1,1]/cm[1].sum()\n    bal  = balanced_accuracy_score(y_test, y_pred_s)\n    flag = \" ← ÉQUILIBRÉ\" if abs(rec0-rec1) < 0.15 else \"\"\n    print(f\"{seuil:<8.2f} {acc*100:>5.1f}% {rec0*100:>5.1f}% \"\n          f\"{rec1*100:>5.1f}% {bal*100:>7.1f}%  {cm.tolist()}{flag}\")\n    if bal > meilleur_bal:\n        meilleur_bal   = bal\n        meilleur_seuil = seuil\n\nprint(f\"\\n★ Seuil optimal : {meilleur_seuil:.2f}\")\ny_pred_final = (y_prob_stable > meilleur_seuil).astype(int)\ncm_final = confusion_matrix(y_test, y_pred_final)\n\nprint(f\"\\n RÉSULTAT FINAL STABLE \")\nprint(f\"val_AUC          : {max(history_stable.history['val_auc']):.4f}\")\nprint(f\"Balanced Accuracy: {meilleur_bal*100:.1f}%\")\nprint(f\"\\nConfusion Matrix finale :\")\nprint(cm_final)\nprint(classification_report(y_test, y_pred_final,\n      target_names=['Non méthylé(0)', 'Méthylé(1)']))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T10:16:46.097465Z","iopub.execute_input":"2026-07-11T10:16:46.097803Z","iopub.status.idle":"2026-07-11T10:17:45.027370Z","shell.execute_reply.started":"2026-07-11T10:16:46.097771Z","shell.execute_reply":"2026-07-11T10:17:45.026512Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# OPTIMISATION FINALE — CNN 3D (MGMT Prediction)\n# Paramètres optimaux validés empiriquement\nimport random, os\nimport numpy as np\nimport tensorflow as tf\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom sklearn.metrics import (classification_report, confusion_matrix,\n                              accuracy_score, balanced_accuracy_score)\n\n#  Reproductibilité \nSEED = 60\nrandom.seed(SEED)\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)\nos.environ['PYTHONHASHSEED'] = str(SEED)\nprint(f\" SEED fixé à {SEED}\")\n\n#  Paramètres optimaux \nLR_OPTIMAL = 0.0006   # validé empiriquement (testé 0.01→0.00001)\nBS_OPTIMAL = 8        # validé empiriquement (testé 8, 16, 32)\nprint(f\" lr={LR_OPTIMAL} | batch_size={BS_OPTIMAL}\")\n\n#  Class weights \ncw = compute_class_weight('balanced',\n     classes=np.unique(y_train), y=y_train)\ncw_dict = {0: float(cw[0]), 1: float(cw[1])}\nprint(f\" Class weights : {cw_dict}\")\n\n# Modèle  même architecture originale \ninput_shape = (X_train.shape[1], X_train.shape[2],\n               X_train.shape[3], X_train.shape[4])\nmodel_opt = build_3d_cnn(input_shape, 1)\nmodel_opt.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=LR_OPTIMAL),\n    loss='binary_crossentropy',\n    metrics=['accuracy', tf.keras.metrics.AUC(name='auc')]\n)\n\n#  Callbacks \ncallbacks_opt = [\n    tf.keras.callbacks.EarlyStopping(\n        patience=8, monitor='val_auc', mode='max',\n        restore_best_weights=True, verbose=1),\n    tf.keras.callbacks.ReduceLROnPlateau(\n        monitor='val_auc', factor=0.5, patience=4,\n        min_lr=1e-7, mode='max', verbose=1),\n    tf.keras.callbacks.ModelCheckpoint(\n        filepath='model_optimise_final.h5',\n        save_best_only=True, monitor='val_auc',\n        mode='max', verbose=1)\n]\n\n#  Entraînement \nprint(\"\\n Entraînement optimisé\")\nhistory_opt = model_opt.fit(\n    X_train, y_train,\n    validation_data=(X_val, y_val),\n    batch_size=BS_OPTIMAL,\n    epochs=30,\n    callbacks=callbacks_opt,\n    class_weight=cw_dict\n)\n\n#  Évaluation \ny_pred_orig = (model.predict(X_test, verbose=0).flatten() > 0.5).astype(int)\ny_pred_opt  = (model_opt.predict(X_test, verbose=0).flatten() > 0.5).astype(int)\n\ncm_orig = confusion_matrix(y_test, y_pred_orig)\ncm_opt  = confusion_matrix(y_test, y_pred_opt)\n\n#  Comparaison finale \nprint(\"\\n\" + \"=\"*55)\nprint(\"         COMPARAISON ORIGINAL vs OPTIMISÉ\")\nprint(\"=\"*55)\nprint(f\"\\n{'Métrique':<25} {'Original':>12} {'Optimisé':>12}\")\nprint(\"-\"*55)\nprint(f\"{'Accuracy':<25} \"\n      f\"{accuracy_score(y_test,y_pred_orig)*100:>11.1f}% \"\n      f\"{accuracy_score(y_test,y_pred_opt)*100:>11.1f}%\")\nprint(f\"{'Recall classe 0':<25} \"\n      f\"{cm_orig[0,0]/cm_orig[0].sum()*100:>11.1f}% \"\n      f\"{cm_opt[0,0]/cm_opt[0].sum()*100:>11.1f}%\")\nprint(f\"{'Recall classe 1':<25} \"\n      f\"{cm_orig[1,1]/cm_orig[1].sum()*100:>11.1f}% \"\n      f\"{cm_opt[1,1]/cm_opt[1].sum()*100:>11.1f}%\")\nprint(f\"{'Balanced Accuracy':<25} \"\n      f\"{balanced_accuracy_score(y_test,y_pred_orig)*100:>11.1f}% \"\n      f\"{balanced_accuracy_score(y_test,y_pred_opt)*100:>11.1f}%\")\nprint(f\"{'val_AUC':<25} \"\n      f\"{'0.696':>12} \"\n      f\"{max(history_opt.history['val_auc']):>11.4f}\")\nprint(f\"\\nConfusion Matrix originale :\")\nprint(cm_orig)\nprint(f\"\\nConfusion Matrix optimisée :\")\nprint(cm_opt)\nprint(\"\\nClassification Report optimisé :\")\nprint(classification_report(y_test, y_pred_opt,\n      target_names=['Non méthylé(0)', 'Méthylé(1)']))\nprint(\" Modèle sauvegardé : model_optimise_final.h5\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T11:00:52.204616Z","iopub.execute_input":"2026-07-11T11:00:52.205394Z","iopub.status.idle":"2026-07-11T11:01:42.760120Z","shell.execute_reply.started":"2026-07-11T11:00:52.205369Z","shell.execute_reply":"2026-07-11T11:01:42.759232Z"}},"outputs":[],"execution_count":null}]}