{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":29653,"databundleVersionId":2420395,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport torch\nimport torch.nn as nn\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib\nimport pydicom as dicom\nimport cv2\nimport ast\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:43:35.064928Z","iopub.execute_input":"2026-07-10T19:43:35.065086Z","iopub.status.idle":"2026-07-10T19:43:39.529803Z","shell.execute_reply.started":"2026-07-10T19:43:35.065071Z","shell.execute_reply":"2026-07-10T19:43:39.528932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom sklearn.metrics import f1_score\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:43:53.175862Z","iopub.execute_input":"2026-07-10T19:43:53.176633Z","iopub.status.idle":"2026-07-10T19:43:54.033761Z","shell.execute_reply.started":"2026-07-10T19:43:53.176607Z","shell.execute_reply":"2026-07-10T19:43:54.033136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/'\nos.listdir(path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:43:59.019908Z","iopub.execute_input":"2026-07-10T19:43:59.020544Z","iopub.status.idle":"2026-07-10T19:43:59.029602Z","shell.execute_reply.started":"2026-07-10T19:43:59.020519Z","shell.execute_reply":"2026-07-10T19:43:59.028950Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/'\nos.listdir(path)\ntrain_data = pd.read_csv(path+'train_labels.csv')\nsamp_subm = pd.read_csv(path+'sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:44:07.440713Z","iopub.execute_input":"2026-07-10T19:44:07.441316Z","iopub.status.idle":"2026-07-10T19:44:07.449234Z","shell.execute_reply.started":"2026-07-10T19:44:07.441289Z","shell.execute_reply":"2026-07-10T19:44:07.448433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('Samples train:', len(train_data))\nprint('Samples test:', len(samp_subm))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:44:27.107946Z","iopub.execute_input":"2026-07-10T19:44:27.108538Z","iopub.status.idle":"2026-07-10T19:44:27.112739Z","shell.execute_reply.started":"2026-07-10T19:44:27.108514Z","shell.execute_reply":"2026-07-10T19:44:27.111797Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:44:29.768319Z","iopub.execute_input":"2026-07-10T19:44:29.768972Z","iopub.status.idle":"2026-07-10T19:44:29.787681Z","shell.execute_reply.started":"2026-07-10T19:44:29.768943Z","shell.execute_reply":"2026-07-10T19:44:29.787109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data[\"MGMT_value\"].value_counts().head(2).plot(kind = 'pie', autopct='%1.1f%%', figsize=(8, 8)).legend()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:44:33.622920Z","iopub.execute_input":"2026-07-10T19:44:33.623548Z","iopub.status.idle":"2026-07-10T19:44:33.844094Z","shell.execute_reply.started":"2026-07-10T19:44:33.623523Z","shell.execute_reply":"2026-07-10T19:44:33.843297Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data[\"MGMT_value\"].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:44:38.410353Z","iopub.execute_input":"2026-07-10T19:44:38.410954Z","iopub.status.idle":"2026-07-10T19:44:38.417028Z","shell.execute_reply.started":"2026-07-10T19:44:38.410930Z","shell.execute_reply":"2026-07-10T19:44:38.416123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"samp_subm.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:44:46.597253Z","iopub.execute_input":"2026-07-10T19:44:46.597774Z","iopub.status.idle":"2026-07-10T19:44:46.605207Z","shell.execute_reply.started":"2026-07-10T19:44:46.597751Z","shell.execute_reply":"2026-07-10T19:44:46.604638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"folder = str(train_data.loc[0, 'BraTS21ID']).zfill(5)\nfolder","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:44:49.959646Z","iopub.execute_input":"2026-07-10T19:44:49.960169Z","iopub.status.idle":"2026-07-10T19:44:49.965247Z","shell.execute_reply.started":"2026-07-10T19:44:49.960146Z","shell.execute_reply":"2026-07-10T19:44:49.964533Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.listdir(path+'train/'+folder)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:44:53.081924Z","iopub.execute_input":"2026-07-10T19:44:53.082747Z","iopub.status.idle":"2026-07-10T19:44:53.137402Z","shell.execute_reply.started":"2026-07-10T19:44:53.082722Z","shell.execute_reply":"2026-07-10T19:44:53.136855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('Number of FLAIR images:', len(os.listdir(path+'train/'+folder+'/'+'FLAIR')))\nprint('Number of T1w images:', len(os.listdir(path+'train/'+folder+'/'+'T1w')))\nprint('Number of T1wCE images:', len(os.listdir(path+'train/'+folder+'/'+'T1wCE')))\nprint('Number of T2w images:', len(os.listdir(path+'train/'+folder+'/'+'T2w')))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:44:56.638217Z","iopub.execute_input":"2026-07-10T19:44:56.638880Z","iopub.status.idle":"2026-07-10T19:44:56.702713Z","shell.execute_reply.started":"2026-07-10T19:44:56.638854Z","shell.execute_reply":"2026-07-10T19:44:56.702132Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path_file = ''.join([path, 'train/', folder, '/', 'FLAIR/'])\nimage = os.listdir(path_file)[0]\ndata_file = dicom.dcmread(path_file+image)\nimg = data_file.pixel_array\nprint('Image shape:', img.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:45:00.206181Z","iopub.execute_input":"2026-07-10T19:45:00.206479Z","iopub.status.idle":"2026-07-10T19:45:00.240870Z","shell.execute_reply.started":"2026-07-10T19:45:00.206453Z","shell.execute_reply":"2026-07-10T19:45:00.240282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Flair Image\ndef plot_examples(row = 0, cat = 'FLAIR'): \n    folder = str(train_data.loc[row, 'BraTS21ID']).zfill(5)\n    path_file = ''.join([path, 'train/', folder, '/', cat, '/'])\n    images = os.listdir(path_file)\n    \n    fig, axs = plt.subplots(1, 5, figsize=(30, 30))\n    fig.subplots_adjust(hspace = .2, wspace=.2)\n    axs = axs.ravel()\n    \n    for num in range(5):\n        data_file = dicom.dcmread(path_file+images[num])\n        img = data_file.pixel_array\n        axs[num].imshow(img, cmap='gray')\n        axs[num].set_title(cat+' '+images[num])\n        axs[num].set_xticklabels([])\n        axs[num].set_yticklabels([])\n        \nrow = 0\nplot_examples(row = row, cat = 'FLAIR')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:45:03.755745Z","iopub.execute_input":"2026-07-10T19:45:03.756106Z","iopub.status.idle":"2026-07-10T19:45:04.454613Z","shell.execute_reply.started":"2026-07-10T19:45:03.756083Z","shell.execute_reply":"2026-07-10T19:45:04.453781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#T1w Images\nplot_examples(row = row, cat = 'T1w')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:45:10.369696Z","iopub.execute_input":"2026-07-10T19:45:10.370044Z","iopub.status.idle":"2026-07-10T19:45:11.149251Z","shell.execute_reply.started":"2026-07-10T19:45:10.370023Z","shell.execute_reply":"2026-07-10T19:45:11.148598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#T1wCE Images\nplot_examples(row = row, cat = 'T1wCE')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:45:13.082689Z","iopub.execute_input":"2026-07-10T19:45:13.083230Z","iopub.status.idle":"2026-07-10T19:45:13.727350Z","shell.execute_reply.started":"2026-07-10T19:45:13.083207Z","shell.execute_reply":"2026-07-10T19:45:13.726528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#T2w Images\nplot_examples(row = row, cat = 'T2w')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:45:17.429996Z","iopub.execute_input":"2026-07-10T19:45:17.430313Z","iopub.status.idle":"2026-07-10T19:45:18.120847Z","shell.execute_reply.started":"2026-07-10T19:45:17.430292Z","shell.execute_reply":"2026-07-10T19:45:18.119885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom sklearn.model_selection import train_test_split\nimport cv2\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import classification_report, confusion_matrix, roc_curve, auc\n\n# Path dataset\npath = '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/'\ntrain_labels = pd.read_csv(path + 'train_labels.csv')\n\n# Konfigurasi\nIMG_SIZE = 128\nBATCH_SIZE = 32\nEPOCHS = 15\nMODALITY = 'FLAIR' \n\n# Fungsi untuk membaca dan memproses gambar DICOM\ndef load_dicom_image(filepath, img_size=IMG_SIZE):\n    dicom = pydicom.dcmread(filepath)\n    img = dicom.pixel_array.astype(float)\n    \n    # Normalisasi\n    img = (img - img.min()) / (img.max() - img.min())\n    \n    # Konversi ke uint8\n    img = (img * 255).astype(np.uint8)\n    \n    # Resize\n    img = cv2.resize(img, (img_size, img_size))\n    \n    # Stack ke 3 channel\n    img = np.stack([img]*3, axis=-1)\n    return img\n\n# Fungsi untuk memuat data pasien\ndef load_patient_data(patient_id, num_slices=16):\n    patient_path = os.path.join(path, 'train', str(patient_id).zfill(5), MODALITY)\n    slices = []\n    \n    if not os.path.exists(patient_path):\n        print(f\"Data tidak ditemukan untuk pasien {patient_id}\")\n        return None\n    \n    # Dapatkan semua file DICOM\n    dicom_files = sorted([f for f in os.listdir(patient_path) if f.endswith('.dcm')])\n    \n    if not dicom_files:\n        print(f\"Tidak ada file DICOM untuk pasien {patient_id}\")\n        return None\n    \n    # Pilih slice secara merata\n    step = max(1, len(dicom_files) // num_slices)\n    selected_files = dicom_files[::step][:num_slices]\n    \n    # Muat slice yang dipilih\n    for filename in selected_files:\n        img_path = os.path.join(patient_path, filename)\n        img = load_dicom_image(img_path)\n        slices.append(img)\n    \n    # Jika tidak cukup slice, duplikat yang terakhir\n    while len(slices) < num_slices:\n        slices.append(slices[-1].copy())  # Gunakan copy untuk menghindari reference yang sama\n    \n    return np.array(slices)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:45:21.734113Z","iopub.execute_input":"2026-07-10T19:45:21.734748Z","iopub.status.idle":"2026-07-10T19:45:32.836940Z","shell.execute_reply.started":"2026-07-10T19:45:21.734723Z","shell.execute_reply":"2026-07-10T19:45:32.836352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#modification2 augmentation du dataset\ndef augment_volume(volume):\n    volume = volume.copy()\n\n    if random.random() > 0.5:\n        volume = np.flip(volume, axis=2).copy()\n\n    if random.random() > 0.5:\n        volume = np.flip(volume, axis=1).copy()\n\n    if random.random() > 0.5:\n        angle = random.uniform(-10, 10)\n        h, w = volume.shape[1], volume.shape[2]\n        center = (w // 2, h // 2)\n        M = cv2.getRotationMatrix2D(center, angle, 1.0)\n        for i in range(volume.shape[0]):\n            volume[i] = cv2.warpAffine(volume[i], M, (w, h))\n\n    if random.random() > 0.5:\n        factor = random.uniform(0.9, 1.1)\n        volume = np.clip(volume * factor, 0, 255)\n\n    return volume\n\n\ndef augment_dataset(X, y, num_augmentations=2):\n    X_aug = [X[i] for i in range(len(X))]\n    y_aug = [y[i] for i in range(len(y))]\n\n    for _ in range(num_augmentations):\n        for i in range(len(X)):\n            X_aug.append(augment_volume(X[i]))\n            y_aug.append(y[i])\n\n    return np.array(X_aug, dtype=np.float32), np.array(y_aug, dtype=np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:45:41.111996Z","iopub.execute_input":"2026-07-10T19:45:41.112676Z","iopub.status.idle":"2026-07-10T19:45:41.120294Z","shell.execute_reply.started":"2026-07-10T19:45:41.112649Z","shell.execute_reply":"2026-07-10T19:45:41.119434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Membuat dataset\nX = []\ny = []\n\nprint(\"Memuat data training...\")\nfor idx, row in train_labels.iterrows():\n    patient_id = row['BraTS21ID']\n    label = row['MGMT_value']\n    \n    patient_data = load_patient_data(patient_id)\n    if patient_data is not None:\n        X.append(patient_data)\n        y.append(label)\n\n# Konversi ke numpy array\nX = np.array(X, dtype=np.float32)\ny = np.array(y, dtype=np.float32)\n\nprint(f\"Total data yang dimuat: {len(X)} sampel\")\nprint(f\"Distribusi kelas: {np.sum(y == 1)} positif, {np.sum(y == 0)} negatif\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:45:45.492309Z","iopub.execute_input":"2026-07-10T19:45:45.492962Z","iopub.status.idle":"2026-07-10T19:47:12.766481Z","shell.execute_reply.started":"2026-07-10T19:45:45.492937Z","shell.execute_reply":"2026-07-10T19:47:12.765749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split data: training (60%), validation (20%), test (20%)\nX_train, X_temp, y_train, y_temp = train_test_split(\n    X, y, test_size=0.4, random_state=42, stratify=y\n)\nX_val, X_test, y_val, y_test = train_test_split(\n    X_temp, y_temp, test_size=0.5, random_state=42, stratify=y_temp\n)\n\nprint(\"\\nDistribusi dataset:\")\nprint(f\"Training:   {len(X_train)} sampel\")\nprint(f\"Validation: {len(X_val)} sampel\")\nprint(f\"Test:       {len(X_test)} sampel\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:47:32.508592Z","iopub.execute_input":"2026-07-10T19:47:32.508852Z","iopub.status.idle":"2026-07-10T19:47:33.256137Z","shell.execute_reply.started":"2026-07-10T19:47:32.508835Z","shell.execute_reply":"2026-07-10T19:47:33.255316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#modification3 Taille avant et après augmentation\nprint(f\"Taille avant augmentation: {len(X_train)}\")\nX_train, y_train = augment_dataset(X_train, y_train, num_augmentations=2)\nprint(f\"Taille après augmentation: {len(X_train)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:47:39.024955Z","iopub.execute_input":"2026-07-10T19:47:39.025794Z","iopub.status.idle":"2026-07-10T19:47:43.927795Z","shell.execute_reply.started":"2026-07-10T19:47:39.025768Z","shell.execute_reply":"2026-07-10T19:47:43.927158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#modification4\nclass_weights = compute_class_weight(\n    class_weight='balanced',\n    classes=np.unique(y_train),\n    y=y_train\n)\nclass_weight_dict = {0: class_weights[0], 1: class_weights[1]}\nprint(\"Class weights:\", class_weight_dict)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:48:14.725348Z","iopub.execute_input":"2026-07-10T19:48:14.725916Z","iopub.status.idle":"2026-07-10T19:48:14.732037Z","shell.execute_reply.started":"2026-07-10T19:48:14.725894Z","shell.execute_reply":"2026-07-10T19:48:14.731360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Arsitektur model CNN 3D\ndef build_3d_cnn(input_shape, num_classes):\n    model = models.Sequential([\n        # Blok konvolusi 1\n        layers.Conv3D(16, (3, 3, 3), activation='relu', padding='same', input_shape=input_shape),\n        layers.BatchNormalization(),\n        layers.MaxPooling3D((2, 2, 2)),\n        layers.Dropout(0.2),\n        \n        # Blok konvolusi 2\n        layers.Conv3D(32, (3, 3, 3), activation='relu', padding='same'),\n        layers.BatchNormalization(),\n        layers.MaxPooling3D((2, 2, 2)),\n        layers.Dropout(0.3),\n        \n        # Blok konvolusi 3\n        layers.Conv3D(64, (3, 3, 3), activation='relu', padding='same'),\n        layers.BatchNormalization(),\n        layers.MaxPooling3D((2, 2, 2)),\n        layers.Dropout(0.4),\n        \n        layers.GlobalAveragePooling3D(),\n        layers.Dense(128, activation='relu'),\n        layers.Dropout(0.5),\n        layers.Dense(num_classes, activation='sigmoid')\n    ])\n    \n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:48:34.043150Z","iopub.execute_input":"2026-07-10T19:48:34.044085Z","iopub.status.idle":"2026-07-10T19:48:34.050278Z","shell.execute_reply.started":"2026-07-10T19:48:34.044052Z","shell.execute_reply":"2026-07-10T19:48:34.049559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Bangun model\ninput_shape = (X_train.shape[1], X_train.shape[2], X_train.shape[3], X_train.shape[4])\nprint(f\"\\nInput shape: {input_shape}\")\nmodel = build_3d_cnn(input_shape, num_classes=1)\n\n# Ringkasan model\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:48:39.658837Z","iopub.execute_input":"2026-07-10T19:48:39.659640Z","iopub.status.idle":"2026-07-10T19:48:41.660442Z","shell.execute_reply.started":"2026-07-10T19:48:39.659614Z","shell.execute_reply":"2026-07-10T19:48:41.659741Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Kompilasi model\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n    loss='binary_crossentropy',\n    metrics=['accuracy', tf.keras.metrics.AUC(name='auc')]\n)\n\n# Callback\ncallbacks = [\n    tf.keras.callbacks.EarlyStopping(\n        patience=10, \n        monitor='val_auc', \n        mode='max', \n        restore_best_weights=True,\n        verbose=1\n    ),\n    tf.keras.callbacks.ReduceLROnPlateau(\n        monitor='val_loss', \n        factor=0.2, \n        patience=4, \n        min_lr=1e-6,\n        verbose=1\n    ),\n    tf.keras.callbacks.ModelCheckpoint(\n        filepath='best_model.h5',\n        save_best_only=True,\n        monitor='val_auc',\n        mode='max',\n        verbose=1\n    )\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:57:18.945792Z","iopub.execute_input":"2026-07-10T19:57:18.946339Z","iopub.status.idle":"2026-07-10T19:57:18.960704Z","shell.execute_reply.started":"2026-07-10T19:57:18.946317Z","shell.execute_reply":"2026-07-10T19:57:18.959937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Training\nprint(\"\\nMemulai training...\")\nhistory = model.fit(\n    X_train, y_train,\n    validation_data=(X_val, y_val),\n    batch_size=BATCH_SIZE,\n    epochs=40,\n    callbacks=callbacks   # sans class_weight\n)\n\n# Plot history training\ndef plot_history(history):\n    plt.figure(figsize=(12, 5))\n    \n    # Plot loss\n    plt.subplot(1, 2, 1)\n    plt.plot(history.history['loss'], label='Training Loss')\n    plt.plot(history.history['val_loss'], label='Validation Loss')\n    plt.title('Training and Validation Loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.legend()\n    \n    # Plot AUC\n    plt.subplot(1, 2, 2)\n    plt.plot(history.history['auc'], label='Training AUC')\n    plt.plot(history.history['val_auc'], label='Validation AUC')\n    plt.title('Training and Validation AUC')\n    plt.xlabel('Epoch')\n    plt.ylabel('AUC')\n    plt.legend()\n    \n    plt.tight_layout()\n    plt.savefig('training_history.png')\n    plt.show()\n\nplot_history(history)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T19:57:24.462868Z","iopub.execute_input":"2026-07-10T19:57:24.463093Z","iopub.status.idle":"2026-07-10T19:59:52.356581Z","shell.execute_reply.started":"2026-07-10T19:57:24.463077Z","shell.execute_reply":"2026-07-10T19:59:52.355779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluasi pada validation set\nprint(\"\\nEvaluasi pada validation set:\")\nval_loss, val_acc, val_auc = model.evaluate(X_val, y_val)\nprint(f\"Validation Loss: {val_loss:.4f}\")\nprint(f\"Validation Accuracy: {val_acc:.4f}\")\nprint(f\"Validation AUC: {val_auc:.4f}\")\n\n# Evaluasi pada test set\nprint(\"\\nEvaluasi pada test set:\")\ntest_loss, test_acc, test_auc = model.evaluate(X_test, y_test)\nprint(f\"Test Loss: {test_loss:.4f}\")\nprint(f\"Test Accuracy: {test_acc:.4f}\")\nprint(f\"Test AUC: {test_auc:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T20:00:17.775001Z","iopub.execute_input":"2026-07-10T20:00:17.775535Z","iopub.status.idle":"2026-07-10T20:00:20.495656Z","shell.execute_reply.started":"2026-07-10T20:00:17.775512Z","shell.execute_reply":"2026-07-10T20:00:20.494788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prediksi pada test set\ny_pred_prob = model.predict(X_test).flatten()\ny_pred = (y_pred_prob > 0.5).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T20:00:30.622843Z","iopub.execute_input":"2026-07-10T20:00:30.623189Z","iopub.status.idle":"2026-07-10T20:00:32.843750Z","shell.execute_reply.started":"2026-07-10T20:00:30.623167Z","shell.execute_reply":"2026-07-10T20:00:32.842905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#modification5\nbest_threshold = 0.5\nbest_f1 = 0\nfor threshold in np.arange(0.3, 0.71, 0.01):\n    y_pred_temp = (y_pred_prob > threshold).astype(int)\n    f1 = f1_score(y_test, y_pred_temp, average='macro')   # <-- macro au lieu de binaire\n    if f1 > best_f1:\n        best_f1 = f1\n        best_threshold = threshold\n\nprint(f\"Meilleur seuil trouvé: {best_threshold:.2f} (F1-macro: {best_f1:.4f})\")\ny_pred = (y_pred_prob > best_threshold).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T20:00:36.851230Z","iopub.execute_input":"2026-07-10T20:00:36.851996Z","iopub.status.idle":"2026-07-10T20:00:36.908993Z","shell.execute_reply.started":"2026-07-10T20:00:36.851970Z","shell.execute_reply":"2026-07-10T20:00:36.908360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Classification report\nprint(\"\\nClassification Report:\")\nprint(classification_report(y_test, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T20:00:41.046082Z","iopub.execute_input":"2026-07-10T20:00:41.046335Z","iopub.status.idle":"2026-07-10T20:00:41.059685Z","shell.execute_reply.started":"2026-07-10T20:00:41.046318Z","shell.execute_reply":"2026-07-10T20:00:41.059080Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Confusion matrix\nconf_matrix = confusion_matrix(y_test, y_pred)\nprint(\"\\nConfusion Matrix:\")\nprint(conf_matrix)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T20:00:55.532632Z","iopub.execute_input":"2026-07-10T20:00:55.532998Z","iopub.status.idle":"2026-07-10T20:00:55.540084Z","shell.execute_reply.started":"2026-07-10T20:00:55.532976Z","shell.execute_reply":"2026-07-10T20:00:55.539214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(6, 5))\nsns.heatmap(conf_matrix, annot=True, fmt='d', cmap='Blues',\n            xticklabels=['MGMT=0', 'MGMT=1'],\n            yticklabels=['MGMT=0', 'MGMT=1'])\nplt.xlabel('Prédiction')\nplt.ylabel('Réalité')\nplt.title('Matrice de confusion (seuil optimisé)')\nplt.savefig('confusion_matrix_improved.png')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T20:01:00.379176Z","iopub.execute_input":"2026-07-10T20:01:00.379812Z","iopub.status.idle":"2026-07-10T20:01:00.602262Z","shell.execute_reply.started":"2026-07-10T20:01:00.379789Z","shell.execute_reply":"2026-07-10T20:01:00.601439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot ROC curve\nfpr, tpr, thresholds = roc_curve(y_test, y_pred_prob)\nroc_auc = auc(fpr, tpr)\n\nplt.figure()\nplt.plot(fpr, tpr, color='darkorange', lw=2, label=f'ROC curve (area = {roc_auc:.2f})')\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic')\nplt.legend(loc=\"lower right\")\nplt.savefig('roc_curve.png')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T20:01:06.151101Z","iopub.execute_input":"2026-07-10T20:01:06.151914Z","iopub.status.idle":"2026-07-10T20:01:06.383913Z","shell.execute_reply.started":"2026-07-10T20:01:06.151880Z","shell.execute_reply":"2026-07-10T20:01:06.383121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Simpan model akhir\nmodel.save('final_model.h5')\nprint(\"\\nModel akhir disimpan sebagai 'final_model.h5'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T20:01:13.682365Z","iopub.execute_input":"2026-07-10T20:01:13.682879Z","iopub.status.idle":"2026-07-10T20:01:13.739830Z","shell.execute_reply.started":"2026-07-10T20:01:13.682855Z","shell.execute_reply":"2026-07-10T20:01:13.739227Z"}},"outputs":[],"execution_count":null}]}