{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":29653,"databundleVersionId":2420395,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport torch\nimport torch.nn as nn\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib\nimport pydicom as dicom\nimport cv2\nimport ast\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:27:27.066224Z","iopub.execute_input":"2026-07-10T22:27:27.066459Z","iopub.status.idle":"2026-07-10T22:27:29.738865Z","shell.execute_reply.started":"2026-07-10T22:27:27.066440Z","shell.execute_reply":"2026-07-10T22:27:29.738000Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/'\nos.listdir(path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:27:29.746446Z","iopub.execute_input":"2026-07-10T22:27:29.746938Z","iopub.status.idle":"2026-07-10T22:27:29.754789Z","shell.execute_reply.started":"2026-07-10T22:27:29.746915Z","shell.execute_reply":"2026-07-10T22:27:29.754024Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/'\nos.listdir(path)\ntrain_data = pd.read_csv(path+'train_labels.csv')\nsamp_subm = pd.read_csv(path+'sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:27:29.756523Z","iopub.execute_input":"2026-07-10T22:27:29.756800Z","iopub.status.idle":"2026-07-10T22:27:29.775736Z","shell.execute_reply.started":"2026-07-10T22:27:29.756779Z","shell.execute_reply":"2026-07-10T22:27:29.774923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('Samples train:', len(train_data))\nprint('Samples test:', len(samp_subm))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:27:29.777028Z","iopub.execute_input":"2026-07-10T22:27:29.777341Z","iopub.status.idle":"2026-07-10T22:27:29.782740Z","shell.execute_reply.started":"2026-07-10T22:27:29.777311Z","shell.execute_reply":"2026-07-10T22:27:29.781618Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:27:29.783916Z","iopub.execute_input":"2026-07-10T22:27:29.784283Z","iopub.status.idle":"2026-07-10T22:27:29.802770Z","shell.execute_reply.started":"2026-07-10T22:27:29.784252Z","shell.execute_reply":"2026-07-10T22:27:29.801878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data[\"MGMT_value\"].value_counts().head(2).plot(kind = 'pie', autopct='%1.1f%%', figsize=(8, 8)).legend()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:27:29.803931Z","iopub.execute_input":"2026-07-10T22:27:29.804361Z","iopub.status.idle":"2026-07-10T22:27:29.970445Z","shell.execute_reply.started":"2026-07-10T22:27:29.804325Z","shell.execute_reply":"2026-07-10T22:27:29.969408Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data[\"MGMT_value\"].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:27:29.972387Z","iopub.execute_input":"2026-07-10T22:27:29.972712Z","iopub.status.idle":"2026-07-10T22:27:29.981119Z","shell.execute_reply.started":"2026-07-10T22:27:29.972689Z","shell.execute_reply":"2026-07-10T22:27:29.980133Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"samp_subm.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:27:29.982058Z","iopub.execute_input":"2026-07-10T22:27:29.982489Z","iopub.status.idle":"2026-07-10T22:27:30.001355Z","shell.execute_reply.started":"2026-07-10T22:27:29.982455Z","shell.execute_reply":"2026-07-10T22:27:29.999817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"folder = str(train_data.loc[0, 'BraTS21ID']).zfill(5)\nfolder","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:27:30.002611Z","iopub.execute_input":"2026-07-10T22:27:30.003000Z","iopub.status.idle":"2026-07-10T22:27:30.016746Z","shell.execute_reply.started":"2026-07-10T22:27:30.002933Z","shell.execute_reply":"2026-07-10T22:27:30.015795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.listdir(path+'train/'+folder)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:27:30.017776Z","iopub.execute_input":"2026-07-10T22:27:30.018112Z","iopub.status.idle":"2026-07-10T22:27:30.034152Z","shell.execute_reply.started":"2026-07-10T22:27:30.018091Z","shell.execute_reply":"2026-07-10T22:27:30.033315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('Number of FLAIR images:', len(os.listdir(path+'train/'+folder+'/'+'FLAIR')))\nprint('Number of T1w images:', len(os.listdir(path+'train/'+folder+'/'+'T1w')))\nprint('Number of T1wCE images:', len(os.listdir(path+'train/'+folder+'/'+'T1wCE')))\nprint('Number of T2w images:', len(os.listdir(path+'train/'+folder+'/'+'T2w')))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:27:30.037361Z","iopub.execute_input":"2026-07-10T22:27:30.038521Z","iopub.status.idle":"2026-07-10T22:27:30.052466Z","shell.execute_reply.started":"2026-07-10T22:27:30.038494Z","shell.execute_reply":"2026-07-10T22:27:30.051423Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path_file = ''.join([path, 'train/', folder, '/', 'FLAIR/'])\nimage = os.listdir(path_file)[0]\ndata_file = dicom.dcmread(path_file+image)\nimg = data_file.pixel_array\nprint('Image shape:', img.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:27:30.053336Z","iopub.execute_input":"2026-07-10T22:27:30.053691Z","iopub.status.idle":"2026-07-10T22:27:30.062337Z","shell.execute_reply.started":"2026-07-10T22:27:30.053661Z","shell.execute_reply":"2026-07-10T22:27:30.061510Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Flair Image\ndef plot_examples(row = 0, cat = 'FLAIR'): \n    folder = str(train_data.loc[row, 'BraTS21ID']).zfill(5)\n    path_file = ''.join([path, 'train/', folder, '/', cat, '/'])\n    images = os.listdir(path_file)\n    \n    fig, axs = plt.subplots(1, 5, figsize=(30, 30))\n    fig.subplots_adjust(hspace = .2, wspace=.2)\n    axs = axs.ravel()\n    \n    for num in range(5):\n        data_file = dicom.dcmread(path_file+images[num])\n        img = data_file.pixel_array\n        axs[num].imshow(img, cmap='gray')\n        axs[num].set_title(cat+' '+images[num])\n        axs[num].set_xticklabels([])\n        axs[num].set_yticklabels([])\n        \nrow = 0\nplot_examples(row = row, cat = 'FLAIR')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:27:30.063768Z","iopub.execute_input":"2026-07-10T22:27:30.064096Z","iopub.status.idle":"2026-07-10T22:27:30.880747Z","shell.execute_reply.started":"2026-07-10T22:27:30.064076Z","shell.execute_reply":"2026-07-10T22:27:30.879750Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#T1w Images\nplot_examples(row = row, cat = 'T1w')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:27:30.881663Z","iopub.execute_input":"2026-07-10T22:27:30.881945Z","iopub.status.idle":"2026-07-10T22:27:31.660921Z","shell.execute_reply.started":"2026-07-10T22:27:30.881924Z","shell.execute_reply":"2026-07-10T22:27:31.660029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#T1wCE Images\nplot_examples(row = row, cat = 'T1wCE')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:27:31.661931Z","iopub.execute_input":"2026-07-10T22:27:31.662254Z","iopub.status.idle":"2026-07-10T22:27:32.609285Z","shell.execute_reply.started":"2026-07-10T22:27:31.662235Z","shell.execute_reply":"2026-07-10T22:27:32.608191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#T2w Images\nplot_examples(row = row, cat = 'T2w')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:27:32.610238Z","iopub.execute_input":"2026-07-10T22:27:32.610576Z","iopub.status.idle":"2026-07-10T22:27:33.420188Z","shell.execute_reply.started":"2026-07-10T22:27:32.610555Z","shell.execute_reply":"2026-07-10T22:27:33.419124Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom sklearn.model_selection import train_test_split\nimport cv2\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import classification_report, confusion_matrix, roc_curve, auc\n\n# Path dataset\npath = '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/'\ntrain_labels = pd.read_csv(path + 'train_labels.csv')\n\n# Konfigurasi\nIMG_SIZE = 128\nBATCH_SIZE = 32\nEPOCHS = 15\nMODALITIES = ['FLAIR']   # une seule modalité\n\ndef load_dicom_slice(filepath, img_size=IMG_SIZE):\n    \"\"\"Charge et normalise UNE slice DICOM en niveaux de gris (pas de duplication en 3 canaux).\"\"\"\n    dicom = pydicom.dcmread(filepath)\n    img = dicom.pixel_array.astype(float)\n\n    # Normalisation\n    denom = (img.max() - img.min())\n    if denom == 0:\n        denom = 1e-6\n    img = (img - img.min()) / denom\n    img = (img * 255).astype(np.uint8)\n\n    # Resize\n    img = cv2.resize(img, (img_size, img_size))\n    return img  # shape (img_size, img_size), 1 seul canal\n\ndef load_modality_slices(patient_id, modality, num_slices=16):\n    \"\"\"Charge num_slices slices pour une modalité donnée, retourne None si absent.\"\"\"\n    patient_path = os.path.join(path, 'train', str(patient_id).zfill(5), modality)\n\n    if not os.path.exists(patient_path):\n        return None\n\n    dicom_files = sorted([f for f in os.listdir(patient_path) if f.endswith('.dcm')])\n    if not dicom_files:\n        return None\n\n    step = max(1, len(dicom_files) // num_slices)\n    selected_files = dicom_files[::step][:num_slices]\n\n    slices = []\n    for filename in selected_files:\n        img_path = os.path.join(patient_path, filename)\n        slices.append(load_dicom_slice(img_path))\n\n    while len(slices) < num_slices:\n        slices.append(slices[-1].copy())\n\n    return np.array(slices)  # shape (num_slices, img_size, img_size)\n\ndef load_patient_data(patient_id, num_slices=16, modalities=MODALITIES):\n    \"\"\"Charge et empile plusieurs modalités comme canaux : shape (num_slices, img_size, img_size, n_modalities).\"\"\"\n    modality_arrays = []\n    for mod in modalities:\n        arr = load_modality_slices(patient_id, mod, num_slices=num_slices)\n        if arr is None:\n            print(f\"Données manquantes pour patient {patient_id}, modalité {mod}\")\n            return None\n        modality_arrays.append(arr)\n\n    # Empiler sur le dernier axe : chaque modalité devient un canal\n    patient_volume = np.stack(modality_arrays, axis=-1)\n    return patient_volume","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:27:33.421261Z","iopub.execute_input":"2026-07-10T22:27:33.421652Z","iopub.status.idle":"2026-07-10T22:27:37.546293Z","shell.execute_reply.started":"2026-07-10T22:27:33.421625Z","shell.execute_reply":"2026-07-10T22:27:37.545436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Membuat dataset\nX = []\ny = []\n\nprint(\"Memuat data training...\")\nfor idx, row in train_labels.iterrows():\n    patient_id = row['BraTS21ID']\n    label = row['MGMT_value']\n    \n    patient_data = load_patient_data(patient_id)\n    if patient_data is not None:\n        X.append(patient_data)\n        y.append(label)\n\n# Konversi ke numpy array\nX = np.array(X, dtype=np.float32)\ny = np.array(y, dtype=np.float32)\n\nprint(f\"Total data yang dimuat: {len(X)} sampel\")\nprint(f\"Distribusi kelas: {np.sum(y == 1)} positif, {np.sum(y == 0)} negatif\")\nprint(f\"Shape d'un échantillon: {X.shape[1:]}\")   # devrait être (16, 128, 128, 2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:27:37.547189Z","iopub.execute_input":"2026-07-10T22:27:37.547452Z","iopub.status.idle":"2026-07-10T22:28:06.533245Z","shell.execute_reply.started":"2026-07-10T22:27:37.547435Z","shell.execute_reply":"2026-07-10T22:28:06.532255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\n\nN_FOLDS = 5\nskf = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=42)\n\nprint(f\"Validation croisée en {N_FOLDS} folds\")\nfor fold, (train_idx, val_idx) in enumerate(skf.split(X, y)):\n    print(f\"Fold {fold+1}: train={len(train_idx)}, val={len(val_idx)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:28:06.534197Z","iopub.execute_input":"2026-07-10T22:28:06.534461Z","iopub.status.idle":"2026-07-10T22:28:06.543068Z","shell.execute_reply.started":"2026-07-10T22:28:06.534434Z","shell.execute_reply":"2026-07-10T22:28:06.542029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Architecture (inchangée, réutilisée à chaque fold)\ndef build_3d_cnn(input_shape, num_classes):\n    model = models.Sequential([\n        layers.Conv3D(16, (3, 3, 3), activation='relu', padding='same', input_shape=input_shape),\n        layers.BatchNormalization(),\n        layers.MaxPooling3D((2, 2, 2)),\n        layers.Dropout(0.2),\n\n        layers.Conv3D(32, (3, 3, 3), activation='relu', padding='same'),\n        layers.BatchNormalization(),\n        layers.MaxPooling3D((2, 2, 2)),\n        layers.Dropout(0.3),\n\n        layers.Conv3D(64, (3, 3, 3), activation='relu', padding='same'),\n        layers.BatchNormalization(),\n        layers.MaxPooling3D((2, 2, 2)),\n        layers.Dropout(0.4),\n\n        layers.GlobalAveragePooling3D(),\n        layers.Dense(128, activation='relu'),\n        layers.Dropout(0.5),\n        layers.Dense(num_classes, activation='sigmoid')\n    ])\n    return model\n\n# Stockage des résultats\nfold_aucs = []\nfold_accs = []\noof_preds = np.zeros(len(y))   # prédictions out-of-fold (une par patient)\noof_true = y.copy()\n\ninput_shape = (X.shape[1], X.shape[2], X.shape[3], X.shape[4])\n\nfor fold, (train_idx, val_idx) in enumerate(skf.split(X, y)):\n    print(f\"\\n========== FOLD {fold+1}/{N_FOLDS} ==========\")\n\n    X_train, X_val = X[train_idx], X[val_idx]\n    y_train, y_val = y[train_idx], y[val_idx]\n\n    model = build_3d_cnn(input_shape, num_classes=1)\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=0.0005),\n        loss='binary_crossentropy',\n        metrics=['accuracy', tf.keras.metrics.AUC(name='auc')]\n    )\n\n    callbacks = [\n        tf.keras.callbacks.EarlyStopping(\n            patience=10, monitor='val_auc', mode='max',\n            restore_best_weights=True, verbose=0\n        ),\n        tf.keras.callbacks.ReduceLROnPlateau(\n            monitor='val_loss', factor=0.5, patience=3,\n            min_lr=1e-6, verbose=0\n        )\n    ]\n\n    history = model.fit(\n        X_train, y_train,\n        validation_data=(X_val, y_val),\n        batch_size=BATCH_SIZE,\n        epochs=EPOCHS,\n        callbacks=callbacks,\n        verbose=1\n    )\n\n    val_loss, val_acc, val_auc = model.evaluate(X_val, y_val, verbose=0)\n    print(f\"Fold {fold+1} -> AUC: {val_auc:.4f} | Accuracy: {val_acc:.4f}\")\n\n    fold_aucs.append(val_auc)\n    fold_accs.append(val_acc)\n\n    oof_preds[val_idx] = model.predict(X_val, verbose=0).flatten()\n\nprint(\"\\n========== RÉSULTATS FINAUX (moyenne sur les folds) ==========\")\nprint(f\"AUC moyen: {np.mean(fold_aucs):.4f} (+/- {np.std(fold_aucs):.4f})\")\nprint(f\"Accuracy moyenne: {np.mean(fold_accs):.4f} (+/- {np.std(fold_accs):.4f})\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:28:06.544160Z","iopub.execute_input":"2026-07-10T22:28:06.544395Z","iopub.status.idle":"2026-07-10T22:33:05.207080Z","shell.execute_reply.started":"2026-07-10T22:28:06.544368Z","shell.execute_reply":"2026-07-10T22:33:05.205844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Test rapide de seuils sur les prédictions déjà calculées — pas de ré-entraînement nécessaire\nfor threshold in [0.35, 0.40, 0.45, 0.50, 0.55]:\n    y_pred_t = (oof_preds >= threshold).astype(int)\n    print(f\"\\n===== Threshold = {threshold} =====\")\n    print(classification_report(oof_true, y_pred_t))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:36:17.878626Z","iopub.execute_input":"2026-07-10T22:36:17.879636Z","iopub.status.idle":"2026-07-10T22:36:17.922385Z","shell.execute_reply.started":"2026-07-10T22:36:17.879605Z","shell.execute_reply":"2026-07-10T22:36:17.921216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BEST_THRESHOLD = 0.35\ny_pred = (oof_preds >= BEST_THRESHOLD).astype(int)\n\nprint(f\"\\nClassification Report (threshold={BEST_THRESHOLD}):\")\nprint(classification_report(oof_true, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:37:54.544865Z","iopub.execute_input":"2026-07-10T22:37:54.545783Z","iopub.status.idle":"2026-07-10T22:37:54.562372Z","shell.execute_reply.started":"2026-07-10T22:37:54.545741Z","shell.execute_reply":"2026-07-10T22:37:54.561198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"conf_matrix = confusion_matrix(oof_true, y_pred)\nprint(\"\\nConfusion Matrix:\")\nprint(conf_matrix)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:38:08.642721Z","iopub.execute_input":"2026-07-10T22:38:08.643613Z","iopub.status.idle":"2026-07-10T22:38:08.653374Z","shell.execute_reply.started":"2026-07-10T22:38:08.643584Z","shell.execute_reply":"2026-07-10T22:38:08.652431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot ROC curve\nfpr, tpr, thresholds = roc_curve(oof_true, oof_preds)\nroc_auc = auc(fpr, tpr)\n\nplt.figure()\nplt.plot(fpr, tpr, color='darkorange', lw=2, label=f'ROC curve (area = {roc_auc:.2f})')\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('ROC Curve (out-of-fold predictions)')\nplt.legend(loc=\"lower right\")\nplt.savefig('roc_curve.png')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:38:13.245230Z","iopub.execute_input":"2026-07-10T22:38:13.246337Z","iopub.status.idle":"2026-07-10T22:38:13.535454Z","shell.execute_reply.started":"2026-07-10T22:38:13.246302Z","shell.execute_reply":"2026-07-10T22:38:13.534496Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Simpan model akhir ! des 5 models\nmodel.save('final_model.h5')\nprint(\"\\nModel akhir disimpan sebagai 'final_model.h5'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T22:38:16.738901Z","iopub.execute_input":"2026-07-10T22:38:16.739330Z","iopub.status.idle":"2026-07-10T22:38:16.809890Z","shell.execute_reply.started":"2026-07-10T22:38:16.739305Z","shell.execute_reply":"2026-07-10T22:38:16.809013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}