{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":29653,"databundleVersionId":2420395,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom sklearn.model_selection import train_test_split\nimport cv2\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import roc_auc_score","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-30T09:34:32.297812Z","iopub.execute_input":"2025-05-30T09:34:32.29814Z","iopub.status.idle":"2025-05-30T09:34:57.653064Z","shell.execute_reply.started":"2025-05-30T09:34:32.298106Z","shell.execute_reply":"2025-05-30T09:34:57.651807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Path dataset\npath = '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/'\ntrain_labels = pd.read_csv(path + 'train_labels.csv')\n\n# Konfigurasi sesuai solusi juara 1\nIMG_SIZE = 256\nBATCH_SIZE = 8\nEPOCHS = 15\nMODALITIES = ['FLAIR', 'T1w', 'T1wCE', 'T2w']  # Semua modalitas\nNUM_SLICES = 16  # 8 sebelum + 1 pusat + 7 setelah = 16","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T09:35:03.271745Z","iopub.execute_input":"2025-05-30T09:35:03.27212Z","iopub.status.idle":"2025-05-30T09:35:03.301051Z","shell.execute_reply.started":"2025-05-30T09:35:03.272094Z","shell.execute_reply":"2025-05-30T09:35:03.299737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Fungsi untuk menemukan slice dengan area otak terbesar\ndef find_central_slice(images):\n    max_area = -1\n    central_idx = len(images) // 2  # Default ke tengah\n    \n    for i, img in enumerate(images):\n        # Konversi ke grayscale dan thresholding\n        gray = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY) if len(img.shape) == 3 else img\n        _, thresh = cv2.threshold(gray, 20, 255, cv2.THRESH_BINARY)\n        \n        # Temukan kontur\n        contours, _ = cv2.findContours(thresh, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n        \n        if contours:\n            # Ambil kontur terbesar\n            largest_contour = max(contours, key=cv2.contourArea)\n            area = cv2.contourArea(largest_contour)\n            \n            if area > max_area:\n                max_area = area\n                central_idx = i\n                \n    return central_idx","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T09:35:26.404834Z","iopub.execute_input":"2025-05-30T09:35:26.406164Z","iopub.status.idle":"2025-05-30T09:35:26.415492Z","shell.execute_reply.started":"2025-05-30T09:35:26.406109Z","shell.execute_reply":"2025-05-30T09:35:26.413601Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Fungsi untuk memuat data pasien dengan \"central image trick\"\ndef load_patient_data(patient_id, modality, num_slices=NUM_SLICES):\n    patient_path = os.path.join(path, 'train', str(patient_id).zfill(5), modality)\n    if not os.path.exists(patient_path): \n        return None\n    \n    dicom_files = sorted([f for f in os.listdir(patient_path) if f.endswith('.dcm')])\n    if not dicom_files: \n        return None\n    \n    # Muat semua gambar\n    images = []\n    for filename in dicom_files:\n        dicom = pydicom.dcmread(os.path.join(patient_path, filename))\n        img = dicom.pixel_array.astype(float)\n        img = (img - img.min()) / (img.max() - img.min()) * 255\n        img = cv2.resize(img.astype(np.uint8), (IMG_SIZE, IMG_SIZE))\n        img = np.stack([img]*3, axis=-1)  # Convert to 3-channel\n        images.append(img)\n    \n    # Temukan slice sentral\n    central_idx = find_central_slice(images)\n    \n    # Hitung indeks awal dan akhir\n    start_idx = max(0, central_idx - num_slices//2)\n    end_idx = min(len(images), start_idx + num_slices)\n    \n    # Pastikan kita mendapatkan tepat num_slices\n    if end_idx - start_idx < num_slices:\n        start_idx = max(0, end_idx - num_slices)\n    \n    selected_images = images[start_idx:end_idx]\n    \n    # Padding jika perlu\n    while len(selected_images) < num_slices:\n        selected_images.append(np.zeros((IMG_SIZE, IMG_SIZE, 3), dtype=np.uint8))\n    \n    return np.array(selected_images[:num_slices])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T09:35:53.766499Z","iopub.execute_input":"2025-05-30T09:35:53.766845Z","iopub.status.idle":"2025-05-30T09:35:53.776624Z","shell.execute_reply.started":"2025-05-30T09:35:53.76682Z","shell.execute_reply":"2025-05-30T09:35:53.775139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Fungsi untuk membangun ResNet10 3D\ndef build_resnet10_3d(input_shape=(NUM_SLICES, IMG_SIZE, IMG_SIZE, 3)):\n    def residual_block(x, filters, stride=1):\n        shortcut = x\n        \n        # Blok konvolusi utama\n        x = layers.Conv3D(filters, kernel_size=3, strides=stride, padding='same')(x)\n        x = layers.BatchNormalization()(x)\n        x = layers.ReLU()(x)\n        \n        x = layers.Conv3D(filters, kernel_size=3, padding='same')(x)\n        x = layers.BatchNormalization()(x)\n        \n        # Shortcut connection jika dimensi berubah\n        if stride != 1 or shortcut.shape[-1] != filters:\n            shortcut = layers.Conv3D(filters, kernel_size=1, strides=stride)(shortcut)\n            shortcut = layers.BatchNormalization()(shortcut)\n        \n        x = layers.Add()([x, shortcut])\n        x = layers.ReLU()(x)\n        return x\n    \n    # Input layer\n    inputs = layers.Input(shape=input_shape)\n    \n    # Initial convolution\n    x = layers.Conv3D(64, kernel_size=7, strides=2, padding='same')(inputs)\n    x = layers.BatchNormalization()(x)\n    x = layers.ReLU()(x)\n    x = layers.MaxPool3D(pool_size=3, strides=2, padding='same')(x)\n    \n    # Residual blocks\n    x = residual_block(x, 64)\n    x = residual_block(x, 128, stride=2)\n    x = residual_block(x, 256, stride=2)\n    x = residual_block(x, 512, stride=2)\n    \n    # Head network\n    x = layers.GlobalAveragePooling3D()(x)\n    x = layers.Dense(256, activation='relu')(x)\n    outputs = layers.Dense(1, activation='sigmoid')(x)\n    \n    model = models.Model(inputs, outputs)\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T09:36:24.894771Z","iopub.execute_input":"2025-05-30T09:36:24.895206Z","iopub.status.idle":"2025-05-30T09:36:24.905456Z","shell.execute_reply.started":"2025-05-30T09:36:24.895177Z","shell.execute_reply":"2025-05-30T09:36:24.904139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Membuat dataset untuk satu modalitas\ndef create_dataset(modalitie):\n    X, y = [], []\n    \n    for idx, row in train_labels.iterrows():\n        patient_data = load_patient_data(row['BraTS21ID'], modalitie)\n        if patient_data is not None:\n            X.append(patient_data)\n            y.append(row['MGMT_value'])\n    \n    return np.array(X, dtype=np.float32) / 255.0, np.array(y, dtype=np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T09:36:43.582501Z","iopub.execute_input":"2025-05-30T09:36:43.582894Z","iopub.status.idle":"2025-05-30T09:36:43.589601Z","shell.execute_reply.started":"2025-05-30T09:36:43.582868Z","shell.execute_reply":"2025-05-30T09:36:43.588253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Inisialisasi model untuk setiap modalitas\nmodels_dict = {}\nfor modality in MODALITIES:\n    print(f\"\\nMembangun model untuk {modality}...\")\n    models_dict[modality] = build_resnet10_3d()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T09:37:00.549533Z","iopub.execute_input":"2025-05-30T09:37:00.549872Z","iopub.status.idle":"2025-05-30T09:37:02.530444Z","shell.execute_reply.started":"2025-05-30T09:37:00.549847Z","shell.execute_reply":"2025-05-30T09:37:02.529382Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Kompilasi model\nfor modality, model in models_dict.items():\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=0.0001),\n        loss='binary_crossentropy',\n        metrics=['accuracy', tf.keras.metrics.AUC(name='auc')]\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T09:37:20.197115Z","iopub.execute_input":"2025-05-30T09:37:20.19745Z","iopub.status.idle":"2025-05-30T09:37:20.248908Z","shell.execute_reply.started":"2025-05-30T09:37:20.197423Z","shell.execute_reply":"2025-05-30T09:37:20.247954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Callback untuk learning rate\ndef lr_scheduler(epoch, lr):\n    return 0.00005 if epoch >= 10 else 0.0001\n\nlr_callback = tf.keras.callbacks.LearningRateScheduler(lr_scheduler)\nearly_stop = tf.keras.callbacks.EarlyStopping(patience=5, restore_best_weights=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T09:37:31.531668Z","iopub.execute_input":"2025-05-30T09:37:31.53274Z","iopub.status.idle":"2025-05-30T09:37:31.538199Z","shell.execute_reply.started":"2025-05-30T09:37:31.53271Z","shell.execute_reply":"2025-05-30T09:37:31.537084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Pelatihan model untuk setiap modalitas\nhistories = {}\nfor modality in MODALITIES:\n    print(f\"\\nPelatihan model untuk {modality}...\")\n    X, y = create_dataset(modality)\n    \n    # Split data\n    X_train, X_val, y_train, y_val = train_test_split(\n        X, y, test_size=0.2, stratify=y, random_state=42\n    )\n    \n    # Pelatihan\n    history = models_dict[modality].fit(\n        X_train, y_train,\n        validation_data=(X_val, y_val),\n        epochs=EPOCHS,\n        batch_size=BATCH_SIZE,\n        callbacks=[lr_callback, early_stop],\n        class_weight={0: 1., 1: len(y_train[y_train==0])/len(y_train[y_train==1])}\n    )\n    \n    histories[modality] = history","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T09:37:52.364123Z","iopub.execute_input":"2025-05-30T09:37:52.364442Z","execution_failed":"2025-05-31T03:34:43.054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Fungsi untuk prediksi ensemble\ndef ensemble_predict(patient_id):\n    predictions = []\n    \n    for modality in MODALITIES:\n        data = load_patient_data(patient_id, modality)\n        if data is not None:\n            data = np.expand_dims(data, axis=0) / 255.0\n            pred = models_dict[modality].predict(data)[0][0]\n            predictions.append(pred)\n    \n    # Kembalikan rata-rata prediksi jika ada, jika tidak kembalikan 0.5\n    return np.mean(predictions) if predictions else 0.5","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluasi ensemble pada data validasi\ndef evaluate_ensemble():\n    y_true, y_pred = [], []\n    \n    # Gunakan data validasi dari salah satu modalitas sebagai referensi\n    X_sample, _ = create_dataset(MODALITIES[0])\n    patient_ids = [row['BraTS21ID'] for _, row in train_labels.iterrows()]\n    \n    for patient_id, label in zip(patient_ids, train_labels['MGMT_value']):\n        # Pastikan pasien ada di set validasi\n        pred = ensemble_predict(patient_id)\n        y_true.append(label)\n        y_pred.append(pred)\n    \n    auc = roc_auc_score(y_true, y_pred)\n    print(f\"\\nEnsemble AUC: {auc:.4f}\")\n    return auc","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}