{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.10","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":29653,"databundleVersionId":2420395,"sourceType":"competition"}],"dockerImageVersionId":30120,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Predicting Genetic Biomarker in Brain Tumor.\rPredict the genetic subtype of glioblastoma using MRI (magnetic resonance imaging) scans to train and test your model to detect for the presence of MGMT promoter methylation\n\n* Glossary:\r* \nMGMT promoter methylation - The presence of a specific genetic sequence in the tumor known as MGMT promoter methylation has been shown to be a favorable predictive factor and a strong predictor of responsiveness to chemotherapy.* \r\nRadio genomics - the field of predicting the genetics of the cancer throug \nimaging\r\nTypes of mpMRI\n* Fluid Attenuated Inversion Recovery (FLAIR)\n* T1-weighted pre-contrast (T1w)\n* T1-weighted post-contrast (T1Gd)\n* T2-weighted (TT2)","metadata":{}},{"cell_type":"markdown","source":"# Packages:","metadata":{}},{"cell_type":"code","source":"import os\nimport re \nimport glob\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport seaborn as sns\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n# Pydicom related imports\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport SimpleITK as sitk\n\n# Deep learning packages\nimport tensorflow as tf\n\n# For gif creation\nimport imageio\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T14:55:23.033585Z","iopub.execute_input":"2024-11-20T14:55:23.034052Z","iopub.status.idle":"2024-11-20T14:55:30.738663Z","shell.execute_reply.started":"2024-11-20T14:55:23.033956Z","shell.execute_reply":"2024-11-20T14:55:30.737619Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Loading and Inspection\nDefinition: Loading data into a DataFrame and inspecting it is the first step in EDA. This allows you to understand the structure, column names, and sample values within the dataset.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# Load data\ndataset = pd.read_csv(r'/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv')\n\n# Inspect the data\nprint(dataset.head())\nprint(dataset.info())\nprint(dataset.describe())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T14:55:30.740139Z","iopub.execute_input":"2024-11-20T14:55:30.740473Z","iopub.status.idle":"2024-11-20T14:55:30.792176Z","shell.execute_reply.started":"2024-11-20T14:55:30.740442Z","shell.execute_reply":"2024-11-20T14:55:30.791117Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Checking Class Distribution\nDefinition: It's essential to understand the distribution of the target variable (in this case, the MGMT promoter methylation status). This helps you assess any class imbalance.","metadata":{}},{"cell_type":"code","source":"# Check distribution of MGMT promoter methylation status\nprint(dataset['MGMT_value'].value_counts())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T14:55:30.793802Z","iopub.execute_input":"2024-11-20T14:55:30.794061Z","iopub.status.idle":"2024-11-20T14:55:30.80046Z","shell.execute_reply.started":"2024-11-20T14:55:30.794036Z","shell.execute_reply":"2024-11-20T14:55:30.79933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import glob\n\npatients = glob.glob('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/*')\nprint(f'There are {len(patients)} patients in the dataset')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T14:55:30.802656Z","iopub.execute_input":"2024-11-20T14:55:30.803115Z","iopub.status.idle":"2024-11-20T14:55:30.827571Z","shell.execute_reply.started":"2024-11-20T14:55:30.803073Z","shell.execute_reply":"2024-11-20T14:55:30.826611Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_files = glob.glob('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/*/*/*')\nprint(f'There are {len(train_files)} dicom files in the dataset')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T14:55:30.828875Z","iopub.execute_input":"2024-11-20T14:55:30.829178Z","iopub.status.idle":"2024-11-20T14:56:04.195655Z","shell.execute_reply.started":"2024-11-20T14:55:30.829147Z","shell.execute_reply":"2024-11-20T14:56:04.194473Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tqdm import tqdm\nimport seaborn as sns\n\nkeys = ['FLAIR', 'T1w', 'T1wCE', 'T2w']\n\nlabel_dict = {\n    'FLAIR': [],\n    'T1w': [],\n    'T1wCE': [],\n    'T2w': []\n}\n\nlabel_dict_counts = {}\n\nfor filename in tqdm(train_files):\n    \n    scan = filename.split('/')[-2]\n    \n    if scan=='FLAIR':\n        label_dict['FLAIR'].append(filename)\n        \n    elif scan=='T1w':\n        label_dict['T1w'].append(filename)\n\n    elif scan=='T1wCE':\n        label_dict['T1wCE'].append(filename)\n\n    else:\n        label_dict['T2w'].append(filename)\n    \nfor key in keys:\n    label_dict_counts[key] = len(label_dict[key])\n\nvalues = label_dict_counts.values()\nsns.barplot(x=keys, y=list(values))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T14:56:04.196964Z","iopub.execute_input":"2024-11-20T14:56:04.197254Z","iopub.status.idle":"2024-11-20T14:56:04.957124Z","shell.execute_reply.started":"2024-11-20T14:56:04.197224Z","shell.execute_reply":"2024-11-20T14:56:04.956025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# Number of files per patient per Key.\ntrain_folders = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/'\ndf_patient_records_train = pd.DataFrame(columns=['PatientId'] + keys)\ndf_patient_records_train.set_index('PatientId')\nfor f in tqdm(os.listdir(train_folders)):\n    patientId = f\n    df_patient_records_train = df_patient_records_train.append({'PatientId': patientId, 'FLAIR': 0, 'T1w': 0, 'T1wCE': 0, 'T2w' : 0}, ignore_index=True)\n    for key in keys:\n        patientId_key_path = f'../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/{patientId}/{key}/*.dcm'\n        df_patient_records_train.loc[df_patient_records_train['PatientId'] == patientId, [key]] = len(glob.glob(patientId_key_path))\ndf_patient_records_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T14:56:04.958457Z","iopub.execute_input":"2024-11-20T14:56:04.958798Z","iopub.status.idle":"2024-11-20T14:56:12.616665Z","shell.execute_reply.started":"2024-11-20T14:56:04.958765Z","shell.execute_reply":"2024-11-20T14:56:12.615696Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Loading MRI Images\nDefinition: MRI images are often stored in DICOM format. ","metadata":{}},{"cell_type":"code","source":"import pydicom\nimport cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport os\nimport glob\nimport random  # Import the random module\n\n# Function created to load DICOM files\ndef load_dicom_file(path):\n    dicom_file = pydicom.read_file(path)\n    pixel_array = dicom_file.pixel_array\n    pixel_array = pixel_array - np.min(pixel_array)\n    if np.max(pixel_array) != 0:\n        pixel_array = pixel_array / np.max(pixel_array)\n    pixel_array = (pixel_array * 255).astype(np.uint8)\n    return pixel_array\n\n# Function created to visualize a sample of brain tumor radiogenomic data\ndef visualize_sample(brats21id, slice_i, mgmt_value, types=(\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\")):\n    plt.figure(figsize=(16, 5))\n    patient_path = os.path.join(\n        \"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/\", \n        str(brats21id).zfill(5),\n    )\n    for i, t in enumerate(types, 1):\n        t_paths = sorted(\n            glob.glob(os.path.join(patient_path, t, \"*\")), \n            key=lambda x: int(x[:-4].split(\"-\")[-1]),\n        )\n        pixel_array = load_dicom_file(t_paths[int(len(t_paths) * slice_i)])\n        plt.subplot(1, 4, i)\n        plt.imshow(pixel_array, cmap=\"gray\")\n        plt.title(f\"{t}\", fontsize=16)\n        plt.axis(\"off\")\n\n    plt.suptitle(f\"MGMT_value: {mgmt_value}\", fontsize=16)\n    plt.show()\n\n# Selecting a random sample of 10 brain tumor radiogenomic images from the training dataset\nrandom_indices = random.sample(range(dataset.shape[0]), 10)\n\n# Visualizing each sample using the `visualize_sample()` function\nfor i in random_indices:\n    brats21id = dataset.iloc[i][\"BraTS21ID\"]\n    mgmt_value = dataset.iloc[i][\"MGMT_value\"]\n    visualize_sample(brats21id=brats21id, mgmt_value=mgmt_value, slice_i=0.5)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T14:56:12.619436Z","iopub.execute_input":"2024-11-20T14:56:12.619828Z","iopub.status.idle":"2024-11-20T14:56:16.545958Z","shell.execute_reply.started":"2024-11-20T14:56:12.619783Z","shell.execute_reply":"2024-11-20T14:56:16.544973Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Visualization","metadata":{}},{"cell_type":"markdown","source":"Bar Chart Visualization of Modalities:","metadata":{}},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\nimport shutil\nfrom sklearn.model_selection import train_test_split\n\n# Define the directory paths\ndataset_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\noutput_train_dir = \"./train\"\noutput_test_dir = \"./test\"\n\n# Function to count images for each modality in each patient folder\ndef count_images_per_modality(directory):\n    modalities = [\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"]\n    modality_counts = {modality: 0 for modality in modalities}\n    \n    for patient_folder in os.listdir(directory):\n        patient_path = os.path.join(directory, patient_folder)\n        \n        # Check if it's a folder\n        if os.path.isdir(patient_path):\n            # Count images in each modality folder\n            for modality in modalities:\n                modality_path = os.path.join(patient_path, modality)\n                if os.path.isdir(modality_path):\n                    modality_counts[modality] += len([img for img in os.listdir(modality_path) if img.endswith(\".dcm\")])\n    \n    return modality_counts\n\n# Split dataset into train and test\nall_patients = os.listdir(dataset_dir)\ntrain_patients, test_patients = train_test_split(all_patients, test_size=0.2, random_state=42)\n\n# Function to copy patient folders to respective train/test directories\ndef copy_patient_folders(patient_list, source_dir, destination_dir):\n    os.makedirs(destination_dir, exist_ok=True)\n    for patient in patient_list:\n        source_path = os.path.join(source_dir, patient)\n        dest_path = os.path.join(destination_dir, patient)\n        if os.path.isdir(source_path):\n            shutil.copytree(source_path, dest_path)\n\n# Copy train and test patient data\ncopy_patient_folders(train_patients, dataset_dir, output_train_dir)\ncopy_patient_folders(test_patients, dataset_dir, output_test_dir)\n\n# Get modality image counts for training and testing sets\ntrain_modality_counts = count_images_per_modality(output_train_dir)\ntest_modality_counts = count_images_per_modality(output_test_dir)\n\n# Plotting the results with improvements\nfig, ax = plt.subplots(1, 2, figsize=(16, 8))\ncolors = [\"#4e79a7\", \"#f28e2b\", \"#e15759\", \"#76b7b2\"]  # Different colors for each modality\n\n# Bar plot for train dataset\nax[0].bar(train_modality_counts.keys(), train_modality_counts.values(), color=colors)\nax[0].set_title(\"Number of Images per Modality in Training Dataset\", fontsize=14, weight='bold')\nax[0].set_xlabel(\"Modality\", fontsize=12)\nax[0].set_ylabel(\"Number of Images\", fontsize=12)\n\n# Adding annotations to the bars\nfor i, (modality, count) in enumerate(train_modality_counts.items()):\n    ax[0].text(i, count + 50, str(count), ha='center', va='bottom', fontsize=10)\n\n# Bar plot for test dataset\nax[1].bar(test_modality_counts.keys(), test_modality_counts.values(), color=colors)\nax[1].set_title(\"Number of Images per Modality in Testing Dataset\", fontsize=14, weight='bold')\nax[1].set_xlabel(\"Modality\", fontsize=12)\nax[1].set_ylabel(\"Number of Images\", fontsize=12)\n\n# Adding annotations to the bars\nfor i, (modality, count) in enumerate(test_modality_counts.items()):\n    ax[1].text(i, count + 50, str(count), ha='center', va='bottom', fontsize=10)\n\n# Adding a legend\nfig.legend(train_modality_counts.keys(), loc=\"upper center\", ncol=4, fontsize=12, frameon=False)\n\nplt.tight_layout(rect=[0, 0, 1, 0.95])  # Adjust layout to make space for legend\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T14:56:16.547515Z","iopub.execute_input":"2024-11-20T14:56:16.547881Z","iopub.status.idle":"2024-11-20T14:57:46.05934Z","shell.execute_reply.started":"2024-11-20T14:56:16.547849Z","shell.execute_reply":"2024-11-20T14:57:46.057064Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Pie Chart Visualization of Tumor","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\n# Load the CSV file\nfile_path = '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv'  \ndata = pd.read_csv(file_path)\n\n# Count the number of patients with and without tumors\ntumor_counts = data['MGMT_value'].value_counts()\n\n# Plotting\nlabels = ['No Tumor', 'Tumor']\nsizes = [tumor_counts.get(0, 0), tumor_counts.get(1, 0)]  # Get counts for 0 and 1, default to 0 if not found\ncolors = ['lightblue', 'salmon']\nexplode = (0.1, 0)  # explode 1st slice (No Tumor)\n\nplt.figure(figsize=(8, 6))\nplt.pie(sizes, explode=explode, labels=labels, colors=colors,\n        autopct='%1.1f%%', shadow=True, startangle=140)\n\nplt.title('Distribution of Patients with and without Tumors')\nplt.axis('equal')  # Equal aspect ratio ensures that pie chart is circular.\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T14:57:46.06066Z","iopub.status.idle":"2024-11-20T14:57:46.061045Z","shell.execute_reply":"2024-11-20T14:57:46.060863Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocessing, Training, Evaluation the MRI Images \nYou’ll need to load and preprocess these images (e.g., resizing, normalization) to make them suitable for model training.","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport pydicom\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.layers import GlobalAveragePooling2D, Dense, Dropout\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.regularizers import l2\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, EarlyStopping, ModelCheckpoint\nfrom sklearn.metrics import classification_report\nimport matplotlib.pyplot as plt\n\n# Load the labels dataset\nlabels_path = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv'\nimages_dir = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/train'\nlabels_df = pd.read_csv(labels_path)\n\n# Use a subset of patients\nsubset_patients = 250  # Increase the number of patients\nlabels_df = labels_df.head(subset_patients)\n\n# Define image processing parameters\nimage_size = (128, 128)\nmax_images_per_mri_type = 100  # Process more images per MRI type\n\n# Preprocess images and collect data\ndef preprocess_image(dicom):\n    \"\"\"Preprocess DICOM image by normalizing and resizing.\"\"\"\n    image = dicom.pixel_array.astype(np.float32)\n    image = (image - np.min(image)) / (np.max(image) - np.min(image) + 1e-7)  # Normalize to [0, 1]\n    return cv2.resize(image, image_size)\n\ndata = []\nfor _, row in labels_df.iterrows():\n    brats_id = str(row['BraTS21ID']).zfill(5)\n    label = row['MGMT_value']\n    \n    # Define paths for each MRI type\n    image_paths = {\n        'FLAIR': os.path.join(images_dir, brats_id, 'FLAIR'),\n        'T1w': os.path.join(images_dir, brats_id, 'T1w'),\n        'T1wCE': os.path.join(images_dir, brats_id, 'T1wCE'),\n        'T2w': os.path.join(images_dir, brats_id, 'T2w')\n    }\n    \n    for mri_type, path in image_paths.items():\n        if os.path.exists(path):\n            for image_file in os.listdir(path)[:max_images_per_mri_type]:\n                image_path = os.path.join(path, image_file)\n                try:\n                    dicom = pydicom.dcmread(image_path)\n                    image = preprocess_image(dicom)\n                    data.append((image, label))\n                except Exception as e:\n                    print(f\"Error processing {image_path}: {e}\")\n\n# Convert to NumPy arrays\nimages, labels = zip(*data)\nimages = np.array(images).reshape(-1, image_size[0], image_size[1], 1)\nlabels = np.array(labels)\n\n# Normalize labels to binary format\nlabels = (labels > 0.5).astype(int)\n\n# Compute class weights for imbalanced datasets\nclass_weights = compute_class_weight('balanced', classes=np.unique(labels), y=labels)\nclass_weights = dict(enumerate(class_weights))\n\n# Split dataset into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(images, labels, test_size=0.2, random_state=42)\n\n# Data Augmentation\ndatagen = ImageDataGenerator(\n    rotation_range=40,\n    width_shift_range=0.3,\n    height_shift_range=0.3,\n    shear_range=0.2,\n    zoom_range=0.3,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\n# Define the model with Transfer Learning (EfficientNetB0)\nbase_model = EfficientNetB0(weights='imagenet', include_top=False, input_shape=(128, 128, 3))\nbase_model.trainable = False  # Freeze base model layers\n\nmodel = Sequential([\n    base_model,\n    GlobalAveragePooling2D(),\n    Dense(128, activation='relu', kernel_regularizer=l2(0.01)),\n    Dropout(0.6),\n    Dense(1, activation='sigmoid')  # Binary classification\n])\n\n# Unfreeze the last few layers of the base model for fine-tuning\nfor layer in base_model.layers[-50:]:\n    layer.trainable = True\n\n# Compile the model with a smaller learning rate for fine-tuning\nmodel.compile(optimizer=Adam(learning_rate=1e-5), loss='binary_crossentropy', metrics=['accuracy'])\n\n# Callbacks for training\nearly_stopping = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\nmodel_checkpoint = ModelCheckpoint('best_model.h5', save_best_only=True)\nlr_scheduler = ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=3, min_lr=1e-7)\n\n# Train the model\nbatch_size = 32\nhistory = model.fit(\n    datagen.flow(np.repeat(X_train, 3, axis=-1), y_train, batch_size=batch_size),  # Convert to 3 channels\n    validation_data=(np.repeat(X_test, 3, axis=-1), y_test),  # Convert to 3 channels\n    epochs=50,\n    class_weight=class_weights,\n    callbacks=[early_stopping, model_checkpoint, lr_scheduler]\n)\n\n# Evaluate the model\ntrain_loss, train_accuracy = model.evaluate(np.repeat(X_train, 3, axis=-1), y_train)\ntest_loss, test_accuracy = model.evaluate(np.repeat(X_test, 3, axis=-1), y_test)\n\nprint(f\"\\nTraining Accuracy: {train_accuracy * 100:.2f}%\")\nprint(f\"Testing Accuracy: {test_accuracy * 100:.2f}%\")\n\n# Classification Report\ny_pred = (model.predict(np.repeat(X_test, 3, axis=-1)) > 0.5).astype(\"int32\")\nprint(\"\\nClassification Report:\")\nprint(classification_report(y_test, y_pred))\n\n# Plot Training and Validation Accuracy/Loss\nplt.figure(figsize=(12, 5))\nplt.subplot(1, 2, 1)\nplt.plot(history.history['accuracy'], label='Train Accuracy')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\nplt.title('Accuracy')\nplt.legend()\n\nplt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='Train Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Loss')\nplt.legend()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T14:57:52.280532Z","iopub.execute_input":"2024-11-20T14:57:52.280858Z"}},"outputs":[],"execution_count":null}]}