{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":29653,"databundleVersionId":2420395,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\n# List all files in the input directory and subdirectories\ninput_dir = '/kaggle/input'\nfile_paths = []\n\nfor dirname, _, filenames in os.walk(input_dir):\n    for filename in filenames:\n        file_paths.append(os.path.join(dirname, filename))\n\n# Print confirmation message\nif file_paths:\n    print(\"Input added successfully.\")\nelse:\n    print(\"No input files found.\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:07:37.305997Z","iopub.execute_input":"2024-12-22T12:07:37.306359Z","iopub.status.idle":"2024-12-22T12:16:41.845654Z","shell.execute_reply.started":"2024-12-22T12:07:37.306322Z","shell.execute_reply":"2024-12-22T12:16:41.844724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\n# Verify GPU availability\nprint(\"Num GPUs Available: \", len(tf.config.list_physical_devices('GPU')))\n\n# List available GPU devices\nif tf.config.list_physical_devices('GPU'):\n    print(\"TensorFlow is set to use the GPU.\")\nelse:\n    print(\"No GPU detected. The model will run on the CPU.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:16:41.847356Z","iopub.execute_input":"2024-12-22T12:16:41.847648Z","iopub.status.idle":"2024-12-22T12:16:41.852659Z","shell.execute_reply.started":"2024-12-22T12:16:41.847615Z","shell.execute_reply":"2024-12-22T12:16:41.851745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\n\n# Define the directory paths\ntrain_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\ntest_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/test\"\n\n# Function to count patients and images\ndef count_patients_and_images(directory):\n    patient_count = 0\n    image_count = 0\n    \n    for patient_folder in os.listdir(directory):\n        patient_path = os.path.join(directory, patient_folder)\n        \n        # Check if it's a folder\n        if os.path.isdir(patient_path):\n            patient_count += 1\n            \n            # Count images in each modality folder\n            for modality in [\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"]:\n                modality_path = os.path.join(patient_path, modality)\n                if os.path.isdir(modality_path):\n                    image_count += len([img for img in os.listdir(modality_path) if img.endswith(\".dcm\")])\n\n    return patient_count, image_count\n\n# Get counts for training and testing sets\ntrain_patients, train_images = count_patients_and_images(train_dir)\ntest_patients, test_images = count_patients_and_images(test_dir)\n\n# Plotting the results\nfig, ax = plt.subplots(1, 2, figsize=(12, 6))\n\n# Bar plot for number of patients\nax[0].bar([\"Train\", \"Test\"], [train_patients, test_patients], color=[\"blue\", \"orange\"])\nax[0].set_title(\"Total Number of Patients\")\nax[0].set_ylabel(\"Number of Patients\")\n\n# Bar plot for number of images\nax[1].bar([\"Train\", \"Test\"], [train_images, test_images], color=[\"blue\", \"orange\"])\nax[1].set_title(\"Total Number of Images\")\nax[1].set_ylabel(\"Number of Images\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:16:41.853787Z","iopub.execute_input":"2024-12-22T12:16:41.854104Z","iopub.status.idle":"2024-12-22T12:16:45.016520Z","shell.execute_reply.started":"2024-12-22T12:16:41.854068Z","shell.execute_reply":"2024-12-22T12:16:45.015708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\n\n# Define the directory paths\ntrain_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\ntest_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/test\"\n\n# Function to count images for each modality in each patient folder\ndef count_images_per_modality(directory):\n    modalities = [\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"]\n    modality_counts = {modality: 0 for modality in modalities}\n    \n    for patient_folder in os.listdir(directory):\n        patient_path = os.path.join(directory, patient_folder)\n        \n        # Check if it's a folder\n        if os.path.isdir(patient_path):\n            # Count images in each modality folder\n            for modality in modalities:\n                modality_path = os.path.join(patient_path, modality)\n                if os.path.isdir(modality_path):\n                    modality_counts[modality] += len([img for img in os.listdir(modality_path) if img.endswith(\".dcm\")])\n    \n    return modality_counts\n\n# Get modality image counts for training and testing sets\ntrain_modality_counts = count_images_per_modality(train_dir)\ntest_modality_counts = count_images_per_modality(test_dir)\n\n# Plotting the results\nfig, ax = plt.subplots(1, 2, figsize=(14, 6))\n\n# Bar plot for train dataset\nax[0].bar(train_modality_counts.keys(), train_modality_counts.values(), color=\"blue\")\nax[0].set_title(\"Number of Images per Modality in Training Dataset\")\nax[0].set_xlabel(\"Modality\")\nax[0].set_ylabel(\"Number of Images\")\n\n# Bar plot for test dataset\nax[1].bar(test_modality_counts.keys(), test_modality_counts.values(), color=\"orange\")\nax[1].set_title(\"Number of Images per Modality in Testing Dataset\")\nax[1].set_xlabel(\"Modality\")\nax[1].set_ylabel(\"Number of Images\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:16:45.018856Z","iopub.execute_input":"2024-12-22T12:16:45.019244Z","iopub.status.idle":"2024-12-22T12:16:46.864685Z","shell.execute_reply.started":"2024-12-22T12:16:45.019205Z","shell.execute_reply":"2024-12-22T12:16:46.863808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\n\n# Define the directory path\ntrain_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\n\n# Function to count images for each modality in each patient folder\ndef count_images_per_modality(directory):\n    modalities = [\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"]\n    modality_counts = {modality: 0 for modality in modalities}\n    \n    for patient_folder in os.listdir(directory):\n        patient_path = os.path.join(directory, patient_folder)\n        \n        # Check if it's a folder\n        if os.path.isdir(patient_path):\n            # Count images in each modality folder\n            for modality in modalities:\n                modality_path = os.path.join(patient_path, modality)\n                if os.path.isdir(modality_path):\n                    modality_counts[modality] += len([img for img in os.listdir(modality_path) if img.endswith(\".dcm\")])\n    \n    return modality_counts\n\n# Get modality image counts for training set\ntrain_modality_counts = count_images_per_modality(train_dir)\n\n# Plotting the results for training dataset\nplt.figure(figsize=(8, 6))\ncolors = [\"#4e79a7\", \"#f28e2b\", \"#e15759\", \"#76b7b2\"]  # Different colors for each modality\n\n# Bar plot for train dataset\nplt.bar(train_modality_counts.keys(), train_modality_counts.values(), color=colors)\nplt.title(\"Number of Images per Modality in Training Dataset\", fontsize=14, weight='bold')\nplt.xlabel(\"Modality\", fontsize=12)\nplt.ylabel(\"Number of Images\", fontsize=12)\n\n# Adding annotations to the bars\nfor i, (modality, count) in enumerate(train_modality_counts.items()):\n    plt.text(i, count + 50, str(count), ha='center', va='bottom', fontsize=10)\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:41:22.574508Z","iopub.execute_input":"2024-12-22T12:41:22.574880Z","iopub.status.idle":"2024-12-22T12:41:23.840861Z","shell.execute_reply.started":"2024-12-22T12:41:22.574836Z","shell.execute_reply":"2024-12-22T12:41:23.839897Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# Define the directory path\ntrain_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\n\n# Function to find the patient with the maximum number of images for each modality\ndef find_max_images_per_modality(directory):\n    # Dictionary to store the max count for each modality\n    max_images_per_modality = {modality: {\"patient_id\": None, \"image_count\": 0} for modality in [\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"]}\n    \n    for patient_folder in os.listdir(directory):\n        patient_path = os.path.join(directory, patient_folder)\n        \n        # Check if it's a folder\n        if os.path.isdir(patient_path):\n            # Count images in each modality folder\n            for modality in [\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"]:\n                modality_path = os.path.join(patient_path, modality)\n                if os.path.isdir(modality_path):\n                    image_count = len([img for img in os.listdir(modality_path) if img.endswith(\".dcm\")])\n                    \n                    # Update max if this patient has more images for the modality\n                    if image_count > max_images_per_modality[modality][\"image_count\"]:\n                        max_images_per_modality[modality] = {\"patient_id\": patient_folder, \"image_count\": image_count}\n    \n    return max_images_per_modality\n\n# Get the patient with the maximum images for each modality\nmax_images_per_modality = find_max_images_per_modality(train_dir)\n\n# Display the results\nprint(\"Patient IDs with Maximum Number of Images per Modality:\")\nfor modality, data in max_images_per_modality.items():\n    print(f\"Modality: {modality}, Patient ID: {data['patient_id']}, Image Count: {data['image_count']}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:16:48.717543Z","iopub.execute_input":"2024-12-22T12:16:48.717938Z","iopub.status.idle":"2024-12-22T12:16:49.834152Z","shell.execute_reply.started":"2024-12-22T12:16:48.717898Z","shell.execute_reply":"2024-12-22T12:16:49.833277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\n# Load the CSV file\nfile_path = '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv'  # Replace with your actual file path\ndata = pd.read_csv(file_path)\n\n# Count the number of patients with and without tumors\ntumor_counts = data['MGMT_value'].value_counts()\n\n# Plotting\nlabels = ['No Tumor', 'Tumor']\nsizes = [tumor_counts.get(0, 0), tumor_counts.get(1, 0)]  # Get counts for 0 and 1, default to 0 if not found\ncolors = ['lightblue', 'salmon']\nexplode = (0.1, 0)  # explode 1st slice (No Tumor)\n\nplt.figure(figsize=(8, 6))\nplt.pie(sizes, explode=explode, labels=labels, colors=colors,\n        autopct='%1.1f%%', shadow=True, startangle=140)\n\nplt.title('Distribution of Patients with and without Tumors')\nplt.axis('equal')  # Equal aspect ratio ensures that pie chart is circular.\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:16:49.835403Z","iopub.execute_input":"2024-12-22T12:16:49.836274Z","iopub.status.idle":"2024-12-22T12:16:49.995196Z","shell.execute_reply.started":"2024-12-22T12:16:49.836231Z","shell.execute_reply":"2024-12-22T12:16:49.994138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# Define the directories and file paths\ntrain_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\ncsv_path = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\"\n\n# Load the CSV file\nlabels_df = pd.read_csv(csv_path)\n\n# Process patient IDs to remove leading zeros and use the same format in both places\nlabels_df['BraTS21ID'] = labels_df['BraTS21ID'].apply(lambda x: str(x).zfill(5))\n\n# Initialize lists for analysis data and skipped IDs\nanalysis_data = []\nskipped_ids = []\n\n# Function to count images in a given modality folder\ndef count_images_in_folder(folder_path):\n    return len([img for img in os.listdir(folder_path) if img.endswith(\".dcm\")])\n\n# Get all patient IDs from both CSV and folders for cross-referencing\nall_patient_folders = set(os.listdir(train_dir))\nall_csv_ids = set(labels_df['BraTS21ID'].values)\n\n# Iterate through each patient folder in the training directory\nfor patient_folder in all_patient_folders:\n    patient_id = patient_folder.zfill(5)  # Ensure ID format consistency\n\n    # Check if the patient ID exists in the labels DataFrame\n    if patient_id in all_csv_ids:\n        # Find corresponding label for the patient\n        label_row = labels_df[labels_df['BraTS21ID'] == patient_id]\n        tumor_status = label_row['MGMT_value'].values[0]\n\n        # Initialize counts for FLAIR and T1wCE\n        flair_count, t1wce_count = 0, 0\n\n        # Define paths for FLAIR and T1wCE folders\n        flair_path = os.path.join(train_dir, patient_folder, \"FLAIR\")\n        t1wce_path = os.path.join(train_dir, patient_folder, \"T1wCE\")\n\n        # Count images if the folder exists and is not empty\n        if os.path.isdir(flair_path):\n            flair_count = count_images_in_folder(flair_path)\n        if os.path.isdir(t1wce_path):\n            t1wce_count = count_images_in_folder(t1wce_path)\n\n        # Append the data to the analysis list\n        analysis_data.append({\n            \"Patient ID\": patient_id,\n            \"FLAIR Count\": flair_count,\n            \"T1wCE Count\": t1wce_count,\n            \"Tumor Status\": \"Yes\" if tumor_status == 1 else \"No\"\n        })\n    else:\n        # Record IDs skipped due to missing labels\n        skipped_ids.append({\"Patient ID\": patient_id, \"Reason\": \"No label in CSV\"})\n\n# Identify and record IDs from CSV with no corresponding image folder\nfor csv_id in all_csv_ids:\n    if csv_id not in all_patient_folders:\n        skipped_ids.append({\"Patient ID\": csv_id, \"Reason\": \"No image folder in train directory\"})\n\n# Create DataFrames to display analysis data and skipped IDs\nanalysis_df = pd.DataFrame(analysis_data)\nskipped_ids_df = pd.DataFrame(skipped_ids)\n\n# Display the tables\nprint(\"Analysis Table:\")\nprint(analysis_df)\n\nprint(\"\\nSkipped IDs Table:\")\nprint(skipped_ids_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:16:49.997111Z","iopub.execute_input":"2024-12-22T12:16:49.997593Z","iopub.status.idle":"2024-12-22T12:16:50.869644Z","shell.execute_reply.started":"2024-12-22T12:16:49.997537Z","shell.execute_reply":"2024-12-22T12:16:50.868787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# Define the directories and file paths\ntrain_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\ncsv_path = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\"\n\n# Load the CSV file\nlabels_df = pd.read_csv(csv_path)\n\n# Process patient IDs to remove leading zeros and use the same format in both places\nlabels_df['BraTS21ID'] = labels_df['BraTS21ID'].apply(lambda x: str(x).zfill(5))\n\n# Initialize lists for analysis data and skipped IDs\nanalysis_data = []\nskipped_ids = []\n\n# Function to count images in a given modality folder\ndef count_images_in_folder(folder_path):\n    return len([img for img in os.listdir(folder_path) if img.endswith(\".dcm\")])\n\n# Get all patient IDs from both CSV and folders for cross-referencing\nall_patient_folders = set(os.listdir(train_dir))\nall_csv_ids = set(labels_df['BraTS21ID'].values)\n\n# Iterate through each patient folder in the training directory\nfor patient_folder in all_patient_folders:\n    patient_id = patient_folder.zfill(5)  # Ensure ID format consistency\n\n    # Check if the patient ID exists in the labels DataFrame\n    if patient_id in all_csv_ids:\n        # Find corresponding label for the patient\n        label_row = labels_df[labels_df['BraTS21ID'] == patient_id]\n        tumor_status = label_row['MGMT_value'].values[0]\n\n        # Initialize counts for FLAIR and T1wCE\n        flair_count, t1wce_count = 0, 0\n\n        # Define paths for FLAIR and T1wCE folders\n        flair_path = os.path.join(train_dir, patient_folder, \"FLAIR\")\n        t1wce_path = os.path.join(train_dir, patient_folder, \"T1wCE\")\n\n        # Count images if the folder exists and is not empty\n        if os.path.isdir(flair_path):\n            flair_count = count_images_in_folder(flair_path)\n        if os.path.isdir(t1wce_path):\n            t1wce_count = count_images_in_folder(t1wce_path)\n\n        # Append the data to the analysis list\n        analysis_data.append({\n            \"Patient ID\": patient_id,\n            \"FLAIR Count\": flair_count,\n            \"T1wCE Count\": t1wce_count,\n            \"Tumor Status\": \"Yes\" if tumor_status == 1 else \"No\"\n        })\n    else:\n        # Record IDs skipped due to missing labels\n        skipped_ids.append({\"Patient ID\": patient_id, \"Reason\": \"No label in CSV\"})\n\n# Identify and record IDs from CSV with no corresponding image folder\nfor csv_id in all_csv_ids:\n    if csv_id not in all_patient_folders:\n        skipped_ids.append({\"Patient ID\": csv_id, \"Reason\": \"No image folder in train directory\"})\n\n# Create DataFrames to display analysis data and skipped IDs\nanalysis_df = pd.DataFrame(analysis_data)\nskipped_ids_df = pd.DataFrame(skipped_ids)\n\n# Display the tables\nprint(\"Analysis Table:\")\nprint(analysis_df)\n\n# Check if there are any skipped IDs\nif skipped_ids:\n    print(\"\\nSkipped IDs Table:\")\n    print(skipped_ids_df)\nelse:\n    print(\"\\nNo missing data found. All patient IDs have corresponding labels and image folders.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:16:50.871173Z","iopub.execute_input":"2024-12-22T12:16:50.871733Z","iopub.status.idle":"2024-12-22T12:16:51.719339Z","shell.execute_reply.started":"2024-12-22T12:16:50.871692Z","shell.execute_reply":"2024-12-22T12:16:51.718408Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom IPython.display import display, HTML\n\n# Define the directories and file paths\ntrain_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\ncsv_path = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\"\n\n# Load the CSV file\nlabels_df = pd.read_csv(csv_path)\n\n# Process patient IDs to remove leading zeros and use consistent format in both sources\nlabels_df['BraTS21ID'] = labels_df['BraTS21ID'].apply(lambda x: str(x).zfill(5))\n\n# Initialize lists for analysis data and skipped IDs\nanalysis_data = []\nskipped_ids = []\n\n# Function to count images in a given modality folder\ndef count_images_in_folder(folder_path):\n    return len([img for img in os.listdir(folder_path) if img.endswith(\".dcm\")])\n\n# Get all patient IDs from both CSV and folders for cross-referencing\nall_patient_folders = set(os.listdir(train_dir))\nall_csv_ids = set(labels_df['BraTS21ID'].values)\n\n# Iterate through each patient folder in the training directory\nfor patient_folder in all_patient_folders:\n    patient_id = patient_folder.zfill(5)  # Ensure consistent ID format\n\n    # Check if the patient ID exists in the labels DataFrame\n    if patient_id in all_csv_ids:\n        # Retrieve tumor status for the patient\n        tumor_status = labels_df.loc[labels_df['BraTS21ID'] == patient_id, 'MGMT_value'].values[0]\n\n        # Initialize counts for FLAIR and T1wCE\n        flair_count, t1wce_count = 0, 0\n\n        # Define paths for FLAIR and T1wCE folders\n        flair_path = os.path.join(train_dir, patient_folder, \"FLAIR\")\n        t1wce_path = os.path.join(train_dir, patient_folder, \"T1wCE\")\n\n        # Count images if folders exist and are not empty\n        if os.path.isdir(flair_path):\n            flair_count = count_images_in_folder(flair_path)\n        if os.path.isdir(t1wce_path):\n            t1wce_count = count_images_in_folder(t1wce_path)\n\n        # Append formatted data for the analysis table\n        analysis_data.append({\n            \"Patient ID\": patient_id,\n            \"FLAIR Count\": flair_count,\n            \"T1wCE Count\": t1wce_count,\n            \"Tumor Status\": \"Yes\" if tumor_status == 1 else \"No\"\n        })\n    else:\n        # Record IDs skipped due to missing labels\n        skipped_ids.append({\"Patient ID\": patient_id, \"Reason\": \"No label in CSV\"})\n\n# Identify and record IDs from CSV with no corresponding image folder\nfor csv_id in all_csv_ids:\n    if csv_id not in all_patient_folders:\n        skipped_ids.append({\"Patient ID\": csv_id, \"Reason\": \"No image folder in train directory\"})\n\n# Create DataFrames for analysis and skipped IDs tables\nanalysis_df = pd.DataFrame(analysis_data)\nskipped_ids_df = pd.DataFrame(skipped_ids)\n\n# Sort the analysis DataFrame by 'Patient ID' in ascending order\nanalysis_df = analysis_df.sort_values(by=\"Patient ID\").reset_index(drop=True)\n\n# Add a numbering column to the analysis table\nanalysis_df.index = range(1, len(analysis_df) + 1)\nanalysis_df.index.name = \"Entry Number\"\n\n# Display the tables with enhanced formatting\nprint(\"\\n--- Analysis Table ---\\n\")\nif not analysis_df.empty:\n    # Center-align the table content and make it wider for improved readability\n    display(HTML(analysis_df.to_html(index=True, justify='center', border=1)))\n\n# Check if there are any skipped IDs and display accordingly\nif skipped_ids:\n    print(\"\\n--- Skipped IDs Table ---\\n\")\n    # Sort skipped IDs by 'Patient ID' and apply similar formatting\n    skipped_ids_df = skipped_ids_df.sort_values(by=\"Patient ID\").reset_index(drop=True)\n    skipped_ids_df.index = range(1, len(skipped_ids_df) + 1)\n    skipped_ids_df.index.name = \"Entry Number\"\n    display(HTML(skipped_ids_df.to_html(index=True, justify='center', border=1)))\nelse:\n    print(\"\\nNo missing data found. All patient IDs have corresponding labels and image folders.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:16:51.722992Z","iopub.execute_input":"2024-12-22T12:16:51.723668Z","iopub.status.idle":"2024-12-22T12:16:52.619773Z","shell.execute_reply.started":"2024-12-22T12:16:51.723638Z","shell.execute_reply":"2024-12-22T12:16:52.618929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Ensure that Seaborn styles are used for a polished look\nsns.set(style=\"whitegrid\")\n\n# 1. Distribution of FLAIR Image Counts\nplt.figure(figsize=(10, 6))\nsns.histplot(analysis_df['FLAIR Count'], color='blue', kde=True, label='FLAIR Count')\nplt.title(\"Distribution of FLAIR Image Counts\")\nplt.xlabel(\"Number of Images\")\nplt.ylabel(\"Frequency\")\nplt.legend()\nplt.show()\n\n# 2. Tumor vs. Non-Tumor Patients\n# Distribution of patients based on tumor status\nflair_tumor_counts = analysis_df['Tumor Status'].value_counts()\nplt.figure(figsize=(8, 6))\nplt.pie(flair_tumor_counts, labels=['No Tumor', 'Tumor'], autopct='%1.1f%%', colors=['lightblue', 'salmon'], startangle=140)\nplt.title(\"Distribution of Tumor and Non-Tumor Patients\")\nplt.show()\n\n# 3. Box Plot of FLAIR Image Counts by Tumor Status\nplt.figure(figsize=(10, 6))\nsns.boxplot(x='Tumor Status', y='FLAIR Count', data=analysis_df, palette=\"Blues\")\nplt.title(\"FLAIR Image Count by Tumor Status\")\nplt.xlabel(\"Tumor Status\")\nplt.ylabel(\"FLAIR Image Count\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:16:52.620853Z","iopub.execute_input":"2024-12-22T12:16:52.621116Z","iopub.status.idle":"2024-12-22T12:16:53.354421Z","shell.execute_reply.started":"2024-12-22T12:16:52.621090Z","shell.execute_reply":"2024-12-22T12:16:53.353792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Assuming analysis_df is already created with patient data, here's the updated code\n\n# Convert 'Tumor Status' to numeric: 1 for 'Yes' and 0 for 'No'\nanalysis_df['Tumor Status Numeric'] = analysis_df['Tumor Status'].apply(lambda x: 1 if x == 'Yes' else 0)\n\n# 1. Correlation Heatmap\nplt.figure(figsize=(10, 8))\ncorr_matrix = analysis_df[['FLAIR Count', 'T1wCE Count', 'Tumor Status Numeric']].corr()\nsns.heatmap(corr_matrix, annot=True, cmap='coolwarm', fmt='.2f')\nplt.title(\"Correlation Heatmap between Image Counts and Tumor Status\")\nplt.show()\n\n# 2. Tumor Status Distribution\nplt.figure(figsize=(8, 6))\nsns.countplot(x='Tumor Status', data=analysis_df, palette='Set2')\nplt.title(\"Tumor vs Non-Tumor Distribution\")\nplt.xlabel(\"Tumor Status\")\nplt.ylabel(\"Number of Patients\")\nplt.show()\n\n# 3. FLAIR Count vs Tumor Status (Bar Plot)\nplt.figure(figsize=(10, 6))\nsns.boxplot(x='Tumor Status', y='FLAIR Count', data=analysis_df, palette='Set1')\nplt.title(\"FLAIR Count vs Tumor Status\")\nplt.xlabel(\"Tumor Status\")\nplt.ylabel(\"FLAIR Image Count\")\nplt.show()\n\n# 4. T1wCE Count vs Tumor Status (Bar Plot)\nplt.figure(figsize=(10, 6))\nsns.boxplot(x='Tumor Status', y='T1wCE Count', data=analysis_df, palette='Set1')\nplt.title(\"T1wCE Count vs Tumor Status\")\nplt.xlabel(\"Tumor Status\")\nplt.ylabel(\"T1wCE Image Count\")\nplt.show()\n\n# 5. Tumor Status vs Total Image Count (FLAIR + T1wCE)\nanalysis_df['Total Image Count'] = analysis_df['FLAIR Count'] + analysis_df['T1wCE Count']\nplt.figure(figsize=(8, 6))\nsns.boxplot(x='Tumor Status', y='Total Image Count', data=analysis_df, palette='Set2')\nplt.title(\"Total Image Count vs Tumor Status\")\nplt.xlabel(\"Tumor Status\")\nplt.ylabel(\"Total Image Count (FLAIR + T1wCE)\")\nplt.show()\n\n# 6. Pairplot for FLAIR, T1wCE Count and Tumor Status (Numeric)\nsns.pairplot(analysis_df[['FLAIR Count', 'T1wCE Count', 'Tumor Status Numeric']], hue='Tumor Status Numeric', palette='coolwarm')\nplt.suptitle(\"Pairplot of FLAIR and T1wCE Counts with Tumor Status\", y=1.02)\nplt.show()\n\n# 7. Histograms for Image Counts (FLAIR and T1wCE)\nplt.figure(figsize=(14, 6))\nplt.subplot(1, 2, 1)\nsns.histplot(analysis_df['FLAIR Count'], kde=True, color='blue', bins=30)\nplt.title(\"Distribution of FLAIR Image Count\")\nplt.xlabel(\"FLAIR Count\")\nplt.ylabel(\"Frequency\")\n\nplt.subplot(1, 2, 2)\nsns.histplot(analysis_df['T1wCE Count'], kde=True, color='green', bins=30)\nplt.title(\"Distribution of T1wCE Image Count\")\nplt.xlabel(\"T1wCE Count\")\nplt.ylabel(\"Frequency\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:16:53.355336Z","iopub.execute_input":"2024-12-22T12:16:53.355605Z","iopub.status.idle":"2024-12-22T12:16:56.506346Z","shell.execute_reply.started":"2024-12-22T12:16:53.355579Z","shell.execute_reply":"2024-12-22T12:16:56.505518Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nfrom sklearn.utils import shuffle\n\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.initializers import TruncatedNormal\nfrom tensorflow.keras.losses import CategoricalCrossentropy\nfrom tensorflow.keras.metrics import CategoricalAccuracy\nfrom tensorflow.keras.layers import Input, Dense","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:16:56.507625Z","iopub.execute_input":"2024-12-22T12:16:56.508251Z","iopub.status.idle":"2024-12-22T12:16:56.515248Z","shell.execute_reply.started":"2024-12-22T12:16:56.508205Z","shell.execute_reply":"2024-12-22T12:16:56.514375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install pydicom","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:16:56.516292Z","iopub.execute_input":"2024-12-22T12:16:56.516575Z","iopub.status.idle":"2024-12-22T12:17:05.184299Z","shell.execute_reply.started":"2024-12-22T12:16:56.516549Z","shell.execute_reply":"2024-12-22T12:17:05.183356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pydicom\nfrom skimage.transform import resize  # For resizing images\n\n# Path to the \"train\" folder\ntrain_folder = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\n\n# Desired shape for the 3D CNN (Depth, Height, Width, Channels)\ncnn_input_shape = (128, 64, 64, 1)\n\n# List to store patient 3D data\ncnn_input = []\n\n# Counter to keep track of processed patient IDs\npatient_count = 0\nmax_patients = 550  # Limit to the first 100 patients\n\n# Iterate through each patient's folder\nfor patient_id in sorted(os.listdir(train_folder)):  # Ensure consistent order with `sorted`\n    if patient_count >= max_patients:  # Stop after processing 100 patients\n        break\n    \n    patient_path = os.path.join(train_folder, patient_id)\n    \n    # Path to the FLAIR folder inside the patient folder\n    flair_folder = os.path.join(patient_path, \"FLAIR\")\n    \n    if os.path.isdir(flair_folder):\n        # Read all DICOM files in the FLAIR folder\n        slices = []\n        for dcm_file in sorted(os.listdir(flair_folder)):  # Sorting ensures correct slice order\n            dcm_path = os.path.join(flair_folder, dcm_file)\n            if dcm_file.endswith(\".dcm\"):\n                ds = pydicom.dcmread(dcm_path)\n                slices.append(ds.pixel_array)  # Extract the pixel array\n        \n        # Stack the slices into a 3D array if slices are available\n        if slices:\n            flair_stack = np.stack(slices, axis=-1)  # Stack along the last axis\n            \n            # Normalize pixel values to [0, 1]\n            flair_stack = flair_stack.astype(np.float32) / np.max(flair_stack)\n            \n            # Resize each slice to the target height and width\n            resized_stack = np.array([\n                resize(slice, cnn_input_shape[1:3], mode='constant', anti_aliasing=True)\n                for slice in flair_stack.transpose(2, 0, 1)  # (Depth, Height, Width)\n            ])\n            \n            # Adjust depth to match the target depth\n            current_depth = resized_stack.shape[0]\n            if current_depth < cnn_input_shape[0]:\n                # Pad with zeros if current depth is less than target\n                pad_width = cnn_input_shape[0] - current_depth\n                padding = ((0, pad_width), (0, 0), (0, 0))  # Pad depth only\n                resized_stack = np.pad(resized_stack, padding, mode='constant')\n            else:\n                # Crop if current depth is greater than target\n                resized_stack = resized_stack[:cnn_input_shape[0], :, :]\n            \n            # Add the channel dimension (for grayscale)\n            resized_stack = np.expand_dims(resized_stack, axis=-1)\n            \n            # Add to the CNN input list\n            cnn_input.append(resized_stack)\n            patient_count += 1  # Increment the counter\n            print(f\"Processed Patient ID: {patient_id}, Preprocessed Shape: {resized_stack.shape}\")\n\n# Convert the list to a 5D NumPy array for CNN input\ncnn_input = np.array(cnn_input)  # Shape: (num_patients, Depth, Height, Width, Channels)\n\nprint(f\"Final CNN Input Shape: {cnn_input.shape}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:17:05.186273Z","iopub.execute_input":"2024-12-22T12:17:05.186615Z","iopub.status.idle":"2024-12-22T12:38:25.404205Z","shell.execute_reply.started":"2024-12-22T12:17:05.186585Z","shell.execute_reply":"2024-12-22T12:38:25.403272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nimport os\n\n# Path to the CSV file and training folder\ncsv_file = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\"\ntrain_folder = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\n\n# Load the CSV file\nlabels_df = pd.read_csv(csv_file)\n\n# Ensure Patient IDs are zero-padded to match folder names\nlabels_df['BraTS21ID'] = labels_df['BraTS21ID'].apply(lambda x: str(x).zfill(5))\n\n# Extract Patient IDs and Labels into a dictionary\npatient_labels = {row['BraTS21ID']: row['MGMT_value'] for _, row in labels_df.iterrows()}\n\n# Prepare aligned data for CNN input and labels\naligned_cnn_input = []\naligned_labels = []\n\n# Ensure consistent ordering of patient IDs as in preprocessing\npatient_ids = sorted(os.listdir(train_folder))[:len(cnn_input)]  # Use same order and limit as cnn_input\n\n# Match labels with data and filter only valid patients\nfor patient_id, data in zip(patient_ids, cnn_input):  \n    if patient_id in patient_labels:\n        aligned_cnn_input.append(data)\n        aligned_labels.append(patient_labels[patient_id])\n    else:\n        print(f\"Warning: No label found for Patient ID: {patient_id}\")\n\n# Check if aligned_cnn_input is still empty\nif len(aligned_cnn_input) == 0:\n    raise ValueError(\"No valid patient data found. Ensure patient IDs in folders match those in the CSV file.\")\n\n# Convert aligned data and labels to NumPy arrays\naligned_cnn_input = np.array(aligned_cnn_input)\naligned_labels = np.array(aligned_labels)\n\n# Split the data into training and validation sets\nX_train, X_val, y_train, y_val = train_test_split(\n    aligned_cnn_input, aligned_labels, test_size=0.2, random_state=42\n)\n\n# Output shapes for debugging\nprint(f\"X_train shape: {X_train.shape}, y_train shape: {y_train.shape}\")\nprint(f\"X_val shape: {X_val.shape}, y_val shape: {y_val.shape}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:38:25.405634Z","iopub.execute_input":"2024-12-22T12:38:25.406259Z","iopub.status.idle":"2024-12-22T12:38:28.137436Z","shell.execute_reply.started":"2024-12-22T12:38:25.406217Z","shell.execute_reply":"2024-12-22T12:38:28.136515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Assuming `aligned_cnn_input` is your 3D CNN input data\n# and `aligned_labels` is the corresponding label array\nprint(f\"Data Shape: {aligned_cnn_input.shape}, Labels Shape: {aligned_labels.shape}\")\n\n# Ensure labels are numpy arrays\naligned_labels = np.array(aligned_labels)\n\n# Split the data into training and validation sets (stratified to ensure class balance)\nX_train, X_val, y_train, y_val = train_test_split(\n    aligned_cnn_input, aligned_labels, test_size=0.2, random_state=42, stratify=aligned_labels\n)\n\n# Display shapes for debugging and confirmation\nprint(f\"X_train shape: {X_train.shape}, y_train shape: {y_train.shape}\")\nprint(f\"X_val shape: {X_val.shape}, y_val shape: {y_val.shape}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:38:28.138839Z","iopub.execute_input":"2024-12-22T12:38:28.139503Z","iopub.status.idle":"2024-12-22T12:38:29.378080Z","shell.execute_reply.started":"2024-12-22T12:38:28.139440Z","shell.execute_reply":"2024-12-22T12:38:29.377171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from keras import layers, models\nfrom keras.optimizers import Adam  # Import Adam optimizer with adjustable learning rate\n\n# Define the 3D CNN model\ndef build_3d_cnn(input_shape):\n    model = models.Sequential()\n    model.add(layers.Conv3D(filters=32, kernel_size=(3, 3, 3), activation='relu', input_shape=input_shape))\n    model.add(layers.MaxPooling3D(pool_size=(2, 2, 2)))\n    model.add(layers.Conv3D(filters=64, kernel_size=(3, 3, 3), activation='relu'))\n    model.add(layers.MaxPooling3D(pool_size=(2, 2, 2)))\n    model.add(layers.Conv3D(filters=128, kernel_size=(3, 3, 3), activation='relu'))\n    model.add(layers.MaxPooling3D(pool_size=(2, 2, 2)))\n    model.add(layers.Flatten())\n    model.add(layers.Dense(256, activation='relu'))\n    model.add(layers.Dropout(0.5))\n    model.add(layers.Dense(1, activation='sigmoid'))\n    return model\n\n# Use the pre-defined cnn_input_shape from preprocessing\ninput_shape = (128, 64, 64, 1)  # (514, 128, 128, 1) or whatever was defined in preprocessing\nmodel = build_3d_cnn(input_shape)\n\n# Compile the model with a learning rate of 0.0005\noptimizer = Adam(learning_rate=0.0005)  # Set learning rate to 0.0005\n\n# optimizer = Adam(\n#     learning_rate=5e-5,\n#     epsilon=1e-08,\n#     decay=0.01,\n#     clipnorm=1.0)\n\n# loss = CategoricalCrossentropy(from_logits=True)\n# metric = CategoricalAccuracy('bal_accuracy')\n# model.compile(optimizer = optimizer, loss = loss, metrics = [metric])\n\nmodel.compile(optimizer=optimizer, loss='binary_crossentropy', metrics=['accuracy'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:38:29.379372Z","iopub.execute_input":"2024-12-22T12:38:29.379749Z","iopub.status.idle":"2024-12-22T12:38:30.427020Z","shell.execute_reply.started":"2024-12-22T12:38:29.379720Z","shell.execute_reply":"2024-12-22T12:38:30.426063Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:38:30.428341Z","iopub.execute_input":"2024-12-22T12:38:30.429120Z","iopub.status.idle":"2024-12-22T12:38:30.453042Z","shell.execute_reply.started":"2024-12-22T12:38:30.429078Z","shell.execute_reply":"2024-12-22T12:38:30.452217Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the model with validation\nhistory = model.fit(X_train, y_train, validation_data=(X_val, y_val), epochs=10, batch_size=16)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:38:30.454130Z","iopub.execute_input":"2024-12-22T12:38:30.454389Z","iopub.status.idle":"2024-12-22T12:41:03.136992Z","shell.execute_reply.started":"2024-12-22T12:38:30.454363Z","shell.execute_reply":"2024-12-22T12:41:03.136331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Plot accuracy\nplt.figure(figsize=(12, 4))\n\n# Accuracy plot\nplt.subplot(1, 2, 1)\nplt.plot(history.history['accuracy'], label='Training Accuracy')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\nplt.title('Accuracy Over Epochs')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\n\n# Loss plot\nplt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='Training Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Loss Over Epochs')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:41:03.138554Z","iopub.execute_input":"2024-12-22T12:41:03.139233Z","iopub.status.idle":"2024-12-22T12:41:03.837665Z","shell.execute_reply.started":"2024-12-22T12:41:03.139190Z","shell.execute_reply":"2024-12-22T12:41:03.836769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Plot accuracy\nplt.figure(figsize=(12, 4))\n\n# Accuracy plot\nplt.subplot(1, 2, 1)\nplt.plot(history.history['accuracy'], label='Training Accuracy')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\nplt.title('Accuracy Over Epochs')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\n\n# Loss plot\nplt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='Training Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Loss Over Epochs')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T12:41:03.838705Z","iopub.execute_input":"2024-12-22T12:41:03.838977Z","iopub.status.idle":"2024-12-22T12:41:04.549397Z","shell.execute_reply.started":"2024-12-22T12:41:03.838950Z","shell.execute_reply":"2024-12-22T12:41:04.548543Z"}},"outputs":[],"execution_count":null}]}