{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":29653,"databundleVersionId":2420395,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 1-Kaggle Input","metadata":{}},{"cell_type":"code","source":"import os\n\n# List all files in the input directory and subdirectories\ninput_dir = '/kaggle/input'\nfile_paths = []\n\nfor dirname, _, filenames in os.walk(input_dir):\n    for filename in filenames:\n        file_paths.append(os.path.join(dirname, filename))\n\n# Print confirmation message\nif file_paths:\n    print(\"Input adde1`1`qqqed successfully.\")\nelse:\n    print(\"No input files found.\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-23T18:49:19.284190Z","iopub.execute_input":"2024-11-23T18:49:19.284470Z","iopub.status.idle":"2024-11-23T19:06:58.687595Z","shell.execute_reply.started":"2024-11-23T18:49:19.284444Z","shell.execute_reply":"2024-11-23T19:06:58.686604Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2-Count Images and Patients","metadata":{}},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\n\n# Define the directory paths\ntrain_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\ntest_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/test\"\n\n# Function to count patients and images\ndef count_patients_and_images(directory):\n    patient_count = 0\n    image_count = 0\n    \n    for patient_folder in os.listdir(directory):\n        patient_path = os.path.join(directory, patient_folder)\n        \n        # Check if it's a folder\n        if os.path.isdir(patient_path):\n            patient_count += 1\n            \n            # Count images in each modality folder\n            for modality in [\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"]:\n                modality_path = os.path.join(patient_path, modality)\n                if os.path.isdir(modality_path):\n                    image_count += len([img for img in os.listdir(modality_path) if img.endswith(\".dcm\")])\n\n    return patient_count, image_count\n\n# Get counts for training and testing sets\ntrain_patients, train_images = count_patients_and_images(train_dir)\ntest_patients, test_images = count_patients_and_images(test_dir)\n\n# Plotting the results\nfig, ax = plt.subplots(1, 2, figsize=(12, 6))\n\n# Bar plot for number of patients\nax[0].bar([\"Train\", \"Test\"], [train_patients, test_patients], color=[\"blue\", \"orange\"])\nax[0].set_title(\"Total Number of Patients\")\nax[0].set_ylabel(\"Number of Patients\")\n\n# Bar plot for number of images\nax[1].bar([\"Train\", \"Test\"], [train_images, test_images], color=[\"blue\", \"orange\"])\nax[1].set_title(\"Total Number of Images\")\nax[1].set_ylabel(\"Number of Images\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T19:13:17.478605Z","iopub.execute_input":"2024-11-23T19:13:17.478950Z","iopub.status.idle":"2024-11-23T19:13:21.414248Z","shell.execute_reply.started":"2024-11-23T19:13:17.478914Z","shell.execute_reply":"2024-11-23T19:13:21.413414Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3-Count Total no. of Images in each Modality","metadata":{}},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\n\n# Define the directory paths\ntrain_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\ntest_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/test\"\n\n# Function to count images for each modality in each patient folder\ndef count_images_per_modality(directory):\n    modalities = [\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"]\n    modality_counts = {modality: 0 for modality in modalities}\n    \n    for patient_folder in os.listdir(directory):\n        patient_path = os.path.join(directory, patient_folder)\n        \n        # Check if it's a folder\n        if os.path.isdir(patient_path):\n            # Count images in each modality folder\n            for modality in modalities:\n                modality_path = os.path.join(patient_path, modality)\n                if os.path.isdir(modality_path):\n                    modality_counts[modality] += len([img for img in os.listdir(modality_path) if img.endswith(\".dcm\")])\n    \n    return modality_counts\n\n# Get modality image counts for training and testing sets\ntrain_modality_counts = count_images_per_modality(train_dir)\ntest_modality_counts = count_images_per_modality(test_dir)\n\n# Plotting the results\nfig, ax = plt.subplots(1, 2, figsize=(14, 6))\n\n# Bar plot for train dataset\nax[0].bar(train_modality_counts.keys(), train_modality_counts.values(), color=\"blue\")\nax[0].set_title(\"Number of Images per Modality in Training Dataset\")\nax[0].set_xlabel(\"Modality\")\nax[0].set_ylabel(\"Number of Images\")\n\n# Bar plot for test dataset\nax[1].bar(test_modality_counts.keys(), test_modality_counts.values(), color=\"orange\")\nax[1].set_title(\"Number of Images per Modality in Testing Dataset\")\nax[1].set_xlabel(\"Modality\")\nax[1].set_ylabel(\"Number of Images\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T19:13:31.195472Z","iopub.execute_input":"2024-11-23T19:13:31.195802Z","iopub.status.idle":"2024-11-23T19:13:33.263094Z","shell.execute_reply.started":"2024-11-23T19:13:31.195772Z","shell.execute_reply":"2024-11-23T19:13:33.262302Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4-Total Images in each Modality of Train/Test Set","metadata":{}},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\n\n# Define the directory paths\ntrain_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\ntest_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/test\"\n\n# Function to count images for each modality in each patient folder\ndef count_images_per_modality(directory):\n    modalities = [\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"]\n    modality_counts = {modality: 0 for modality in modalities}\n    \n    for patient_folder in os.listdir(directory):\n        patient_path = os.path.join(directory, patient_folder)\n        \n        # Check if it's a folder\n        if os.path.isdir(patient_path):\n            # Count images in each modality folder\n            for modality in modalities:\n                modality_path = os.path.join(patient_path, modality)\n                if os.path.isdir(modality_path):\n                    modality_counts[modality] += len([img for img in os.listdir(modality_path) if img.endswith(\".dcm\")])\n    \n    return modality_counts\n\n# Get modality image counts for training and testing sets\ntrain_modality_counts = count_images_per_modality(train_dir)\ntest_modality_counts = count_images_per_modality(test_dir)\n\n# Plotting the results with improvements\nfig, ax = plt.subplots(1, 2, figsize=(16, 8))\ncolors = [\"#4e79a7\", \"#f28e2b\", \"#e15759\", \"#76b7b2\"]  # Different colors for each modality\n\n# Bar plot for train dataset\nax[0].bar(train_modality_counts.keys(), train_modality_counts.values(), color=colors)\nax[0].set_title(\"Number of Images per Modality in Training Dataset\", fontsize=14, weight='bold')\nax[0].set_xlabel(\"Modality\", fontsize=12)\nax[0].set_ylabel(\"Number of Images\", fontsize=12)\n\n# Adding annotations to the bars\nfor i, (modality, count) in enumerate(train_modality_counts.items()):\n    ax[0].text(i, count + 50, str(count), ha='center', va='bottom', fontsize=10)\n\n# Bar plot for test dataset\nax[1].bar(test_modality_counts.keys(), test_modality_counts.values(), color=colors)\nax[1].set_title(\"Number of Images per Modality in Testing Dataset\", fontsize=14, weight='bold')\nax[1].set_xlabel(\"Modality\", fontsize=12)\nax[1].set_ylabel(\"Number of Images\", fontsize=12)\n\n# Adding annotations to the bars\nfor i, (modality, count) in enumerate(test_modality_counts.items()):\n    ax[1].text(i, count + 50, str(count), ha='center', va='bottom', fontsize=10)\n\n# Adding a legend\nfig.legend(train_modality_counts.keys(), loc=\"upper center\", ncol=4, fontsize=12, frameon=False)\n\nplt.tight_layout(rect=[0, 0, 1, 0.95])  # Adjust layout to make space for legend\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T19:13:49.212885Z","iopub.execute_input":"2024-11-23T19:13:49.213210Z","iopub.status.idle":"2024-11-23T19:13:51.237100Z","shell.execute_reply.started":"2024-11-23T19:13:49.213180Z","shell.execute_reply":"2024-11-23T19:13:51.236034Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5-Patient with the Maximum number of Images for each Modality","metadata":{}},{"cell_type":"code","source":"import os\n\n# Define the directory path\ntrain_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\n\n# Function to find the patient with the maximum number of images for each modality\ndef find_max_images_per_modality(directory):\n    # Dictionary to store the max count for each modality\n    max_images_per_modality = {modality: {\"patient_id\": None, \"image_count\": 0} for modality in [\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"]}\n    \n    for patient_folder in os.listdir(directory):\n        patient_path = os.path.join(directory, patient_folder)\n        \n        # Check if it's a folder\n        if os.path.isdir(patient_path):\n            # Count images in each modality folder\n            for modality in [\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"]:\n                modality_path = os.path.join(patient_path, modality)\n                if os.path.isdir(modality_path):\n                    image_count = len([img for img in os.listdir(modality_path) if img.endswith(\".dcm\")])\n                    \n                    # Update max if this patient has more images for the modality\n                    if image_count > max_images_per_modality[modality][\"image_count\"]:\n                        max_images_per_modality[modality] = {\"patient_id\": patient_folder, \"image_count\": image_count}\n    \n    return max_images_per_modality\n\n# Get the patient with the maximum images for each modality\nmax_images_per_modality = find_max_images_per_modality(train_dir)\n\n# Display the results\nprint(\"Patient IDs with Maximum Number of Images per Modality:\")\nfor modality, data in max_images_per_modality.items():\n    print(f\"Modality: {modality}, Patient ID: {data['patient_id']}, Image Count: {data['image_count']}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T19:14:00.595318Z","iopub.execute_input":"2024-11-23T19:14:00.595975Z","iopub.status.idle":"2024-11-23T19:14:02.041089Z","shell.execute_reply.started":"2024-11-23T19:14:00.595942Z","shell.execute_reply":"2024-11-23T19:14:02.040239Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6-CSV Analysis - Percentage of Tumor and Non Tumor","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\n# Load the CSV file\nfile_path = '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv'  # Replace with your actual file path\ndata = pd.read_csv(file_path)\n\n# Count the number of patients with and without tumors\ntumor_counts = data['MGMT_value'].value_counts()\n\n# Plotting\nlabels = ['No Tumor', 'Tumor']\nsizes = [tumor_counts.get(0, 0), tumor_counts.get(1, 0)]  # Get counts for 0 and 1, default to 0 if not found\ncolors = ['lightblue', 'salmon']\nexplode = (0.1, 0)  # explode 1st slice (No Tumor)\n\nplt.figure(figsize=(8, 6))\nplt.pie(sizes, explode=explode, labels=labels, colors=colors,\n        autopct='%1.1f%%', shadow=True, startangle=140)\n\nplt.title('Distribution of Patients with and without Tumors')\nplt.axis('equal')  # Equal aspect ratio ensures that pie chart is circular.\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T19:14:06.723580Z","iopub.execute_input":"2024-11-23T19:14:06.723914Z","iopub.status.idle":"2024-11-23T19:14:07.224382Z","shell.execute_reply.started":"2024-11-23T19:14:06.723885Z","shell.execute_reply":"2024-11-23T19:14:07.222421Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 7-Analyse both CSV and Training Data\nIn this we are trying to find total number of images in FLAIR and T1wCE Modality, Its Tumor Status for each patient.","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# Define the directories and file paths\ntrain_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\ncsv_path = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\"\n\n# Load the CSV file\nlabels_df = pd.read_csv(csv_path)\n\n# Process patient IDs to remove leading zeros and use the same format in both places\nlabels_df['BraTS21ID'] = labels_df['BraTS21ID'].apply(lambda x: str(x).zfill(5))\n\n# Initialize lists for analysis data and skipped IDs\nanalysis_data = []\nskipped_ids = []\n\n# Function to count images in a given modality folder\ndef count_images_in_folder(folder_path):\n    return len([img for img in os.listdir(folder_path) if img.endswith(\".dcm\")])\n\n# Get all patient IDs from both CSV and folders for cross-referencing\nall_patient_folders = set(os.listdir(train_dir))\nall_csv_ids = set(labels_df['BraTS21ID'].values)\n\n# Iterate through each patient folder in the training directory\nfor patient_folder in all_patient_folders:\n    patient_id = patient_folder.zfill(5)  # Ensure ID format consistency\n\n    # Check if the patient ID exists in the labels DataFrame\n    if patient_id in all_csv_ids:\n        # Find corresponding label for the patient\n        label_row = labels_df[labels_df['BraTS21ID'] == patient_id]\n        tumor_status = label_row['MGMT_value'].values[0]\n\n        # Initialize counts for FLAIR and T1wCE\n        flair_count, t1wce_count = 0, 0\n\n        # Define paths for FLAIR and T1wCE folders\n        flair_path = os.path.join(train_dir, patient_folder, \"FLAIR\")\n        t1wce_path = os.path.join(train_dir, patient_folder, \"T1wCE\")\n\n        # Count images if the folder exists and is not empty\n        if os.path.isdir(flair_path):\n            flair_count = count_images_in_folder(flair_path)\n        if os.path.isdir(t1wce_path):\n            t1wce_count = count_images_in_folder(t1wce_path)\n\n        # Append the data to the analysis list\n        analysis_data.append({\n            \"Patient ID\": patient_id,\n            \"FLAIR Count\": flair_count,\n            \"T1wCE Count\": t1wce_count,\n            \"Tumor Status\": \"Yes\" if tumor_status == 1 else \"No\"\n        })\n    else:\n        # Record IDs skipped due to missing labels\n        skipped_ids.append({\"Patient ID\": patient_id, \"Reason\": \"No label in CSV\"})\n\n# Identify and record IDs from CSV with no corresponding image folder\nfor csv_id in all_csv_ids:\n    if csv_id not in all_patient_folders:\n        skipped_ids.append({\"Patient ID\": csv_id, \"Reason\": \"No image folder in train directory\"})\n\n# Create DataFrames to display analysis data and skipped IDs\nanalysis_df = pd.DataFrame(analysis_data)\nskipped_ids_df = pd.DataFrame(skipped_ids)\n\n# Display the tables\nprint(\"Analysis Table:\")\nprint(analysis_df)\n\nprint(\"\\nSkipped IDs Table:\")\nprint(skipped_ids_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T19:14:10.101107Z","iopub.execute_input":"2024-11-23T19:14:10.101748Z","iopub.status.idle":"2024-11-23T19:14:11.197457Z","shell.execute_reply.started":"2024-11-23T19:14:10.101710Z","shell.execute_reply":"2024-11-23T19:14:11.196601Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 8-Check for Missing Data","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# Define the directories and file paths\ntrain_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\ncsv_path = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\"\n\n# Load the CSV file\nlabels_df = pd.read_csv(csv_path)\n\n# Process patient IDs to remove leading zeros and use the same format in both places\nlabels_df['BraTS21ID'] = labels_df['BraTS21ID'].apply(lambda x: str(x).zfill(5))\n\n# Initialize lists for analysis data and skipped IDs\nanalysis_data = []\nskipped_ids = []\n\n# Function to count images in a given modality folder\ndef count_images_in_folder(folder_path):\n    return len([img for img in os.listdir(folder_path) if img.endswith(\".dcm\")])\n\n# Get all patient IDs from both CSV and folders for cross-referencing\nall_patient_folders = set(os.listdir(train_dir))\nall_csv_ids = set(labels_df['BraTS21ID'].values)\n\n# Iterate through each patient folder in the training directory\nfor patient_folder in all_patient_folders:\n    patient_id = patient_folder.zfill(5)  # Ensure ID format consistency\n\n    # Check if the patient ID exists in the labels DataFrame\n    if patient_id in all_csv_ids:\n        # Find corresponding label for the patient\n        label_row = labels_df[labels_df['BraTS21ID'] == patient_id]\n        tumor_status = label_row['MGMT_value'].values[0]\n\n        # Initialize counts for FLAIR and T1wCE\n        flair_count, t1wce_count = 0, 0\n\n        # Define paths for FLAIR and T1wCE folders\n        flair_path = os.path.join(train_dir, patient_folder, \"FLAIR\")\n        t1wce_path = os.path.join(train_dir, patient_folder, \"T1wCE\")\n\n        # Count images if the folder exists and is not empty\n        if os.path.isdir(flair_path):\n            flair_count = count_images_in_folder(flair_path)\n        if os.path.isdir(t1wce_path):\n            t1wce_count = count_images_in_folder(t1wce_path)\n\n        # Append the data to the analysis list\n        analysis_data.append({\n            \"Patient ID\": patient_id,\n            \"FLAIR Count\": flair_count,\n            \"T1wCE Count\": t1wce_count,\n            \"Tumor Status\": \"Yes\" if tumor_status == 1 else \"No\"\n        })\n    else:\n        # Record IDs skipped due to missing labels\n        skipped_ids.append({\"Patient ID\": patient_id, \"Reason\": \"No label in CSV\"})\n\n# Identify and record IDs from CSV with no corresponding image folder\nfor csv_id in all_csv_ids:\n    if csv_id not in all_patient_folders:\n        skipped_ids.append({\"Patient ID\": csv_id, \"Reason\": \"No image folder in train directory\"})\n\n# Create DataFrames to display analysis data and skipped IDs\nanalysis_df = pd.DataFrame(analysis_data)\nskipped_ids_df = pd.DataFrame(skipped_ids)\n\n# Display the tables\nprint(\"Analysis Table:\")\nprint(analysis_df)\n\n# Check if there are any skipped IDs\nif skipped_ids:\n    print(\"\\nSkipped IDs Table:\")\n    print(skipped_ids_df)\nelse:\n    print(\"\\nNo missing data found. All patient IDs have corresponding labels and image folders.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T19:14:14.825242Z","iopub.execute_input":"2024-11-23T19:14:14.825968Z","iopub.status.idle":"2024-11-23T19:14:15.845806Z","shell.execute_reply.started":"2024-11-23T19:14:14.825933Z","shell.execute_reply":"2024-11-23T19:14:15.844948Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 9-Data Formatting in Coloum","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom IPython.display import display, HTML\n\n# Define the directories and file paths\ntrain_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\ncsv_path = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\"\n\n# Load the CSV file\nlabels_df = pd.read_csv(csv_path)\n\n# Process patient IDs to remove leading zeros and use consistent format in both sources\nlabels_df['BraTS21ID'] = labels_df['BraTS21ID'].apply(lambda x: str(x).zfill(5))\n\n# Initialize lists for analysis data and skipped IDs\nanalysis_data = []\nskipped_ids = []\n\n# Function to count images in a given modality folder\ndef count_images_in_folder(folder_path):\n    return len([img for img in os.listdir(folder_path) if img.endswith(\".dcm\")])\n\n# Get all patient IDs from both CSV and folders for cross-referencing\nall_patient_folders = set(os.listdir(train_dir))\nall_csv_ids = set(labels_df['BraTS21ID'].values)\n\n# Iterate through each patient folder in the training directory\nfor patient_folder in all_patient_folders:\n    patient_id = patient_folder.zfill(5)  # Ensure consistent ID format\n\n    # Check if the patient ID exists in the labels DataFrame\n    if patient_id in all_csv_ids:\n        # Retrieve tumor status for the patient\n        tumor_status = labels_df.loc[labels_df['BraTS21ID'] == patient_id, 'MGMT_value'].values[0]\n\n        # Initialize counts for FLAIR and T1wCE\n        flair_count, t1wce_count = 0, 0\n\n        # Define paths for FLAIR and T1wCE folders\n        flair_path = os.path.join(train_dir, patient_folder, \"FLAIR\")\n        t1wce_path = os.path.join(train_dir, patient_folder, \"T1wCE\")\n\n        # Count images if folders exist and are not empty\n        if os.path.isdir(flair_path):\n            flair_count = count_images_in_folder(flair_path)\n        if os.path.isdir(t1wce_path):\n            t1wce_count = count_images_in_folder(t1wce_path)\n\n        # Append formatted data for the analysis table\n        analysis_data.append({\n            \"Patient ID\": patient_id,\n            \"FLAIR Count\": flair_count,\n            \"T1wCE Count\": t1wce_count,\n            \"Tumor Status\": \"Yes\" if tumor_status == 1 else \"No\"\n        })\n    else:\n        # Record IDs skipped due to missing labels\n        skipped_ids.append({\"Patient ID\": patient_id, \"Reason\": \"No label in CSV\"})\n\n# Identify and record IDs from CSV with no corresponding image folder\nfor csv_id in all_csv_ids:\n    if csv_id not in all_patient_folders:\n        skipped_ids.append({\"Patient ID\": csv_id, \"Reason\": \"No image folder in train directory\"})\n\n# Create DataFrames for analysis and skipped IDs tables\nanalysis_df = pd.DataFrame(analysis_data)\nskipped_ids_df = pd.DataFrame(skipped_ids)\n\n# Sort the analysis DataFrame by 'Patient ID' in ascending order\nanalysis_df = analysis_df.sort_values(by=\"Patient ID\").reset_index(drop=True)\n\n# Add a numbering column to the analysis table\nanalysis_df.index = range(1, len(analysis_df) + 1)\nanalysis_df.index.name = \"Entry Number\"\n\n# Display the tables with enhanced formatting\nprint(\"\\n--- Analysis Table ---\\n\")\nif not analysis_df.empty:\n    # Center-align the table content and make it wider for improved readability\n    display(HTML(analysis_df.to_html(index=True, justify='center', border=1)))\n\n# Check if there are any skipped IDs and display accordingly\nif skipped_ids:\n    print(\"\\n--- Skipped IDs Table ---\\n\")\n    # Sort skipped IDs by 'Patient ID' and apply similar formatting\n    skipped_ids_df = skipped_ids_df.sort_values(by=\"Patient ID\").reset_index(drop=True)\n    skipped_ids_df.index = range(1, len(skipped_ids_df) + 1)\n    skipped_ids_df.index.name = \"Entry Number\"\n    display(HTML(skipped_ids_df.to_html(index=True, justify='center', border=1)))\nelse:\n    print(\"\\nNo missing data found. All patient IDs have corresponding labels and image folders.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T19:14:18.144593Z","iopub.execute_input":"2024-11-23T19:14:18.144936Z","iopub.status.idle":"2024-11-23T19:14:19.336298Z","shell.execute_reply.started":"2024-11-23T19:14:18.144907Z","shell.execute_reply":"2024-11-23T19:14:19.335418Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 10-Visual Representation Part-1","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Ensure that Seaborn styles are used for a polished look\nsns.set(style=\"whitegrid\")\n\n# 1. Distribution of FLAIR and T1wCE Image Counts\nplt.figure(figsize=(14, 6))\nsns.histplot(analysis_df['FLAIR Count'], color='blue', kde=True, label='FLAIR Count')\nsns.histplot(analysis_df['T1wCE Count'], color='orange', kde=True, label='T1wCE Count')\nplt.title(\"Distribution of FLAIR and T1wCE Image Counts\")\nplt.xlabel(\"Number of Images\")\nplt.ylabel(\"Frequency\")\nplt.legend()\nplt.show()\n\n# 2. Tumor vs. Non-Tumor Patients\ntumor_counts = analysis_df['Tumor Status'].value_counts()\nplt.figure(figsize=(8, 6))\nplt.pie(tumor_counts, labels=['No Tumor', 'Tumor'], autopct='%1.1f%%', colors=['lightblue', 'salmon'], startangle=140)\nplt.title(\"Distribution of Tumor and Non-Tumor Patients\")\nplt.show()\n\n# 3. Box Plot of Image Counts by Tumor Status\nplt.figure(figsize=(14, 6))\nplt.subplot(1, 2, 1)\nsns.boxplot(x='Tumor Status', y='FLAIR Count', data=analysis_df, palette=\"Blues\")\nplt.title(\"FLAIR Image Count by Tumor Status\")\nplt.subplot(1, 2, 2)\nsns.boxplot(x='Tumor Status', y='T1wCE Count', data=analysis_df, palette=\"Oranges\")\nplt.title(\"T1wCE Image Count by Tumor Status\")\nplt.show()\n\n# 4. Image Count Correlation between FLAIR and T1wCE\nplt.figure(figsize=(8, 6))\nsns.scatterplot(x='FLAIR Count', y='T1wCE Count', hue='Tumor Status', data=analysis_df, palette={'Yes': 'red', 'No': 'green'})\nplt.title(\"Correlation between FLAIR and T1wCE Image Counts\")\nplt.xlabel(\"FLAIR Image Count\")\nplt.ylabel(\"T1wCE Image Count\")\nplt.legend(title=\"Tumor Status\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T19:14:27.834229Z","iopub.execute_input":"2024-11-23T19:14:27.834818Z","iopub.status.idle":"2024-11-23T19:14:29.809275Z","shell.execute_reply.started":"2024-11-23T19:14:27.834782Z","shell.execute_reply":"2024-11-23T19:14:29.808427Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 11-Visual Representation Part-2","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Assuming analysis_df is already created with patient data, here's the updated code\n\n# Convert 'Tumor Status' to numeric: 1 for 'Yes' and 0 for 'No'\nanalysis_df['Tumor Status Numeric'] = analysis_df['Tumor Status'].apply(lambda x: 1 if x == 'Yes' else 0)\n\n# 1. Correlation Heatmap\nplt.figure(figsize=(10, 8))\ncorr_matrix = analysis_df[['FLAIR Count', 'T1wCE Count', 'Tumor Status Numeric']].corr()\nsns.heatmap(corr_matrix, annot=True, cmap='coolwarm', fmt='.2f')\nplt.title(\"Correlation Heatmap between Image Counts and Tumor Status\")\nplt.show()\n\n# 2. Tumor Status Distribution\nplt.figure(figsize=(8, 6))\nsns.countplot(x='Tumor Status', data=analysis_df, palette='Set2')\nplt.title(\"Tumor vs Non-Tumor Distribution\")\nplt.xlabel(\"Tumor Status\")\nplt.ylabel(\"Number of Patients\")\nplt.show()\n\n# 3. FLAIR Count vs Tumor Status (Bar Plot)\nplt.figure(figsize=(10, 6))\nsns.boxplot(x='Tumor Status', y='FLAIR Count', data=analysis_df, palette='Set1')\nplt.title(\"FLAIR Count vs Tumor Status\")\nplt.xlabel(\"Tumor Status\")\nplt.ylabel(\"FLAIR Image Count\")\nplt.show()\n\n# 4. T1wCE Count vs Tumor Status (Bar Plot)\nplt.figure(figsize=(10, 6))\nsns.boxplot(x='Tumor Status', y='T1wCE Count', data=analysis_df, palette='Set1')\nplt.title(\"T1wCE Count vs Tumor Status\")\nplt.xlabel(\"Tumor Status\")\nplt.ylabel(\"T1wCE Image Count\")\nplt.show()\n\n# 5. Tumor Status vs Total Image Count (FLAIR + T1wCE)\nanalysis_df['Total Image Count'] = analysis_df['FLAIR Count'] + analysis_df['T1wCE Count']\nplt.figure(figsize=(8, 6))\nsns.boxplot(x='Tumor Status', y='Total Image Count', data=analysis_df, palette='Set2')\nplt.title(\"Total Image Count vs Tumor Status\")\nplt.xlabel(\"Tumor Status\")\nplt.ylabel(\"Total Image Count (FLAIR + T1wCE)\")\nplt.show()\n\n# 6. Pairplot for FLAIR, T1wCE Count and Tumor Status (Numeric)\nsns.pairplot(analysis_df[['FLAIR Count', 'T1wCE Count', 'Tumor Status Numeric']], hue='Tumor Status Numeric', palette='coolwarm')\nplt.suptitle(\"Pairplot of FLAIR and T1wCE Counts with Tumor Status\", y=1.02)\nplt.show()\n\n# 7. Histograms for Image Counts (FLAIR and T1wCE)\nplt.figure(figsize=(14, 6))\nplt.subplot(1, 2, 1)\nsns.histplot(analysis_df['FLAIR Count'], kde=True, color='blue', bins=30)\nplt.title(\"Distribution of FLAIR Image Count\")\nplt.xlabel(\"FLAIR Count\")\nplt.ylabel(\"Frequency\")\n\nplt.subplot(1, 2, 2)\nsns.histplot(analysis_df['T1wCE Count'], kde=True, color='green', bins=30)\nplt.title(\"Distribution of T1wCE Image Count\")\nplt.xlabel(\"T1wCE Count\")\nplt.ylabel(\"Frequency\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T19:14:34.705907Z","iopub.execute_input":"2024-11-23T19:14:34.706805Z","iopub.status.idle":"2024-11-23T19:14:38.243142Z","shell.execute_reply.started":"2024-11-23T19:14:34.706772Z","shell.execute_reply":"2024-11-23T19:14:38.242300Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\n# Verify GPU availability\nprint(\"Num GPUs Available: \", len(tf.config.list_physical_devices('GPU')))\n\n# List available GPU devices\nif tf.config.list_physical_devices('GPU'):\n    print(\"TensorFlow is set to use the GPU.\")\nelse:\n    print(\"No GPU detected. The model will run on the CPU.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T19:34:42.772090Z","iopub.execute_input":"2024-11-23T19:34:42.772799Z","iopub.status.idle":"2024-11-23T19:34:42.778040Z","shell.execute_reply.started":"2024-11-23T19:34:42.772765Z","shell.execute_reply":"2024-11-23T19:34:42.777059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport matplotlib.pyplot as plt \nimport cv2 as cv\nfrom path import Path\nimport os \nimport glob\nimport tensorflow_hub as hub\nimport os \nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom keras import layers\nfrom tqdm import tqdm\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nfrom tensorflow.keras.utils import to_categorical","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T20:43:48.455583Z","iopub.execute_input":"2024-11-23T20:43:48.455946Z","iopub.status.idle":"2024-11-23T20:43:48.461553Z","shell.execute_reply.started":"2024-11-23T20:43:48.455914Z","shell.execute_reply":"2024-11-23T20:43:48.460587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nimport numpy as np\n\ndef load_dicom(path):\n    dicom = pydicom.dcmread(path)  # Corrected to use dcmread\n    data = dicom.pixel_array       # Extract pixel data from the DICOM\n    data = data - np.min(data)     # Normalize the pixel values\n    return data\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T20:43:53.415121Z","iopub.execute_input":"2024-11-23T20:43:53.415819Z","iopub.status.idle":"2024-11-23T20:43:53.420231Z","shell.execute_reply.started":"2024-11-23T20:43:53.415783Z","shell.execute_reply":"2024-11-23T20:43:53.419298Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv')\ntrain_dir = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/train'\n\ntrainset = []\ntrainlabel = []\n\nfor i in tqdm(range(len(train_df))):\n    idt = train_df.loc[i, 'BraTS21ID']\n    idt2 = ('00000' + str(idt))[-5:]\n    path = os.path.join(train_dir, idt2, 'T1wCE')\n    \n    # Load and process DICOM images\n    temp_images = []\n    for im in os.listdir(path):\n        img = load_dicom(os.path.join(path, im))\n        img = cv.resize(img, (64, 64))  # Resize the 2D image to 64x64\n        temp_images.append(img)\n    \n    # Stack the images into a 3D volume\n    temp_images = np.stack(temp_images, axis=-1)  # Shape: (64, 64, 128)\n    \n    # Resize or pad to ensure consistent depth (128 slices)\n    temp_images = resize_or_pad_image(temp_images, target_depth=128)\n    \n    trainset.append(temp_images)\n    trainlabel.append(train_df.loc[i, 'MGMT_value'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T21:03:42.140125Z","iopub.execute_input":"2024-11-23T21:03:42.141105Z","iopub.status.idle":"2024-11-23T21:20:34.117919Z","shell.execute_reply.started":"2024-11-23T21:03:42.141067Z","shell.execute_reply":"2024-11-23T21:20:34.117134Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_dir = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/test'\ntestset = []\ntestidt = []\n\nfor i in tqdm(range(len(sample_df))):\n    idt = sample_df.loc[i, 'BraTS21ID']\n    idt2 = ('00000' + str(idt))[-5:]\n    path = os.path.join(test_dir, idt2, 'T1wCE')\n    \n    # Load and process DICOM images\n    temp_images = []\n    for im in os.listdir(path):\n        img = load_dicom(os.path.join(path, im))\n        img = cv.resize(img, (64, 64))  # Resize the 2D image to 64x64\n        temp_images.append(img)\n    \n    # Stack the images into a 3D volume\n    temp_images = np.stack(temp_images, axis=-1)  # Shape: (64, 64, 128)\n    \n    # Resize or pad to ensure consistent depth (128 slices)\n    temp_images = resize_or_pad_image(temp_images, target_depth=128)\n    \n    testset.append(temp_images)\n    testidt.append(idt)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T21:21:29.206203Z","iopub.execute_input":"2024-11-23T21:21:29.206657Z","iopub.status.idle":"2024-11-23T21:22:55.063481Z","shell.execute_reply.started":"2024-11-23T21:21:29.206623Z","shell.execute_reply":"2024-11-23T21:22:55.062556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train = np.array(trainset)  # Shape: (num_samples, 64, 64, 128)\ny = np.array(trainlabel)\nY_train = to_categorical(y, num_classes=2)  # One-hot encoding of labels\n\n# Add the 'channels' dimension (for grayscale)\nX_train = np.expand_dims(X_train, axis=-1)  # Shape: (num_samples, 64, 64, 128, 1)\n\nX_test = np.array(testset)  # Shape: (num_samples, 64, 64, 128)\nX_test = np.expand_dims(X_test, axis=-1)  # Add the 'channels' dimension\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T21:23:17.583295Z","iopub.execute_input":"2024-11-23T21:23:17.583925Z","iopub.status.idle":"2024-11-23T21:23:19.182064Z","shell.execute_reply.started":"2024-11-23T21:23:17.583891Z","shell.execute_reply":"2024-11-23T21:23:19.181375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Shape of X_train: {X_train.shape}\")\nprint(f\"Shape of X_test: {X_test.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T21:23:23.676621Z","iopub.execute_input":"2024-11-23T21:23:23.677224Z","iopub.status.idle":"2024-11-23T21:23:23.681752Z","shell.execute_reply.started":"2024-11-23T21:23:23.677191Z","shell.execute_reply":"2024-11-23T21:23:23.680746Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_val, Y_train, Y_val = train_test_split(X_train, Y_train, test_size=0.3, random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T21:23:48.750813Z","iopub.execute_input":"2024-11-23T21:23:48.751642Z","iopub.status.idle":"2024-11-23T21:23:49.040477Z","shell.execute_reply.started":"2024-11-23T21:23:48.751605Z","shell.execute_reply":"2024-11-23T21:23:49.039722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense\n\n\nmodel = models.Sequential()\n\n# 3D Convolutional layers\nmodel.add(layers.Conv3D(filters=32, kernel_size=(3, 3, 3), activation='relu', input_shape=(64, 64, 128, 1)))\nmodel.add(layers.MaxPooling3D(pool_size=(2, 2, 2)))\nmodel.add(layers.BatchNormalization())\n\nmodel.add(layers.Conv3D(filters=64, kernel_size=(3, 3, 3), activation='relu'))\nmodel.add(layers.MaxPooling3D(pool_size=(2, 2, 2)))\nmodel.add(layers.BatchNormalization())\n\nmodel.add(layers.Dropout(0.25))\n\nmodel.add(layers.Conv3D(filters=128, kernel_size=(3, 3, 3), activation='relu'))\nmodel.add(layers.MaxPooling3D(pool_size=(2, 2, 2)))\nmodel.add(layers.BatchNormalization())\n\nmodel.add(layers.Conv3D(filters=256, kernel_size=(3, 3, 3), activation='relu'))\nmodel.add(layers.MaxPooling3D(pool_size=(2, 2, 2)))\nmodel.add(layers.BatchNormalization())\n\n# Flatten the 3D output to feed into dense layers\nmodel.add(layers.Flatten())\n\n# Dense layers\nmodel.add(layers.Dense(128, activation='relu'))\nmodel.add(layers.Dropout(0.5))\n\nmodel.add(layers.Dense(2, activation='softmax'))  # Output layer for binary classification\n\n# Define the optimizer with a specific learning rate\nlearning_rate = 0.0005  # You can adjust this value\noptimizer = Adam(learning_rate=learning_rate)\n\n# Compile the model with the custom optimizer\nmodel.compile(optimizer=optimizer, loss='categorical_crossentropy', metrics=['accuracy'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T21:23:52.780461Z","iopub.execute_input":"2024-11-23T21:23:52.781247Z","iopub.status.idle":"2024-11-23T21:23:52.961732Z","shell.execute_reply.started":"2024-11-23T21:23:52.781207Z","shell.execute_reply":"2024-11-23T21:23:52.960859Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"callback = tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T21:24:01.842196Z","iopub.execute_input":"2024-11-23T21:24:01.842569Z","iopub.status.idle":"2024-11-23T21:24:01.850601Z","shell.execute_reply.started":"2024-11-23T21:24:01.842533Z","shell.execute_reply":"2024-11-23T21:24:01.849631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(X_train, Y_train, epochs=100, batch_size=16, validation_data=(X_val, Y_val), callbacks=[callback], verbose=1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T21:24:10.117019Z","iopub.execute_input":"2024-11-23T21:24:10.117699Z","iopub.status.idle":"2024-11-23T21:25:51.283891Z","shell.execute_reply.started":"2024-11-23T21:24:10.117664Z","shell.execute_reply":"2024-11-23T21:25:51.283214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot accuracy and loss\nplt.figure(figsize=(12, 6))\nplt.subplot(1, 2, 1)\nplt.plot(history.history['accuracy'], label='Training Accuracy')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\nplt.title('Training and Validation Accuracy')\nplt.legend()\n\nplt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='Training Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Training and Validation Loss')\nplt.legend()\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T21:26:00.885688Z","iopub.execute_input":"2024-11-23T21:26:00.886097Z","iopub.status.idle":"2024-11-23T21:26:01.457732Z","shell.execute_reply.started":"2024-11-23T21:26:00.886064Z","shell.execute_reply":"2024-11-23T21:26:01.456850Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print overall training and validation accuracy\nfinal_training_accuracy = history.history['accuracy'][-1]\nfinal_validation_accuracy = history.history['val_accuracy'][-1]\nprint(f\"Overall Training Accuracy: {final_training_accuracy:.2f}\")\nprint(f\"Overall Validation Accuracy: {final_validation_accuracy:.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T21:27:12.647591Z","iopub.execute_input":"2024-11-23T21:27:12.647942Z","iopub.status.idle":"2024-11-23T21:27:12.653488Z","shell.execute_reply.started":"2024-11-23T21:27:12.647914Z","shell.execute_reply":"2024-11-23T21:27:12.652513Z"}},"outputs":[],"execution_count":null}]}