{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":29653,"databundleVersionId":2420395,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\n# List all files in the input directory and subdirectories\ninput_dir = '/kaggle/input'\nfile_paths = []\n\nfor dirname, _, filenames in os.walk(input_dir):\n    for filename in filenames:\n        \n        file_paths.append(os.path.join(dirname, filename))\n\n# Print confirmation message\nif file_paths:\n    print(\"Input added successfully.\")\nelse:\n    print(\"No input files found.\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-06T18:21:40.312160Z","iopub.execute_input":"2024-12-06T18:21:40.312567Z","iopub.status.idle":"2024-12-06T18:34:07.242279Z","shell.execute_reply.started":"2024-12-06T18:21:40.312534Z","shell.execute_reply":"2024-12-06T18:34:07.241222Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport pydicom as dicom\nimport numpy as np\nfrom IPython.display import display, HTML\n\n# Define directory paths and CSV file path\ntrain_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\ncsv_path = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\"\n\n# Load the CSV file containing patient tumor status\nlabels_df = pd.read_csv(csv_path)\n\n# Ensure 'BraTS21ID' is formatted consistently\nlabels_df['BraTS21ID'] = labels_df['BraTS21ID'].apply(lambda x: str(x).zfill(5))\n\n# Separate the tumor and non-tumor patients\ntumor_patients = labels_df[labels_df['MGMT_value'] == 1]['BraTS21ID']\nnon_tumor_patients = labels_df[labels_df['MGMT_value'] == 0]['BraTS21ID']\n\n# Step 1: Count Tumor and Non-Tumor Images\ntumor_count = 0\nnon_tumor_count = 0\n\n# Lists to hold image paths (just for visualization or further use)\ntumor_images = []\nnon_tumor_images = []\n\n# Iterate through the patient folders and collect FLAIR images\nfor patient_folder in os.listdir(train_dir):\n    patient_id = patient_folder.zfill(5)  # Ensure consistent ID format\n    \n    # Check if the patient has tumor or not\n    if patient_id in tumor_patients.values:\n        tumor_status = 'Tumor'\n        # Path to the FLAIR modality folder for the patient\n        flair_path = os.path.join(train_dir, patient_folder, \"FLAIR\")\n        \n        if os.path.isdir(flair_path):  # Ensure the folder exists\n            for img_name in os.listdir(flair_path):\n                if img_name.endswith(\".dcm\"):\n                    img_path = os.path.join(flair_path, img_name)\n                    tumor_images.append(img_path)\n                    tumor_count += 1\n    \n    elif patient_id in non_tumor_patients.values:\n        tumor_status = 'Non-Tumor'\n        # Path to the FLAIR modality folder for the patient\n        flair_path = os.path.join(train_dir, patient_folder, \"FLAIR\")\n        \n        if os.path.isdir(flair_path):  # Ensure the folder exists\n            for img_name in os.listdir(flair_path):\n                if img_name.endswith(\".dcm\"):\n                    img_path = os.path.join(flair_path, img_name)\n                    non_tumor_images.append(img_path)\n                    non_tumor_count += 1\n\n# Print out the total counts\nprint(f\"Total Tumor Images: {tumor_count}\")\nprint(f\"Total Non-Tumor Images: {non_tumor_count}\")\n\n# Step 2: Plot the Distribution of Tumor vs Non-Tumor Images\nplt.figure(figsize=(8, 6))\ncategories = ['Tumor', 'Non-Tumor']\ncounts = [tumor_count, non_tumor_count]\nplt.bar(categories, counts, color=['salmon', 'lightblue'])\nplt.title('Distribution of Tumor vs Non-Tumor Images', fontsize=14)\nplt.xlabel('Category', fontsize=12)\nplt.ylabel('Number of Images', fontsize=12)\nplt.show()\n\n# Step 3: Display Sample Images from Both Categories\n\ndef display_sample_images(image_paths, title, num_samples=3):\n    sample_paths = np.random.choice(image_paths, num_samples, replace=False)\n    \n    plt.figure(figsize=(12, 4))\n    for i, img_path in enumerate(sample_paths):\n        ds = dicom.dcmread(img_path)  # Read the DICOM image\n        img = ds.pixel_array  # Extract pixel data\n        plt.subplot(1, num_samples, i+1)\n        plt.imshow(img, cmap='gray')\n        plt.axis('off')\n        plt.title(f'Sample {i+1}')\n    plt.suptitle(title, fontsize=16)\n    plt.show()\n\n# Display sample tumor and non-tumor images\ndisplay_sample_images(tumor_images, \"Tumor Images (FLAIR Modality)\", num_samples=3)\ndisplay_sample_images(non_tumor_images, \"Non-Tumor Images (FLAIR Modality)\", num_samples=3)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T18:34:07.243905Z","iopub.execute_input":"2024-12-06T18:34:07.244208Z","iopub.status.idle":"2024-12-06T18:34:09.788746Z","shell.execute_reply.started":"2024-12-06T18:34:07.244178Z","shell.execute_reply":"2024-12-06T18:34:09.787850Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# Step 1: Define directory paths and CSV file path\ntrain_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\ncsv_path = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\"\n\n# Load the CSV file containing patient tumor status\nlabels_df = pd.read_csv(csv_path)\n\n# Ensure 'BraTS21ID' is formatted consistently\nlabels_df['BraTS21ID'] = labels_df['BraTS21ID'].apply(lambda x: str(x).zfill(5))\n\n# Separate the tumor and non-tumor patients\ntumor_patients = labels_df[labels_df['MGMT_value'] == 1]['BraTS21ID']\nnon_tumor_patients = labels_df[labels_df['MGMT_value'] == 0]['BraTS21ID']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T18:34:09.789792Z","iopub.execute_input":"2024-12-06T18:34:09.790035Z","iopub.status.idle":"2024-12-06T18:34:09.801291Z","shell.execute_reply.started":"2024-12-06T18:34:09.790011Z","shell.execute_reply":"2024-12-06T18:34:09.800497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom as dicom\n\n# Lists to hold image paths and image details\ntumor_images = []\nnon_tumor_images = []\n\n# Step 2: Count Tumor and Non-Tumor Images\ntumor_count = 0\nnon_tumor_count = 0\n\n# Image dimension statistics\ntumor_image_sizes = []\nnon_tumor_image_sizes = []\n\n# Iterate through the patient folders and collect FLAIR images\nfor patient_folder in os.listdir(train_dir):\n    patient_id = patient_folder.zfill(5)  # Ensure consistent ID format\n    \n    # Check if the patient has tumor or not\n    if patient_id in tumor_patients.values:\n        # Path to the FLAIR modality folder for the patient\n        flair_path = os.path.join(train_dir, patient_folder, \"FLAIR\")\n        \n        if os.path.isdir(flair_path):  # Ensure the folder exists\n            for img_name in os.listdir(flair_path):\n                if img_name.endswith(\".dcm\"):\n                    img_path = os.path.join(flair_path, img_name)\n                    tumor_images.append(img_path)\n                    tumor_count += 1\n                    \n                    # Read image and get dimensions\n                    ds = dicom.dcmread(img_path)\n                    img = ds.pixel_array\n                    tumor_image_sizes.append(img.shape)\n    \n    elif patient_id in non_tumor_patients.values:\n        # Path to the FLAIR modality folder for the patient\n        flair_path = os.path.join(train_dir, patient_folder, \"FLAIR\")\n        \n        if os.path.isdir(flair_path):  # Ensure the folder exists\n            for img_name in os.listdir(flair_path):\n                if img_name.endswith(\".dcm\"):\n                    img_path = os.path.join(flair_path, img_name)\n                    non_tumor_images.append(img_path)\n                    non_tumor_count += 1\n                    \n                    # Read image and get dimensions\n                    ds = dicom.dcmread(img_path)\n                    img = ds.pixel_array\n                    non_tumor_image_sizes.append(img.shape)\n\n# Print out the total counts\nprint(f\"Total Tumor Images: {tumor_count}\")\nprint(f\"Total Non-Tumor Images: {non_tumor_count}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T18:34:09.803120Z","iopub.execute_input":"2024-12-06T18:34:09.803444Z","iopub.status.idle":"2024-12-06T18:41:16.959110Z","shell.execute_reply.started":"2024-12-06T18:34:09.803389Z","shell.execute_reply":"2024-12-06T18:41:16.958105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Step 3: Plot the Distribution of Tumor vs Non-Tumor Images\nplt.figure(figsize=(8, 6))\ncategories = ['Tumor', 'Non-Tumor']\ncounts = [tumor_count, non_tumor_count]\nplt.bar(categories, counts, color=['salmon', 'lightblue'])\nplt.title('Distribution of Tumor vs Non-Tumor Images', fontsize=14)\nplt.xlabel('Category', fontsize=12)\nplt.ylabel('Number of Images', fontsize=12)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T18:41:16.960354Z","iopub.execute_input":"2024-12-06T18:41:16.960642Z","iopub.status.idle":"2024-12-06T18:41:17.099123Z","shell.execute_reply.started":"2024-12-06T18:41:16.960612Z","shell.execute_reply":"2024-12-06T18:41:17.098318Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# Step 4: Statistical Analysis of Image Dimensions\ntumor_heights, tumor_widths = zip(*tumor_image_sizes)\nnon_tumor_heights, non_tumor_widths = zip(*non_tumor_image_sizes)\n\n# Calculate mean and std for height and width\ntumor_height_mean, tumor_width_mean = np.mean(tumor_heights), np.mean(tumor_widths)\ntumor_height_std, tumor_width_std = np.std(tumor_heights), np.std(tumor_widths)\nnon_tumor_height_mean, non_tumor_width_mean = np.mean(non_tumor_heights), np.mean(non_tumor_widths)\nnon_tumor_height_std, non_tumor_width_std = np.std(non_tumor_heights), np.std(non_tumor_widths)\n\nprint(f\"Tumor Image Dimensions (mean ± std): Height: {tumor_height_mean} ± {tumor_height_std}, Width: {tumor_width_mean} ± {tumor_width_std}\")\nprint(f\"Non-Tumor Image Dimensions (mean ± std): Height: {non_tumor_height_mean} ± {non_tumor_height_std}, Width: {non_tumor_width_mean} ± {non_tumor_width_std}\")\n\n# Plot histograms of image dimensions\nplt.figure(figsize=(12, 6))\n\n# Plot height distributions\nplt.subplot(1, 2, 1)\nplt.hist(tumor_heights, bins=20, alpha=0.5, label='Tumor', color='salmon')\nplt.hist(non_tumor_heights, bins=20, alpha=0.5, label='Non-Tumor', color='lightblue')\nplt.title('Histogram of Image Heights')\nplt.xlabel('Height')\nplt.ylabel('Frequency')\nplt.legend()\n\n# Plot width distributions\nplt.subplot(1, 2, 2)\nplt.hist(tumor_widths, bins=20, alpha=0.5, label='Tumor', color='salmon')\nplt.hist(non_tumor_widths, bins=20, alpha=0.5, label='Non-Tumor', color='lightblue')\nplt.title('Histogram of Image Widths')\nplt.xlabel('Width')\nplt.ylabel('Frequency')\nplt.legend()\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T18:41:17.100184Z","iopub.execute_input":"2024-12-06T18:41:17.100562Z","iopub.status.idle":"2024-12-06T18:41:18.080805Z","shell.execute_reply.started":"2024-12-06T18:41:17.100522Z","shell.execute_reply":"2024-12-06T18:41:18.079963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 6: Compare Number of Images per Patient\ntumor_patient_image_count = [len([img for img in os.listdir(os.path.join(train_dir, patient, \"FLAIR\")) if img.endswith(\".dcm\")])\n                             for patient in tumor_patients]\nnon_tumor_patient_image_count = [len([img for img in os.listdir(os.path.join(train_dir, patient, \"FLAIR\")) if img.endswith(\".dcm\")])\n                                 for patient in non_tumor_patients]\n\nplt.figure\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T18:41:18.081816Z","iopub.execute_input":"2024-12-06T18:41:18.082068Z","iopub.status.idle":"2024-12-06T18:41:18.607047Z","shell.execute_reply.started":"2024-12-06T18:41:18.082042Z","shell.execute_reply":"2024-12-06T18:41:18.606106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(len(tumor_images), len(non_tumor_images))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T18:47:04.951158Z","iopub.execute_input":"2024-12-06T18:47:04.952006Z","iopub.status.idle":"2024-12-06T18:47:04.956531Z","shell.execute_reply.started":"2024-12-06T18:47:04.951969Z","shell.execute_reply":"2024-12-06T18:47:04.955725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"non_tumor_images[10]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T18:46:14.017680Z","iopub.execute_input":"2024-12-06T18:46:14.018820Z","iopub.status.idle":"2024-12-06T18:46:14.024524Z","shell.execute_reply.started":"2024-12-06T18:46:14.018777Z","shell.execute_reply":"2024-12-06T18:46:14.023624Z"}},"outputs":[],"execution_count":null}]}