{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":29653,"databundleVersionId":2420395,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\n\n# Define the directory paths\ntrain_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\ntest_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/test\"\n\n# Function to count patients and images\ndef count_patients_and_images(directory):\n    patient_count = 0\n    image_count = 0\n    \n    for patient_folder in os.listdir(directory):\n        patient_path = os.path.join(directory, patient_folder)\n        \n        # Check if it's a folder\n        if os.path.isdir(patient_path):\n            patient_count += 1\n            \n            # Count images in each modality folder\n            for modality in [\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"]:\n                modality_path = os.path.join(patient_path, modality)\n                if os.path.isdir(modality_path):\n                    image_count += len([img for img in os.listdir(modality_path) if img.endswith(\".dcm\")])\n\n    return patient_count, image_count\n\n# Get counts for training and testing sets\ntrain_patients, train_images = count_patients_and_images(train_dir)\ntest_patients, test_images = count_patients_and_images(test_dir)\n\n# Plotting the results\nfig, ax = plt.subplots(1, 2, figsize=(12, 6))\n\n# Bar plot for number of patients\nax[0].bar([\"Train\", \"Test\"], [train_patients, test_patients], color=[\"blue\", \"orange\"])\nax[0].set_title(\"Total Number of Patients\")\nax[0].set_ylabel(\"Number of Patients\")\n\n# Bar plot for number of images\nax[1].bar([\"Train\", \"Test\"], [train_images, test_images], color=[\"blue\", \"orange\"])\nax[1].set_title(\"Total Number of Images\")\nax[1].set_ylabel(\"Number of Images\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T20:10:24.127037Z","iopub.execute_input":"2024-11-26T20:10:24.127975Z","iopub.status.idle":"2024-11-26T20:10:50.437355Z","shell.execute_reply.started":"2024-11-26T20:10:24.127928Z","shell.execute_reply":"2024-11-26T20:10:50.436470Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\n\n# Define the directory paths\ntrain_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\ntest_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/test\"\n\n# Function to count images for each modality in each patient folder\ndef count_images_per_modality(directory):\n    modalities = [\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"]\n    modality_counts = {modality: 0 for modality in modalities}\n    \n    for patient_folder in os.listdir(directory):\n        patient_path = os.path.join(directory, patient_folder)\n        \n        # Check if it's a folder\n        if os.path.isdir(patient_path):\n            # Count images in each modality folder\n            for modality in modalities:\n                modality_path = os.path.join(patient_path, modality)\n                if os.path.isdir(modality_path):\n                    modality_counts[modality] += len([img for img in os.listdir(modality_path) if img.endswith(\".dcm\")])\n    \n    return modality_counts\n\n# Get modality image counts for training and testing sets\ntrain_modality_counts = count_images_per_modality(train_dir)\ntest_modality_counts = count_images_per_modality(test_dir)\n\n# Plotting the results with improvements\nfig, ax = plt.subplots(1, 2, figsize=(16, 8))\ncolors = [\"#4e79a7\", \"#f28e2b\", \"#e15759\", \"#76b7b2\"]  # Different colors for each modality\n\n# Bar plot for train dataset\nax[0].bar(train_modality_counts.keys(), train_modality_counts.values(), color=colors)\nax[0].set_title(\"Number of Images per Modality in Training Dataset\", fontsize=14, weight='bold')\nax[0].set_xlabel(\"Modality\", fontsize=12)\nax[0].set_ylabel(\"Number of Images\", fontsize=12)\n\n# Adding annotations to the bars\nfor i, (modality, count) in enumerate(train_modality_counts.items()):\n    ax[0].text(i, count + 50, str(count), ha='center', va='bottom', fontsize=10)\n\n# Bar plot for test dataset\nax[1].bar(test_modality_counts.keys(), test_modality_counts.values(), color=colors)\nax[1].set_title(\"Number of Images per Modality in Testing Dataset\", fontsize=14, weight='bold')\nax[1].set_xlabel(\"Modality\", fontsize=12)\nax[1].set_ylabel(\"Number of Images\", fontsize=12)\n\n# Adding annotations to the bars\nfor i, (modality, count) in enumerate(test_modality_counts.items()):\n    ax[1].text(i, count + 50, str(count), ha='center', va='bottom', fontsize=10)\n\n# Adding a legend\nfig.legend(train_modality_counts.keys(), loc=\"upper center\", ncol=4, fontsize=12, frameon=False)\n\nplt.tight_layout(rect=[0, 0, 1, 0.95])  # Adjust layout to make space for legend\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T20:10:50.438917Z","iopub.execute_input":"2024-11-26T20:10:50.439189Z","iopub.status.idle":"2024-11-26T20:10:52.216888Z","shell.execute_reply.started":"2024-11-26T20:10:50.439162Z","shell.execute_reply":"2024-11-26T20:10:52.216104Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\n# Load the CSV file\nfile_path = '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv'  # Replace with your actual file path\ndata = pd.read_csv(file_path)\n\n# Count the number of patients with and without tumors\ntumor_counts = data['MGMT_value'].value_counts()\n\n# Plotting\nlabels = ['No Tumor', 'Tumor']\nsizes = [tumor_counts.get(0, 0), tumor_counts.get(1, 0)]  # Get counts for 0 and 1, default to 0 if not found\ncolors = ['lightblue', 'salmon']\nexplode = (0.1, 0)  # explode 1st slice (No Tumor)\n\nplt.figure(figsize=(8, 6))\nplt.pie(sizes, explode=explode, labels=labels, colors=colors,\n        autopct='%1.1f%%', shadow=True, startangle=140)\n\nplt.title('Distribution of Patients with and without Tumors')\nplt.axis('equal')  # Equal aspect ratio ensures that pie chart is circular.\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T20:10:52.218119Z","iopub.execute_input":"2024-11-26T20:10:52.218487Z","iopub.status.idle":"2024-11-26T20:10:52.682543Z","shell.execute_reply.started":"2024-11-26T20:10:52.218449Z","shell.execute_reply":"2024-11-26T20:10:52.681344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# Define the directories and file paths\ntrain_dir = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\ncsv_path = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\"\n\n# Load the CSV file\nlabels_df = pd.read_csv(csv_path)\n\n# Process patient IDs to remove leading zeros and use the same format in both places\nlabels_df['BraTS21ID'] = labels_df['BraTS21ID'].apply(lambda x: str(x).zfill(5))\n\n# Initialize lists for analysis data and skipped IDs\nanalysis_data = []\nskipped_ids = []\n\n# Function to count images in a given modality folder\ndef count_images_in_folder(folder_path):\n    return len([img for img in os.listdir(folder_path) if img.endswith(\".dcm\")])\n\n# Get all patient IDs from both CSV and folders for cross-referencing\nall_patient_folders = set(os.listdir(train_dir))\nall_csv_ids = set(labels_df['BraTS21ID'].values)\n\n# Iterate through each patient folder in the training directory\nfor patient_folder in all_patient_folders:\n    patient_id = patient_folder.zfill(5)  # Ensure ID format consistency\n\n    # Check if the patient ID exists in the labels DataFrame\n    if patient_id in all_csv_ids:\n        # Find corresponding label for the patient\n        label_row = labels_df[labels_df['BraTS21ID'] == patient_id]\n        tumor_status = label_row['MGMT_value'].values[0]\n\n        # Initialize counts for FLAIR and T1wCE\n        flair_count, t1wce_count = 0, 0\n\n        # Define paths for FLAIR and T1wCE folders\n        flair_path = os.path.join(train_dir, patient_folder, \"FLAIR\")\n        t1wce_path = os.path.join(train_dir, patient_folder, \"T1wCE\")\n\n        # Count images if the folder exists and is not empty\n        if os.path.isdir(flair_path):\n            flair_count = count_images_in_folder(flair_path)\n        if os.path.isdir(t1wce_path):\n            t1wce_count = count_images_in_folder(t1wce_path)\n\n        # Append the data to the analysis list\n        analysis_data.append({\n            \"Patient ID\": patient_id,\n            \"FLAIR Count\": flair_count,\n            \"T1wCE Count\": t1wce_count,\n            \"Tumor Status\": \"Yes\" if tumor_status == 1 else \"No\"\n        })\n    else:\n        # Record IDs skipped due to missing labels\n        skipped_ids.append({\"Patient ID\": patient_id, \"Reason\": \"No label in CSV\"})\n\n# Identify and record IDs from CSV with no corresponding image folder\nfor csv_id in all_csv_ids:\n    if csv_id not in all_patient_folders:\n        skipped_ids.append({\"Patient ID\": csv_id, \"Reason\": \"No image folder in train directory\"})\n\n# Create DataFrames to display analysis data and skipped IDs\nanalysis_df = pd.DataFrame(analysis_data)\nskipped_ids_df = pd.DataFrame(skipped_ids)\n\n# Display the tables\nprint(\"Analysis Table:\")\nprint(analysis_df)\n\nprint(\"\\nSkipped IDs Table:\")\nprint(skipped_ids_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T20:10:52.684464Z","iopub.execute_input":"2024-11-26T20:10:52.684916Z","iopub.status.idle":"2024-11-26T20:10:53.562876Z","shell.execute_reply.started":"2024-11-26T20:10:52.684864Z","shell.execute_reply":"2024-11-26T20:10:53.562027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pydicom\nfrom skimage.transform import resize\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom torch.utils.tensorboard import SummaryWriter\n\n# Device configuration\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# Paths\ntrain_folder = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\ncsv_file = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\"\n\n# Parameters\ncnn_input_shape = (128, 64, 64)  # Depth, Height, Width\nmodalities = [\"T1wCE\"]\nmax_patients = (500)  # Limit for demonstration\nnum_classes = 2\n\n# Preprocessing\ndef preprocess_patient(patient_id , folder):\n    modality_volumes = []\n    for modality in modalities:\n        modality_folder = os.path.join(folder, patient_id, modality)\n        if not os.path.isdir(modality_folder):\n            return None\n        \n        slices = []\n        for dcm_file in sorted(os.listdir(modality_folder)):\n            if dcm_file.endswith(\".dcm\"):\n                dcm_path = os.path.join(modality_folder, dcm_file)\n                ds = pydicom.dcmread(dcm_path)\n                slices.append(ds.pixel_array)\n        \n        if not slices:\n            return None\n\n        volume = np.stack(slices, axis=0).astype(np.float32)\n        volume = (volume - volume.min()) / (volume.max() - volume.min())\n        resized_slices = np.array([resize(slice, cnn_input_shape[1:3]) for slice in volume])\n\n        current_depth = resized_slices.shape[0]\n        if current_depth < cnn_input_shape[0]:\n            pad_depth = cnn_input_shape[0] - current_depth\n            resized_slices = np.pad(resized_slices, ((0, pad_depth), (0, 0), (0, 0)))\n        else:\n            resized_slices = resized_slices[:cnn_input_shape[0]]\n\n        modality_volumes.append(resized_slices)\n\n    if not modality_volumes:\n        return None\n\n    combined_volume = np.stack(modality_volumes, axis=-1)  # (Depth, Height, Width, Modalities)\n    return combined_volume\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T20:10:53.565943Z","iopub.execute_input":"2024-11-26T20:10:53.566703Z","iopub.status.idle":"2024-11-26T20:11:08.626453Z","shell.execute_reply.started":"2024-11-26T20:10:53.566671Z","shell.execute_reply":"2024-11-26T20:11:08.625584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prepare Data\nlabels_df = pd.read_csv(csv_file)\nlabels_df['BraTS21ID'] = labels_df['BraTS21ID'].apply(lambda x: str(x).zfill(5))\npatient_labels = {row['BraTS21ID']: row['MGMT_value'] for _, row in labels_df.iterrows()}\n\ncnn_input = []\naligned_labels = []\ni=0\nfor patient_id in sorted(os.listdir(train_folder)):\n    if len(cnn_input) >= max_patients:\n        break \n    processed_data = preprocess_patient(patient_id , train_folder)\n    if processed_data is not None and patient_id in patient_labels:\n        cnn_input.append(processed_data)\n        aligned_labels.append(patient_labels[patient_id])\n        print(f\"Processed Patient ID:\",i)\n        i+=1\n        \nif not cnn_input:\n    raise ValueError(\"No valid patient data found.\")\n\ncnn_input = np.array(cnn_input)\naligned_labels = np.array(aligned_labels)\n\n# Train-Test Split\nX_train, X_val, y_train, y_val = train_test_split(\n    cnn_input, aligned_labels, test_size=0.2, random_state=42\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T20:11:08.627703Z","iopub.execute_input":"2024-11-26T20:11:08.628632Z","iopub.status.idle":"2024-11-26T20:28:11.312212Z","shell.execute_reply.started":"2024-11-26T20:11:08.628587Z","shell.execute_reply":"2024-11-26T20:28:11.310997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(X_train.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T20:28:11.313531Z","iopub.execute_input":"2024-11-26T20:28:11.313884Z","iopub.status.idle":"2024-11-26T20:28:11.320047Z","shell.execute_reply.started":"2024-11-26T20:28:11.313845Z","shell.execute_reply":"2024-11-26T20:28:11.318996Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom keras import layers, models\nfrom keras.models import Model\nfrom keras.optimizers import Adam\nfrom keras.layers import Input, Conv3D, MaxPooling3D, Flatten, Dense, Dropout, Multiply, GlobalAveragePooling3D, Reshape\nfrom sklearn.metrics import roc_auc_score  # For AUC\nimport tensorflow as tf  # For other metrics\n\n# Spatial Attention\ndef spatial_attention(feature_map):\n    \"\"\"\n    Apply spatial attention to a 3D feature map.\n    Args:\n        feature_map: Keras tensor of shape (batch_size, depth, height, width, channels).\n    Returns:\n        Attention-weighted feature map.\n    \"\"\"\n    attention = Conv3D(1, kernel_size=1, activation='sigmoid')(feature_map)  # Generate attention weights\n    return Multiply()([feature_map, attention])  # Apply weights to the feature map\n\n# Slice Attention\ndef slice_attention(volume):\n    \"\"\"\n    Apply slice (depth) attention to a 3D volume.\n    Args:\n        volume: Keras tensor of shape (batch_size, depth, height, width, channels).\n    Returns:\n        Attention-weighted volume.\n    \"\"\"\n    depth = volume.shape[1]\n    squeeze = GlobalAveragePooling3D()(volume)  # Global pooling over spatial dimensions\n    squeeze = Dense(depth, activation='sigmoid')(squeeze)  # Compute attention weights for depth\n    squeeze = Reshape((depth, 1, 1, 1))(squeeze)  # Reshape for broadcasting\n    return Multiply()([volume, squeeze])\n\n# Modality Attention\ndef modality_attention(volume):\n    \"\"\"\n    Apply modality attention to a multi-modal volume.\n    Args:\n        volume: Keras tensor of shape (batch_size, depth, height, width, modalities).\n    Returns:\n        Attention-weighted volume.\n    \"\"\"\n    modalities = volume.shape[-1]\n    weights = GlobalAveragePooling3D()(volume)  # Pool over spatial dimensions\n    weights = Dense(modalities, activation='softmax')(weights)  # Modality-wise attention\n    weights = Reshape((1, 1, 1, modalities))(weights)  # Reshape for broadcasting\n    return Multiply()([volume, weights])\n\n# Build 3D CNN with Attention\ndef build_3d_cnn_with_attention(input_shape):\n    \"\"\"\n    Build a 3D CNN with spatial, slice, and modality attention.\n    Args:\n        input_shape: Shape of the input tensor (depth, height, width, modalities).\n    Returns:\n        Compiled Keras model.\n    \"\"\"\n    input_layer = Input(shape=input_shape)\n\n    # Conv3D + Spatial Attention\n    x = Conv3D(filters=32, kernel_size=(3, 3, 3), activation='relu')(input_layer)\n    x = MaxPooling3D(pool_size=(2, 2, 2))(x)\n    x = spatial_attention(x)  # Apply spatial attention\n\n    # Conv3D + Slice Attention\n    x = Conv3D(filters=64, kernel_size=(3, 3, 3), activation='relu')(x)\n    x = MaxPooling3D(pool_size=(2, 2, 2))(x)\n    x = slice_attention(x)  # Apply slice attention\n\n    # Conv3D + Modality Attention\n    x = Conv3D(filters=128, kernel_size=(3, 3, 3), activation='relu')(x)\n    x = MaxPooling3D(pool_size=(2, 2, 2))(x)\n    x = modality_attention(x)  # Apply modality attention\n\n    # Fully connected layers\n    x = Flatten()(x)\n    x = Dense(256, activation='relu')(x)\n    x = Dropout(0.5)(x)\n    output_layer = Dense(1, activation='sigmoid')(x)\n\n    return Model(inputs=input_layer, outputs=output_layer)\n\n# Define input shape\ninput_shape = (128, 64, 64, 1)  # Example: Depth, Height, Width, Modalities\nmodel = build_3d_cnn_with_attention(input_shape)\n\n# Compile the model\noptimizer = Adam(learning_rate=0.0005)  # Set learning rate to 0.0005\nmetrics = [\n    'accuracy',\n    tf.keras.metrics.Precision(),\n    tf.keras.metrics.Recall(),\n    tf.keras.metrics.AUC(name='auc') # For AUC\n]\n\n\n\nmodel.compile(optimizer=optimizer, loss='binary_crossentropy', metrics=metrics)\n\n# Model summary\nmodel.summary()\nhistory = model.fit(X_train, y_train, validation_data=(X_val, y_val), epochs=15, batch_size=5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T20:45:13.596796Z","iopub.execute_input":"2024-11-26T20:45:13.597808Z","iopub.status.idle":"2024-11-26T20:58:33.760033Z","shell.execute_reply.started":"2024-11-26T20:45:13.597765Z","shell.execute_reply":"2024-11-26T20:58:33.759250Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score, confusion_matrix, roc_curve\nimport numpy as np\n\n# ... your training code ...\n\n# Make predictions\ny_pred_prob = model.predict(X_val)\ny_pred = (y_pred_prob > 0.5).astype(int)\n\n# Calculate metrics\naccuracy = accuracy_score(y_val, y_pred)\nprecision = precision_score(y_val, y_pred)\nrecall = recall_score(y_val, y_pred)\nf1 = f1_score(y_val, y_pred)\nauc = roc_auc_score(y_val, y_pred_prob)\ncm = confusion_matrix(y_val, y_pred)\n\n\n# --- Visualization Enhancements ---\n\n# 1. Confusion Matrix Heatmap\nplt.figure(figsize=(6, 4))\nsns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\", cbar=False,\n            xticklabels=[\"Predicted Negative\", \"Predicted Positive\"],\n            yticklabels=[\"Actual Negative\", \"Actual Positive\"])\nplt.title(\"Confusion Matrix\")\nplt.show()\n\n\n\n# 2. ROC Curve\nfpr, tpr, thresholds = roc_curve(y_val, y_pred_prob)\nplt.figure(figsize=(6, 4))\nplt.plot(fpr, tpr, label=f\"AUC = {auc:.2f}\")\nplt.plot([0, 1], [0, 1], linestyle=\"--\", color=\"gray\")  # Diagonal line for reference\nplt.xlabel(\"False Positive Rate\")\nplt.ylabel(\"True Positive Rate\")\nplt.title(\"ROC Curve\")\nplt.legend()\nplt.show()\n\n\n\n# 3. Combined Metrics Plot\nmetrics = {\"Accuracy\": accuracy, \"Precision\": precision, \"Recall\": recall, \"F1-score\": f1, \"AUC\": auc}\nplt.figure(figsize=(8, 4))\nplt.bar(metrics.keys(), metrics.values(), color=\"skyblue\")\nplt.ylim(0, 1) # Set y-axis limit for better visualization\nplt.title(\"Classification Metrics\")\nfor i, v in enumerate(metrics.values()):\n    plt.text(i, v + 0.02, f\"{v:.2f}\", ha='center') # Add labels on top of bars\nplt.show()\n\n\n\n\n# Print metrics (optional, you can remove if visualized)\nprint(f\"Accuracy: {accuracy:.2f}\")\nprint(f\"Precision: {precision:.2f}\")\nprint(f\"Recall: {recall:.2f}\")\nprint(f\"F1-score: {f1:.2f}\")\nprint(f\"AUC: {auc:.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T20:59:37.708590Z","iopub.execute_input":"2024-11-26T20:59:37.708965Z","iopub.status.idle":"2024-11-26T20:59:40.180085Z","shell.execute_reply.started":"2024-11-26T20:59:37.708932Z","shell.execute_reply":"2024-11-26T20:59:40.179158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.save('/kaggle/working/T1wCE_model.h5')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T21:10:30.574141Z","iopub.execute_input":"2024-11-26T21:10:30.575034Z","iopub.status.idle":"2024-11-26T21:10:31.119718Z","shell.execute_reply.started":"2024-11-26T21:10:30.574999Z","shell.execute_reply":"2024-11-26T21:10:31.118628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict_all_test_patients(test_folder, model, output_file=\"predictions.csv\"):\n    \"\"\"\n    Predicts the labels for all patients in the test folder and saves the results.\n    \"\"\"\n    predictions = []\n    patient_ids = []\n\n    for patient_id in sorted(os.listdir(test_folder)):\n        processed_data = preprocess_patient(patient_id, test_folder)\n        if processed_data is not None:\n            # Expand dimensions to match input shape (1, Depth, Height, Width, Modalities)\n            input_data = np.expand_dims(processed_data, axis=0)\n            \n            # Make prediction using the Keras model\n            output = model.predict(input_data, verbose=0)  # Use verbose=0 to suppress output\n            prediction = 1 if output[0] >= 0.5 else 0  # Apply threshold for binary classification\n            \n            predictions.append(prediction)\n            patient_ids.append(patient_id)\n            print(f\"{patient_id}: Prediction = {prediction}\")\n        else:\n            print(f\"Skipping patient {patient_id}: Unable to preprocess data.\")\n\n    # Save predictions to a CSV file\n    result_df = pd.DataFrame({\"PatientID\": patient_ids, \"Prediction\": predictions})\n    result_df.to_csv(output_file, index=False)\n    print(f\"Predictions saved to {output_file}.\")\n\n# Example usage\ntest_folder = \"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/test\"\npredict_all_test_patients(test_folder, model)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T20:29:20.367928Z","iopub.execute_input":"2024-11-26T20:29:20.368534Z","iopub.status.idle":"2024-11-26T20:32:07.363435Z","shell.execute_reply.started":"2024-11-26T20:29:20.368490Z","shell.execute_reply":"2024-11-26T20:32:07.362499Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install visualkeras","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T20:32:07.364692Z","iopub.execute_input":"2024-11-26T20:32:07.365050Z","iopub.status.idle":"2024-11-26T20:32:16.978794Z","shell.execute_reply.started":"2024-11-26T20:32:07.365011Z","shell.execute_reply":"2024-11-26T20:32:16.977636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import visualkeras\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\n\n# Generate the visualkeras layered view with enhancements\nvisualkeras.layered_view(\n    model,\n    legend=True,\n    to_file='/kaggle/working/model_architecture_with_legend.png',\n    scale_xy=2,\n)\n\n# Display the saved image with matplotlib enhancements\nimg = mpimg.imread('/kaggle/working/model_architecture_with_legend.png')\nplt.figure(figsize=(20, 10))\nplt.imshow(img)\nplt.axis('off')\nplt.title('Enhanced Model Architecture Visualization', fontsize=5, fontweight='bold')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T20:32:16.980654Z","iopub.execute_input":"2024-11-26T20:32:16.981091Z","iopub.status.idle":"2024-11-26T20:32:17.361203Z","shell.execute_reply.started":"2024-11-26T20:32:16.981048Z","shell.execute_reply":"2024-11-26T20:32:17.360201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ndef visualize_slices(slices, title=\"\", num_slices=5):\n    \"\"\"\n    Visualizes slices from a 3D volume.\n    \"\"\"\n    fig, axes = plt.subplots(1, num_slices, figsize=(15, 5))\n    step = max(1, len(slices) // num_slices)\n    selected_slices = slices[::step][:num_slices]\n    \n    for i, slice_img in enumerate(selected_slices):\n        axes[i].imshow(slice_img, cmap=\"gray\")\n        axes[i].axis(\"off\")\n        axes[i].set_title(f\"Slice {i * step}\")\n    \n    fig.suptitle(title)\n    plt.tight_layout()\n    plt.show()\n\ndef visualize_volume(volume, title=\"\"):\n    \"\"\"\n    Visualizes all modalities in a combined volume.\n    \"\"\"\n    num_modalities = volume.shape[-1]\n    for modality in range(num_modalities):\n        print(f\"Visualizing Modality {modality + 1}\")\n        visualize_slices(volume[..., modality], title=f\"{title} - Modality {modality + 1}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T20:32:17.362548Z","iopub.execute_input":"2024-11-26T20:32:17.363267Z","iopub.status.idle":"2024-11-26T20:32:17.370048Z","shell.execute_reply.started":"2024-11-26T20:32:17.363224Z","shell.execute_reply":"2024-11-26T20:32:17.369007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess_patientt(patient_id, folder, visualize=False):\n    modality_volumes = []\n    for modality in modalities:\n        modality_folder = os.path.join(folder, patient_id, modality)\n        if not os.path.isdir(modality_folder):\n            return None\n        \n        slices = []\n        for dcm_file in sorted(os.listdir(modality_folder)):\n            if dcm_file.endswith(\".dcm\"):\n                dcm_path = os.path.join(modality_folder, dcm_file)\n                ds = pydicom.dcmread(dcm_path)\n                slices.append(ds.pixel_array)\n        \n        if not slices:\n            return None\n\n        volume = np.stack(slices, axis=0).astype(np.float32)\n        if visualize:\n            visualize_slices(volume, title=f\"Raw Slices - Patient {patient_id}, Modality {modality}\")\n\n        # Normalize\n        volume = (volume - volume.min()) / (volume.max() - volume.min())\n        if visualize:\n            visualize_slices(volume, title=f\"Normalized Slices - Patient {patient_id}, Modality {modality}\")\n\n        # Resize slices\n        resized_slices = np.array([resize(slice, cnn_input_shape[1:3]) for slice in volume])\n        if visualize:\n            visualize_slices(resized_slices, title=f\"Resized Slices - Patient {patient_id}, Modality {modality}\")\n\n        # Pad or truncate depth\n        current_depth = resized_slices.shape[0]\n        if current_depth < cnn_input_shape[0]:\n            pad_depth = cnn_input_shape[0] - current_depth\n            resized_slices = np.pad(resized_slices, ((0, pad_depth), (0, 0), (0, 0)))\n        else:\n            resized_slices = resized_slices[:cnn_input_shape[0]]\n\n        modality_volumes.append(resized_slices)\n\n    if not modality_volumes:\n        return None\n\n    combined_volume = np.stack(modality_volumes, axis=-1)  # (Depth, Height, Width, Modalities)\n    if visualize:\n        visualize_volume(combined_volume, title=f\"Combined Volume - Patient {patient_id}\")\n\n    return combined_volume\n# Example: Visualize a single patient\npatient_id = \"00000\"  # Replace with a valid patient ID\nprocessed_data = preprocess_patientt(patient_id, train_folder, visualize=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T20:32:17.373513Z","iopub.execute_input":"2024-11-26T20:32:17.373764Z","iopub.status.idle":"2024-11-26T20:32:21.843206Z","shell.execute_reply.started":"2024-11-26T20:32:17.373739Z","shell.execute_reply":"2024-11-26T20:32:21.842329Z"}},"outputs":[],"execution_count":null}]}