{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"},{"sourceId":9980543,"sourceType":"datasetVersion","datasetId":6141371}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Load and examine RSNA dataset structure, including the CSV files and the image files.","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport pydicom\nimport matplotlib.pyplot as plt\n\n# Define the dataset directory\ndata_dir = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/'\n\n# Paths to CSV files\ntrain_csv = os.path.join(data_dir, 'train.csv')\ntrain_label_coordinates_csv = os.path.join(data_dir, 'train_label_coordinates.csv')\nsample_submission_csv = os.path.join(data_dir, 'sample_submission.csv')\n\n# Function to print section separator\ndef print_separator(title=\"\"):\n    print(\"\\n\" + \"_\" * 60)\n    if title:\n        print(f\"\\n{title}\\n\" + \"_\" * 60)\n\n# Explore CSV files\nprint_separator(\"Exploring CSV Files\")\n\n# Load train.csv\nprint_separator(\"Previewing train.csv\")\ntrain_df = pd.read_csv(train_csv)\nprint(train_df.head())\nprint(\"\\nColumns in train.csv:\")\nprint(train_df.columns)\nprint(f\"Number of rows in train.csv: {len(train_df)}\")\n\n# Load train_label_coordinates.csv\nprint_separator(\"Previewing train_label_coordinates.csv\")\ntrain_label_coordinates_df = pd.read_csv(train_label_coordinates_csv)\nprint(train_label_coordinates_df.head())\nprint(\"\\nColumns in train_label_coordinates.csv:\")\nprint(train_label_coordinates_df.columns)\nprint(f\"Number of rows in train_label_coordinates.csv: {len(train_label_coordinates_df)}\")\n\n# Load sample_submission.csv\nprint_separator(\"Previewing sample_submission.csv\")\nsample_submission_df = pd.read_csv(sample_submission_csv)\nprint(sample_submission_df.head())\nprint(\"\\nColumns in sample_submission.csv:\")\nprint(sample_submission_df.columns)\nprint(f\"Number of rows in sample_submission.csv: {len(sample_submission_df)}\")\n\n# Explore train_images folder\nprint_separator(\"Exploring train_images Directory\")\ntrain_images_dir = os.path.join(data_dir, 'train_images')\ntrain_image_studies = os.listdir(train_images_dir)\nprint(f\"Number of studies in train_images: {len(train_image_studies)}\")\nprint(\"Example studies:\", train_image_studies[:3])\n\n# Check a specific study directory\nif train_image_studies:\n    example_study_id = train_image_studies[0]\n    example_study_path = os.path.join(train_images_dir, example_study_id)\n    series_ids = os.listdir(example_study_path)\n    print_separator(f\"Exploring Example Study: {example_study_id}\")\n    print(f\"Number of series in this study: {len(series_ids)}\")\n    print(\"Series IDs:\", series_ids)\n\n    # Check a specific series\n    if series_ids:\n        example_series_id = series_ids[0]\n        example_series_path = os.path.join(example_study_path, example_series_id)\n        instance_files = os.listdir(example_series_path)\n        print_separator(f\"Exploring Example Series: {example_series_id}\")\n        print(f\"Number of images in this series: {len(instance_files)}\")\n        print(\"Instance files:\", instance_files[:3])\n\n        # Load a DICOM file\n        if instance_files:\n            example_dicom_file = os.path.join(example_series_path, instance_files[0])\n            dicom_data = pydicom.dcmread(example_dicom_file)\n            print_separator(\"DICOM File Metadata\")\n            print(dicom_data)\n            plt.figure(figsize=(6, 6))\n            plt.imshow(dicom_data.pixel_array, cmap=\"gray\")\n            plt.title(\"Example DICOM Image\")\n            plt.axis(\"off\")\n            plt.show()\n\n# Explore test_images directory\nprint_separator(\"Exploring test_images Directory\")\ntest_images_dir = os.path.join(data_dir, 'test_images')\ntest_image_studies = os.listdir(test_images_dir)\nprint(f\"Number of studies in test_images: {len(test_image_studies)}\")\nprint(\"Example studies:\", test_image_studies[:3])\n\n# Summarize dataset\nprint_separator(\"Dataset Summary\")\nprint(f\"Train Images Directory: {train_images_dir}\")\nprint(f\"Test Images Directory: {test_images_dir}\")\nprint(\"CSV Files:\", os.listdir(data_dir))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T16:25:48.683370Z","iopub.execute_input":"2024-11-25T16:25:48.683668Z","iopub.status.idle":"2024-11-25T16:25:48.917737Z","shell.execute_reply.started":"2024-11-25T16:25:48.683643Z","shell.execute_reply":"2024-11-25T16:25:48.916974Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Now we will display the first 5 studies from the training directory \nprint one dicom for each series in a study labled with its condition and level","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport pydicom\nimport matplotlib.pyplot as plt\n\n# Define the dataset directory\ndata_dir = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/'\ntrain_images_dir = os.path.join(data_dir, 'train_images')\ntrain_csv = os.path.join(data_dir, 'train.csv')\n\n# Load the training labels\ntrain_df = pd.read_csv(train_csv)\n\n# Ensure columns are clean for processing\ntrain_df = train_df.dropna()  # Drop rows with incomplete labels\ntrain_df = train_df.reset_index(drop=True)\n\n# Get the first 5 studies\nstudies = sorted(os.listdir(train_images_dir))[:5]\n\n# Plotting horizontally\ndef plot_study_horizontal(study_id, series_info):\n    num_series = len(series_info)\n    fig, axes = plt.subplots(1, num_series, figsize=(num_series * 3, 3))\n    fig.suptitle(f\"Study ID: {study_id}\", fontsize=14, y=1.05)\n    \n    for ax, (dicom_file, condition, level, severity) in zip(axes, series_info):\n        dicom_data = pydicom.dcmread(dicom_file)\n        ax.imshow(dicom_data.pixel_array, cmap=\"gray\")\n        ax.set_title(f\"{condition}\\n{level}\\n{severity}\", fontsize=8)\n        ax.axis(\"off\")\n    \n    plt.tight_layout()\n    plt.show()\n\nprint(\"\\n\" + \"_\" * 60)\nprint(\"Displaying DICOM images horizontally for the first 5 studies\\n\" + \"_\" * 60)\n\n# Loop through each study\nfor study_id in studies:\n    study_path = os.path.join(train_images_dir, study_id)\n    series_ids = sorted(os.listdir(study_path))\n    series_info = []\n    \n    print(f\"\\nStudy ID: {study_id}\")\n    print(f\"Number of series in this study: {len(series_ids)}\")\n    \n    # Loop through each series\n    for series_id in series_ids:\n        series_path = os.path.join(study_path, series_id)\n        dicom_files = sorted(os.listdir(series_path))\n        \n        # Get the first DICOM file in the series\n        if dicom_files:\n            dicom_file = os.path.join(series_path, dicom_files[0])\n            \n            # Extract condition and level from train_df\n            study_conditions = train_df[train_df['study_id'] == int(study_id)]\n            if not study_conditions.empty:\n                # Loop through conditions and levels for this study\n                for col in train_df.columns[1:]:\n                    condition_level = col.split(\"_\")\n                    condition = \"_\".join(condition_level[:-1])  # e.g., \"spinal_canal_stenosis\"\n                    level = condition_level[-1]  # e.g., \"l1_l2\"\n                    severity = study_conditions.iloc[0][col]\n                    \n                    # Only add if severity exists\n                    if severity in ['Normal/Mild', 'Moderate', 'Severe']:\n                        series_info.append((dicom_file, condition, level, severity))\n                        break  # Only consider one condition/level per series\n    \n    # Plot horizontally for this study\n    if series_info:\n        plot_study_horizontal(study_id, series_info)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T16:25:48.919793Z","iopub.execute_input":"2024-11-25T16:25:48.920179Z","iopub.status.idle":"2024-11-25T16:25:51.285211Z","shell.execute_reply.started":"2024-11-25T16:25:48.920141Z","shell.execute_reply":"2024-11-25T16:25:51.284343Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Some statiscs visulazitions of the traning data\n1.  distribution of conditions, levels,  the severity levels\n- Distribution of Conditions: Bar chart to see how many samples exist for each degenerative condition (e.g., Spinal Canal Stenosis, Left Neural Foraminal Narrowing).\n- Distribution of Levels: Bar chart to explore how the intervertebral disc levels (e.g., L1/L2, L2/L3) are distributed in the dataset.\n- Distribution of Severity Levels: Bar chart showing the proportion of severity levels (e.g., Normal/Mild, Moderate, Severe).\n2. Relationship Between Conditions and Severity Levels\nEach condition has varying levels of severity, which may follow specific patterns or trends. Exploring this can reveal which conditions are more likely to be moderate or severe.\n3. Relationship Between Levels and Severity\nCertain levels of the lumbar spine (e.g., L4/L5, L5/S1) may have a higher prevalence of severe conditions, as these levels typically bear more load. Visualizing this relationship helps identify if specific levels are more prone to degeneration.\n4. Relationship Between Levels and Conditions\nDifferent conditions may be more common at specific levels. For instance, neural foraminal narrowing may be more prevalent in lower lumbar levels (e.g., L5/S1).\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Define dataset paths\ndata_dir = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/'\ntrain_csv = os.path.join(data_dir, 'train.csv')\n\n# Load the training labels\ntrain_df = pd.read_csv(train_csv)\n\n# Drop rows with missing values\ntrain_df = train_df.dropna()\n\n# Extracting Conditions, Levels, and Severity\ntrain_df_melted = train_df.melt(id_vars=[\"study_id\"], var_name=\"Condition_Level\", value_name=\"Severity\")\n\n# Splitting \"Condition_Level\" into \"Condition\" and \"Level\"\ntrain_df_melted[\"Condition\"] = train_df_melted[\"Condition_Level\"].apply(lambda x: \"_\".join(x.split(\"_\")[:-1]))\ntrain_df_melted[\"Level\"] = train_df_melted[\"Condition_Level\"].apply(lambda x: x.split(\"_\")[-1])\n\n# Distribution of Conditions\nplt.figure(figsize=(8, 5))\nsns.countplot(data=train_df_melted, y=\"Condition\", order=train_df_melted[\"Condition\"].value_counts().index)\nplt.title(\"Distribution of Conditions\", fontsize=14)\nplt.xlabel(\"Count\")\nplt.ylabel(\"Condition\")\nplt.show()\n\n# Distribution of Levels\nplt.figure(figsize=(8, 5))\nsns.countplot(data=train_df_melted, x=\"Level\", order=train_df_melted[\"Level\"].value_counts().index)\nplt.title(\"Distribution of Levels\", fontsize=14)\nplt.xlabel(\"Level\")\nplt.ylabel(\"Count\")\nplt.show()\n\n# Distribution of Severity Levels\nplt.figure(figsize=(8, 5))\nsns.countplot(data=train_df_melted, x=\"Severity\", order=train_df_melted[\"Severity\"].value_counts().index)\nplt.title(\"Distribution of Severity Levels\", fontsize=14)\nplt.xlabel(\"Severity\")\nplt.ylabel(\"Count\")\nplt.show()\n\n# Relationship between Condition and Severity\nplt.figure(figsize=(10, 6))\nsns.countplot(data=train_df_melted, x=\"Condition\", hue=\"Severity\", order=train_df_melted[\"Condition\"].value_counts().index)\nplt.title(\"Relationship between Condition and Severity Levels\", fontsize=14)\nplt.xlabel(\"Condition\")\nplt.ylabel(\"Count\")\nplt.legend(title=\"Severity\")\nplt.xticks(rotation=45)\nplt.show()\n\n# Relationship between Level and Severity\nplt.figure(figsize=(10, 6))\nsns.countplot(data=train_df_melted, x=\"Level\", hue=\"Severity\", order=train_df_melted[\"Level\"].value_counts().index)\nplt.title(\"Relationship between Level and Severity Levels\", fontsize=14)\nplt.xlabel(\"Level\")\nplt.ylabel(\"Count\")\nplt.legend(title=\"Severity\")\nplt.show()\n\n# Relationship between Condition and Level\nplt.figure(figsize=(10, 6))\nheatmap_data = train_df_melted.groupby([\"Condition\", \"Level\"]).size().unstack(fill_value=0)\nsns.heatmap(heatmap_data, annot=True, fmt=\"d\", cmap=\"Blues\")\nplt.title(\"Relationship between Condition and Level\", fontsize=14)\nplt.xlabel(\"Level\")\nplt.ylabel(\"Condition\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T16:25:51.286471Z","iopub.execute_input":"2024-11-25T16:25:51.286816Z","iopub.status.idle":"2024-11-25T16:25:54.557074Z","shell.execute_reply.started":"2024-11-25T16:25:51.286778Z","shell.execute_reply":"2024-11-25T16:25:54.556182Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Multi-label Analysis\nEach study contains multiple labels (conditions at various levels). Visualizing how often these conditions co-occur in a single study can provide insights into correlations between conditions.\na Correlation Heatmap show correlations between conditions, levels, and severity labels to identify commonly co-occurring degenerative conditions.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport numpy as np\n\n# Load the train.csv file\ndata_dir = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/'\ntrain_df = pd.read_csv(f\"{data_dir}/train.csv\")\n\n# Convert severity labels to numerical values\nseverity_mapping = {'Normal/Mild': 0, 'Moderate': 1, 'Severe': 2}\nfor col in train_df.columns[1:]:\n    train_df[col] = train_df[col].map(severity_mapping)\n\n# Create a binary matrix for multi-label analysis\nmulti_label_df = train_df.drop(columns=[\"study_id\"]).notnull().astype(int)  # 1 if condition exists, else 0\n\n# Calculate correlations between conditions\ncorrelation_matrix = multi_label_df.corr()\n\n# Plot the heatmap\nplt.figure(figsize=(12, 8))\nsns.heatmap(\n    correlation_matrix,\n    annot=True, \n    cmap=\"coolwarm\", \n    fmt=\".2f\",\n    xticklabels=correlation_matrix.columns, \n    yticklabels=correlation_matrix.columns,\n    cbar_kws={\"label\": \"Correlation Coefficient\"}\n)\nplt.title(\"Correlation Heatmap of Conditions and Levels\")\nplt.xticks(rotation=45, ha='right')\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T16:25:54.558987Z","iopub.execute_input":"2024-11-25T16:25:54.559265Z","iopub.status.idle":"2024-11-25T16:25:56.340004Z","shell.execute_reply.started":"2024-11-25T16:25:54.559238Z","shell.execute_reply":"2024-11-25T16:25:56.339150Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Now lets do classification:\nWe are tasked with classifying the severity levels (Normal/Mild, Moderate, Severe) for five lumbar spine degenerative conditions across intervertebral disc levels (L1/L2 to L5/S1).\n\n- The target output is a probability for each severity level for each row in the test dataset.\n- We also need to calculate an any_severe_spinal label for the spinal canal stenosis condition across all levels for each study.\nThe evaluation metric will use log loss weighted by severity and calculate an additional log loss for the any_severe_spinal label.","metadata":{}},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport os\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.utils import to_categorical\nimport pydicom\nimport numpy as np\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\n\n# Path to the dataset\ndata_dir = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/'\ntrain_csv = pd.read_csv(f\"{data_dir}/train.csv\")\ntrain_images_dir = f\"{data_dir}/train_images\"\n\n# Severity mapping\nseverity_mapping = {\"Normal/Mild\": 0, \"Moderate\": 1, \"Severe\": 2}\n\n# Reshape train labels for multi-class classification\nlabels = []\nrow_ids = []\n\nfor index, row in train_csv.iterrows():\n    study_id = row[\"study_id\"]\n    for col in train_csv.columns[1:]:\n        condition_level = col.split(\"_\")\n        if pd.notna(row[col]):  # If severity level is provided\n            labels.append({\n                \"row_id\": f\"{study_id}_{col}\",\n                \"condition\": \"_\".join(condition_level[:-2]),\n                \"level\": \"_\".join(condition_level[-2:]),\n                \"severity\": severity_mapping[row[col]],\n            })\ntrain_labels = pd.DataFrame(labels)\n\n# Split into train and validation\ntrain_labels, val_labels = train_test_split(train_labels, test_size=0.2, stratify=train_labels[\"severity\"], random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:04:59.777106Z","iopub.execute_input":"2024-11-25T17:04:59.778048Z","iopub.status.idle":"2024-11-25T17:05:00.304003Z","shell.execute_reply.started":"2024-11-25T17:04:59.778000Z","shell.execute_reply":"2024-11-25T17:05:00.302891Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:05:08.305593Z","iopub.execute_input":"2024-11-25T17:05:08.306413Z","iopub.status.idle":"2024-11-25T17:05:08.315233Z","shell.execute_reply.started":"2024-11-25T17:05:08.306376Z","shell.execute_reply":"2024-11-25T17:05:08.314349Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"----------------------------------------\")\nval_labels.head() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T16:26:08.007828Z","iopub.execute_input":"2024-11-25T16:26:08.008185Z","iopub.status.idle":"2024-11-25T16:26:08.026412Z","shell.execute_reply.started":"2024-11-25T16:26:08.008135Z","shell.execute_reply":"2024-11-25T16:26:08.025561Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Image Loading and Preprocessing","metadata":{}},{"cell_type":"markdown","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.utils import to_categorical\nimport os\nimport numpy as np\nimport cv2\nimport pydicom\n\n# Constants\nIMG_SIZE = 224  # Image size for resizing\nBATCH_SIZE = 16  # Batch size for data generator\nBASE_DIR = \"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images\"  # Dataset path\n\ndef load_dicom_image(path, size):\n    \"\"\"Load and preprocess a single DICOM image.\"\"\"\n    try:\n        # Read DICOM file\n        dicom = pydicom.dcmread(path)\n        img = dicom.pixel_array\n        # Resize image\n        img = cv2.resize(img, (size, size))\n        # Normalize to [0, 1]\n        img = (img - np.min(img)) / (np.max(img) - np.min(img))\n        return img\n    except Exception as e:\n        print(f\"Error loading DICOM image from {path}: {e}\")\n        return None\n\ndef load_study_images(study_id, base_dir=BASE_DIR, img_size=IMG_SIZE):\n    \"\"\"Load one image per series from a study.\"\"\"\n    study_path = os.path.join(base_dir, study_id)\n    images = []\n    \n    if not os.path.exists(study_path):\n        print(f\"Study path not found: {study_path}\")\n        return np.array(images)\n\n    for series_id in os.listdir(study_path):\n        series_path = os.path.join(study_path, series_id)\n        instances = os.listdir(series_path)\n\n        if not instances:\n            continue\n\n        # Pick the middle instance of the series\n        middle_instance = instances[len(instances) // 2]\n        instance_path = os.path.join(series_path, middle_instance)\n        \n        img = load_dicom_image(instance_path, img_size)\n        if img is not None:\n            images.append(img)\n    \n    return np.array(images)\n\n# Data augmentation setup\ndatagen = ImageDataGenerator(\n    rotation_range=10,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    horizontal_flip=True\n)\ndef data_generator(labels, base_dir=BASE_DIR, batch_size=BATCH_SIZE):\n    while True:\n        for i in range(0, len(labels), batch_size):\n            batch_labels = labels.iloc[i:i + batch_size]\n            batch_images = []\n            batch_targets = []\n\n            for _, row in batch_labels.iterrows():\n                # Extract the study ID from the row\n                study_id = row[\"row_id\"].split(\"_\")[0]\n                # Load images for the study\n                images = load_study_images(study_id, base_dir)\n\n                for img in images:\n                    # Ensure image has 4 dimensions (add channel dimension if needed)\n                    if img.ndim == 2:  # Grayscale image (height, width)\n                        img = np.expand_dims(img, axis=-1)  # Convert to (height, width, 1)\n                    elif img.ndim == 3 and img.shape[-1] != 1:  # RGB image (height, width, 3)\n                        img = img[..., :3]  # Ensure it has 3 channels\n\n                    batch_images.append(img)\n                    batch_targets.append(to_categorical(row[\"severity\"], num_classes=3))\n\n            if batch_images:\n                # Convert lists to numpy arrays\n                batch_images = np.array(batch_images)  # Shape: (batch_size, height, width, channels)\n                batch_targets = np.array(batch_targets)  # Shape: (batch_size, num_classes)\n                yield batch_images, batch_targets","metadata":{"execution":{"iopub.status.busy":"2024-11-25T17:33:58.592443Z","iopub.execute_input":"2024-11-25T17:33:58.593022Z","iopub.status.idle":"2024-11-25T17:33:58.604194Z","shell.execute_reply.started":"2024-11-25T17:33:58.592987Z","shell.execute_reply":"2024-11-25T17:33:58.603257Z"}}},{"cell_type":"markdown","source":"train_gen = data_generator(train_labels, train_images_dir, BATCH_SIZE)\nbatch_images, batch_targets = next(train_gen)\n\nprint(\"Batch Images Shape:\", batch_images.shape)  # Should be (batch_size, 224, 224, 1)\nprint(\"Batch Targets Shape:\", batch_targets.shape)","metadata":{"execution":{"iopub.status.busy":"2024-11-25T17:34:07.049431Z","iopub.execute_input":"2024-11-25T17:34:07.050288Z","iopub.status.idle":"2024-11-25T17:34:07.593711Z","shell.execute_reply.started":"2024-11-25T17:34:07.050251Z","shell.execute_reply":"2024-11-25T17:34:07.592810Z"}}},{"cell_type":"markdown","source":"# Test `load_dicom_image`\ntest_image_path = os.path.join(data_dir, '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images/100206310/1012284084/1.dcm')  # Update with an actual file path\ntest_image = load_dicom_image(test_image_path, IMG_SIZE)\n\nprint(\"Test Image Shape:\", test_image.shape)\nprint(\"Test Image Pixel Range:\", test_image.min(), test_image.max())\n\n# Visualize the image\nimport matplotlib.pyplot as plt\nplt.imshow(test_image, cmap='gray')\nplt.title(\"Loaded DICOM Image\")\nplt.colorbar()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-25T16:26:09.800235Z","iopub.execute_input":"2024-11-25T16:26:09.800528Z","iopub.status.idle":"2024-11-25T16:26:10.146121Z","shell.execute_reply.started":"2024-11-25T16:26:09.800500Z","shell.execute_reply":"2024-11-25T16:26:10.145250Z"}}},{"cell_type":"markdown","source":"# Model Setting (EfficientNetB0)","metadata":{}},{"cell_type":"markdown","source":"from tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout\nfrom tensorflow.keras.optimizers import Adam\nimport tensorflow as tf\nimport numpy as np\n\n# Ensure the input shape matches the preprocessed image format (3-channel RGB)\nbase_model = EfficientNetB0(weights=\"imagenet\", include_top=False, input_shape=(IMG_SIZE, IMG_SIZE, 3))\n\n# Add custom classification layers\nx = GlobalAveragePooling2D()(base_model.output)\nx = Dropout(0.5)(x)\noutput = Dense(3, activation=\"softmax\")(x)  # 3 classes: normal/mild, moderate, severe\n\n# Create the model\nmodel = Model(inputs=base_model.input, outputs=output)\n\n# Class weights for weighted categorical cross-entropy\nweights = np.array([1, 2, 4], dtype=np.float32)  # Adjust weights if needed based on the dataset distribution\n\ndef weighted_loss(y_true, y_pred):\n    \"\"\"Custom weighted loss function.\"\"\"\n    epsilon = tf.keras.backend.epsilon()  # Small constant to avoid log(0)\n    y_pred = tf.clip_by_value(y_pred, epsilon, 1 - epsilon)\n    return -tf.reduce_sum(weights * y_true * tf.math.log(y_pred))\n\n# Compile the model\nmodel.compile(\n    optimizer=Adam(learning_rate=0.001),\n    loss=weighted_loss,\n    metrics=[\"accuracy\"]\n)\n\n# Print the model summary\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-25T16:26:10.147128Z","iopub.execute_input":"2024-11-25T16:26:10.147365Z","iopub.status.idle":"2024-11-25T16:26:13.475910Z","shell.execute_reply.started":"2024-11-25T16:26:10.147340Z","shell.execute_reply":"2024-11-25T16:26:13.475026Z"}}},{"cell_type":"markdown","source":"# EfficientNetB0 Train the model","metadata":{}},{"cell_type":"markdown","source":"import math\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping\n\n# Train and validation generators \ntrain_gen = data_generator(train_labels, train_images_dir, BATCH_SIZE)\nval_gen = data_generator(val_labels, train_images_dir, BATCH_SIZE)\n\n# Steps per epoch for training and validation\nsteps_per_epoch = math.ceil(len(train_labels) / BATCH_SIZE)\nvalidation_steps = math.ceil(len(val_labels) / BATCH_SIZE)\n\n# Callbacks for training\ncheckpoint = ModelCheckpoint(\n    filepath=\"best_model.keras\",  # Use the `.keras` extension for saving the full model\n    monitor=\"val_loss\",\n    save_best_only=True,\n    save_weights_only=False,  # Save the full model (architecture + weights)\n    mode=\"min\"\n)\n\nearly_stopping = EarlyStopping(\n    monitor=\"val_loss\",\n    patience=10,\n    restore_best_weights=True\n)\n# Train the model\nhistory = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=30,\n    steps_per_epoch=steps_per_epoch,\n    validation_steps=validation_steps,\n    callbacks=[checkpoint, early_stopping]\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-25T16:26:13.476986Z","iopub.execute_input":"2024-11-25T16:26:13.477257Z","iopub.status.idle":"2024-11-25T16:28:23.471037Z","shell.execute_reply.started":"2024-11-25T16:26:13.477231Z","shell.execute_reply":"2024-11-25T16:28:23.469762Z"}}},{"cell_type":"markdown","source":"# EfficientNetB4","metadata":{}},{"cell_type":"markdown","source":"from tensorflow.keras.applications import EfficientNetB4\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout\nfrom tensorflow.keras.optimizers import Adam\nimport tensorflow as tf\nimport numpy as np\n\n# Ensure the input shape matches the preprocessed image format (3-channel RGB)\n# Updated to EfficientNetB4 with settings as listed\nbase_model = EfficientNetB4(weights=\"imagenet\", include_top=False, input_shape=(IMG_SIZE, IMG_SIZE, 3))\n\n# Add custom classification layers\nx = GlobalAveragePooling2D()(base_model.output)  # Feature extraction\nx = Dropout(0.5)(x)  # Regularization\noutput = Dense(3, activation=\"softmax\")(x)  # 3 classes: normal/mild, moderate, severe\n\n# Create the model\nmodel = Model(inputs=base_model.input, outputs=output)\n\n# Class weights for weighted categorical cross-entropy\nweights = np.array([1, 2, 4], dtype=np.float32)  # Adjust weights if needed based on the dataset distribution\n\ndef weighted_loss(y_true, y_pred):\n    \"\"\"Custom weighted loss function.\"\"\"\n    epsilon = tf.keras.backend.epsilon()  # Small constant to avoid log(0)\n    y_pred = tf.clip_by_value(y_pred, epsilon, 1 - epsilon)\n    return -tf.reduce_sum(weights * y_true * tf.math.log(y_pred))\n\n# Compile the model\nmodel.compile(\n    optimizer=Adam(learning_rate=0.001),  # Optimizer with learning rate\n    loss=weighted_loss,                   # Custom loss function\n    metrics=[\"accuracy\"]                  # Metric for evaluation\n)\n\n# Print the model summary\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-25T17:05:50.413249Z","iopub.execute_input":"2024-11-25T17:05:50.413568Z","iopub.status.idle":"2024-11-25T17:05:54.030769Z","shell.execute_reply.started":"2024-11-25T17:05:50.413542Z","shell.execute_reply":"2024-11-25T17:05:54.029887Z"}}},{"cell_type":"markdown","source":"# EfficientNetB4 Train the model\n","metadata":{}},{"cell_type":"markdown","source":"import math\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping\n\n# Train and validation generators \ntrain_gen = data_generator(train_labels, train_images_dir, BATCH_SIZE)\nval_gen = data_generator(val_labels, train_images_dir, BATCH_SIZE)\n\n# Steps per epoch for training and validation\nsteps_per_epoch = math.ceil(len(train_labels) / BATCH_SIZE)\nvalidation_steps = math.ceil(len(val_labels) / BATCH_SIZE)\n\n# Callbacks for training\ncheckpoint = ModelCheckpoint(\n    filepath=\"best_model.keras\",  # Use the `.keras` extension for saving the full model\n    monitor=\"val_loss\",\n    save_best_only=True,\n    save_weights_only=False,  # Save the full model (architecture + weights)\n    mode=\"min\"\n)\n\nearly_stopping = EarlyStopping(\n    monitor=\"val_loss\",\n    patience=10,\n    restore_best_weights=True\n)\n# Train the model\nhistory = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=10,\n    steps_per_epoch=steps_per_epoch,\n    validation_steps=validation_steps,\n    callbacks=[checkpoint, early_stopping]\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-25T17:06:45.286831Z","iopub.execute_input":"2024-11-25T17:06:45.287718Z","iopub.status.idle":"2024-11-25T17:11:21.206071Z","shell.execute_reply.started":"2024-11-25T17:06:45.287683Z","shell.execute_reply":"2024-11-25T17:11:21.204624Z"}}},{"cell_type":"markdown","source":"# Special Data Generator for Inception","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.utils import to_categorical\nimport os\nimport numpy as np\nimport cv2\nimport pydicom\n\n# Constants\nIMG_SIZE = 299  # Image size for resizing (specific to InceptionV3)\nBATCH_SIZE = 16  # Batch size for data generator\nBASE_DIR = \"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images\"  # Dataset path\n\ndef load_dicom_image(path, size):\n    \"\"\"Load and preprocess a single DICOM image.\"\"\"\n    try:\n        # Read DICOM file\n        dicom = pydicom.dcmread(path)\n        img = dicom.pixel_array\n        # Resize image\n        img = cv2.resize(img, (size, size))\n        # Normalize to [0, 1]\n        img = (img - np.min(img)) / (np.max(img) - np.min(img))\n        # Convert grayscale to RGB\n        if img.ndim == 2:  # Grayscale image (height, width)\n            img = np.stack((img,) * 3, axis=-1)  # Duplicate channels to make it (height, width, 3)\n        return img\n    except Exception as e:\n        print(f\"Error loading DICOM image from {path}: {e}\")\n        return None\n\ndef load_study_images(study_id, base_dir=BASE_DIR, img_size=IMG_SIZE):\n    \"\"\"Load one image per series from a study.\"\"\"\n    study_path = os.path.join(base_dir, study_id)\n    images = []\n    \n    if not os.path.exists(study_path):\n        print(f\"Study path not found: {study_path}\")\n        return np.array(images)\n\n    for series_id in os.listdir(study_path):\n        series_path = os.path.join(study_path, series_id)\n        instances = os.listdir(series_path)\n\n        if not instances:\n            continue\n\n        # Pick the middle instance of the series\n        middle_instance = instances[len(instances) // 2]\n        instance_path = os.path.join(series_path, middle_instance)\n        \n        img = load_dicom_image(instance_path, img_size)\n        if img is not None:\n            images.append(img)\n    \n    return np.array(images)\n\n# Data augmentation setup\ndatagen = ImageDataGenerator(\n    rotation_range=10,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    horizontal_flip=True\n)\n\ndef data_generator(labels, base_dir=BASE_DIR, batch_size=BATCH_SIZE):\n    while True:\n        for i in range(0, len(labels), batch_size):\n            batch_labels = labels.iloc[i:i + batch_size]\n            batch_images = []\n            batch_targets = []\n\n            for _, row in batch_labels.iterrows():\n                # Extract the study ID from the row\n                study_id = row[\"row_id\"].split(\"_\")[0]\n                # Load images for the study\n                images = load_study_images(study_id, base_dir)\n\n                for img in images:\n                    batch_images.append(img)\n                    batch_targets.append(to_categorical(row[\"severity\"], num_classes=3))\n\n            if batch_images:\n                # Convert lists to numpy arrays\n                batch_images = np.array(batch_images, dtype=np.float32)  # Shape: (batch_size, height, width, 3)\n                batch_targets = np.array(batch_targets)  # Shape: (batch_size, num_classes)\n\n                # Apply data augmentation\n                yield next(datagen.flow(batch_images, batch_targets, batch_size=batch_size))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:44:02.704141Z","iopub.execute_input":"2024-11-25T17:44:02.704884Z","iopub.status.idle":"2024-11-25T17:44:02.715820Z","shell.execute_reply.started":"2024-11-25T17:44:02.704852Z","shell.execute_reply":"2024-11-25T17:44:02.714957Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Runing InceptionV3","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications import InceptionV3\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout\nfrom tensorflow.keras.optimizers import Adam\nimport tensorflow as tf\nimport numpy as np\n\n# Ensure the input shape matches the preprocessed image format (3-channel RGB)\n#IMG_SIZE = 299  # InceptionV3 typically uses 299x299 images\nbase_model = InceptionV3(weights=\"imagenet\", include_top=False, input_shape=(IMG_SIZE, IMG_SIZE, 3))\n\n# Add custom classification layers\nx = GlobalAveragePooling2D()(base_model.output)  # Feature extraction\nx = Dropout(0.5)(x)  # Regularization\noutput = Dense(3, activation=\"softmax\")(x)  # 3 classes: normal/mild, moderate, severe\n\n# Create the model\nmodel = Model(inputs=base_model.input, outputs=output)\n\n# Class weights for weighted categorical cross-entropy\nweights = np.array([1, 2, 4], dtype=np.float32)  # Adjust weights if needed based on the dataset distribution\n\ndef weighted_loss(y_true, y_pred):\n    \"\"\"Custom weighted loss function.\"\"\"\n    epsilon = tf.keras.backend.epsilon()  # Small constant to avoid log(0)\n    y_pred = tf.clip_by_value(y_pred, epsilon, 1 - epsilon)\n    return -tf.reduce_sum(weights * y_true * tf.math.log(y_pred))\n\n# Compile the model\nmodel.compile(\n    optimizer=Adam(learning_rate=0.001),  # Optimizer with learning rate\n    loss=weighted_loss,                   # Custom loss function\n    metrics=[\"accuracy\"]                  # Metric for evaluation\n)\n\n# Print the model summary\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:41:49.671387Z","iopub.execute_input":"2024-11-25T17:41:49.671714Z","iopub.status.idle":"2024-11-25T17:41:51.487660Z","shell.execute_reply.started":"2024-11-25T17:41:49.671687Z","shell.execute_reply":"2024-11-25T17:41:51.486773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import math\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping\n\n# Train and validation generators \ntrain_gen = data_generator(train_labels, train_images_dir, BATCH_SIZE)\nval_gen = data_generator(val_labels, train_images_dir, BATCH_SIZE)\n\n# Steps per epoch for training and validation\nsteps_per_epoch = math.ceil(len(train_labels) / BATCH_SIZE)\nvalidation_steps = math.ceil(len(val_labels) / BATCH_SIZE)\n\n# Callbacks for training\ncheckpoint = ModelCheckpoint(\n    filepath=\"best_model.keras\",  # Use the `.keras` extension for saving the full model\n    monitor=\"val_loss\",\n    save_best_only=True,\n    save_weights_only=False,  # Save the full model (architecture + weights)\n    mode=\"min\"\n)\n\nearly_stopping = EarlyStopping(\n    monitor=\"val_loss\",\n    patience=10,\n    restore_best_weights=True\n)\n# Train the model\nhistory = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=10,\n    steps_per_epoch=steps_per_epoch,\n    validation_steps=validation_steps,\n    callbacks=[checkpoint, early_stopping]\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:44:12.039546Z","iopub.execute_input":"2024-11-25T17:44:12.040226Z","iopub.status.idle":"2024-11-25T17:57:43.200036Z","shell.execute_reply.started":"2024-11-25T17:44:12.040193Z","shell.execute_reply":"2024-11-25T17:57:43.198356Z"}},"outputs":[],"execution_count":null}]}