{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":30762,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Key points about the challenge: \n\n**The primary objective is to classify five specific conditions:** \n1. Left Neural Foraminal Narrowing: Narrowing of the neural foramen on the left side. \n2. Right Neural Foraminal Narrowing: Narrowing of the neural foramen on the  right side.  \n3. Left Subarticular Stenosis: Stenosis in the left subarticular region. \n4. Right Subarticular Stenosis: Stenosis in the right subarticular region. \n5. Spinal Canal Stenosis: Narrowing of the spinal canal. \n\n**These conditions are evaluated at five inter-vertebral disc levels:** \n- L1/L2\n- L2/L3\n- L3/L4\n- L4/L5\n- L5/S1\n\n**For each condition, the severity levels are:** \n1. Normal/Mild \n2. Moderate \n3. Severe ","metadata":{}},{"cell_type":"markdown","source":"## Importing Libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pydicom\nimport cv2\nfrom glob import glob\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models, applications\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import log_loss\nfrom skimage import exposure\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as transforms\nimport torchvision.models as models\nimport warnings\nwarnings.filterwarnings('ignore')\n\n\n# Set random seeds for reproducibility\nnp.random.seed(42)\ntorch.manual_seed(42)","metadata":{"execution":{"iopub.status.busy":"2024-09-23T15:23:03.340743Z","iopub.execute_input":"2024-09-23T15:23:03.341196Z","iopub.status.idle":"2024-09-23T15:23:20.859385Z","shell.execute_reply.started":"2024-09-23T15:23:03.341151Z","shell.execute_reply":"2024-09-23T15:23:20.858163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading CSV Files ","metadata":{}},{"cell_type":"code","source":"train_csv_path = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train.csv'\ntrain_label_coords_path = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv'\ntrain_series_desc_path = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_series_descriptions.csv'\nsample_submission_path = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/sample_submission.csv'\ntest_series_desc_path = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/test_series_descriptions.csv'\n\n\ntrain_df = pd.read_csv(train_csv_path)\ntrain_label_coords_df = pd.read_csv(train_label_coords_path)\ntrain_series_desc_df = pd.read_csv(train_series_desc_path)\nsample_submission_df = pd.read_csv(sample_submission_path)\ntest_series_desc_df = pd.read_csv(test_series_desc_path)","metadata":{"execution":{"iopub.status.busy":"2024-09-23T14:40:27.740662Z","iopub.execute_input":"2024-09-23T14:40:27.741054Z","iopub.status.idle":"2024-09-23T14:40:27.923738Z","shell.execute_reply.started":"2024-09-23T14:40:27.741016Z","shell.execute_reply":"2024-09-23T14:40:27.922521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Shape of Train_df:\", train_df.shape)\nprint(\"Shape of train_label_coords_df:\", train_label_coords_df.shape )\nprint(\"Shape of train_series_desc_df:\", train_series_desc_df.shape)\nprint(\"Shape of test_series_desc_df:\", test_series_desc_df.shape)\nprint(\"Shape of Submission:\", sample_submission_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-09-23T14:47:41.826737Z","iopub.execute_input":"2024-09-23T14:47:41.827137Z","iopub.status.idle":"2024-09-23T14:47:41.834343Z","shell.execute_reply.started":"2024-09-23T14:47:41.827098Z","shell.execute_reply":"2024-09-23T14:47:41.833154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-23T14:48:51.156468Z","iopub.execute_input":"2024-09-23T14:48:51.156922Z","iopub.status.idle":"2024-09-23T14:48:51.170234Z","shell.execute_reply.started":"2024-09-23T14:48:51.156877Z","shell.execute_reply":"2024-09-23T14:48:51.169059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Shape of test_series_desc_df:\", test_series_desc_df.shape)\ntest_series_desc_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-23T14:45:46.620106Z","iopub.execute_input":"2024-09-23T14:45:46.620549Z","iopub.status.idle":"2024-09-23T14:45:46.632475Z","shell.execute_reply.started":"2024-09-23T14:45:46.620504Z","shell.execute_reply":"2024-09-23T14:45:46.631295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Shape of train_series_desc_df:\", train_series_desc_df.shape)\ntrain_series_desc_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-23T14:45:01.855149Z","iopub.execute_input":"2024-09-23T14:45:01.855697Z","iopub.status.idle":"2024-09-23T14:45:01.869647Z","shell.execute_reply.started":"2024-09-23T14:45:01.855621Z","shell.execute_reply":"2024-09-23T14:45:01.868466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Shape of train_label_coords_df:\", train_label_coords_df.shape )\ntrain_label_coords_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-23T14:44:07.305534Z","iopub.execute_input":"2024-09-23T14:44:07.306097Z","iopub.status.idle":"2024-09-23T14:44:07.324846Z","shell.execute_reply.started":"2024-09-23T14:44:07.306039Z","shell.execute_reply":"2024-09-23T14:44:07.323501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Shape of Train_df:\", train_df.shape)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-23T14:43:19.870740Z","iopub.execute_input":"2024-09-23T14:43:19.871523Z","iopub.status.idle":"2024-09-23T14:43:19.897256Z","shell.execute_reply.started":"2024-09-23T14:43:19.871475Z","shell.execute_reply":"2024-09-23T14:43:19.896161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"# Counting the number of studies and series\nnum_studies = train_df['study_id'].nunique()\nnum_series = train_series_desc_df['series_id'].nunique()\nprint(\"Number of unique studies:\", num_studies)\nprint(\"Number of unique series:\", num_series)\n\n# Counting the number of conditions and levels \nconditions = [col for col in train_df.columns if col != 'study_id']\n\nnum_conditions = len(conditions)\nprint(\"Number of Conditions-level combinations:\", num_conditions)","metadata":{"execution":{"iopub.status.busy":"2024-09-23T14:56:42.476781Z","iopub.execute_input":"2024-09-23T14:56:42.477618Z","iopub.status.idle":"2024-09-23T14:56:42.485975Z","shell.execute_reply.started":"2024-09-23T14:56:42.477569Z","shell.execute_reply":"2024-09-23T14:56:42.484882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Converting all condition columns form train.csv into one melted dataframe for better visualization \nconditions_columns = [col for col in train_df.columns if 'stenosis' in col or 'narrowing' in col]\nmelted_df = train_df.melt(id_vars='study_id', value_vars=conditions_columns, var_name='condition', value_name='severity')\n\n# ploting the distribution of severity levels across all condtions \nplt.figure(figsize=(12, 6))\nsns.countplot(data=melted_df, x='severity', order=['Normal/Mild', 'Moderate', 'Severe'])\nplt.title('Distribution of Severity Levels Across All Conditions')\nplt.show()\n\n# Plotting the condition-wise distribution of severity levels\nplt.figure(figsize=(15, 8))\nsns.countplot(data=melted_df, x='condition', hue='severity', order=conditions_columns)\nplt.title('Condition-wise Severity Distribution Across Vertebrae Levels')\nplt.xticks(rotation=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-23T15:03:47.258991Z","iopub.execute_input":"2024-09-23T15:03:47.259439Z","iopub.status.idle":"2024-09-23T15:03:48.662444Z","shell.execute_reply.started":"2024-09-23T15:03:47.259396Z","shell.execute_reply":"2024-09-23T15:03:48.661051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## MRI Image Visualization","metadata":{}},{"cell_type":"code","source":"dicom_file_path = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images/100206310/1012284084/1.dcm'\n\ndicom_image = pydicom.dcmread(dicom_file_path)\n\nplt.figure(figsize=(8,8))\nplt.imshow(dicom_image.pixel_array, cmap=plt.cm.bone)\nplt.title('DICOM Image')\nplt.axis('off')\nplt.show","metadata":{"execution":{"iopub.status.busy":"2024-09-23T15:33:14.330897Z","iopub.execute_input":"2024-09-23T15:33:14.331969Z","iopub.status.idle":"2024-09-23T15:33:14.836596Z","shell.execute_reply.started":"2024-09-23T15:33:14.331907Z","shell.execute_reply":"2024-09-23T15:33:14.834998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load train_label_coordinates.csv\ntrain_label_coords_path = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv'\ntrain_label_coords_df = pd.read_csv(train_label_coords_path)\n\n# Set the study_id, series_id, and instance_number\nstudy_id = 100206310\nseries_id = 1012284084\n# Set instance_number to 20 (since this one has coordinates)\ninstance_number = 20\n\n# Filter the dataframe to get the corresponding coordinates for the specific image\ncoords_df = train_label_coords_df[(train_label_coords_df['study_id'] == study_id) &\n                                  (train_label_coords_df['series_id'] == series_id) &\n                                  (train_label_coords_df['instance_number'] == instance_number)]\n\n# Print the filtered dataframe to check if there are any entries\nprint(\"Filtered Coordinates DataFrame for instance_number 20:\")\nprint(coords_df)\n\n# Load the DICOM image for instance_number 20\ndicom_file_path = f'/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images/{study_id}/{series_id}/{instance_number}.dcm'\ndicom_image = pydicom.dcmread(dicom_file_path)\n\n# Plot the DICOM image\nplt.figure(figsize=(6, 6))\nplt.imshow(dicom_image.pixel_array, cmap=plt.cm.bone)\n\n# If there are rows in coords_df, plot the coordinates\nif not coords_df.empty:\n    for index, row in coords_df.iterrows():\n        x, y = row['x'], row['y']\n        print(f\"Condition: {row['condition']}, X: {x}, Y: {y}\")  # Print coordinates for debugging\n        plt.scatter(x, y, color='red', s=300, label=row['condition'], edgecolor='black')  # Increase marker size\nelse:\n    print(f\"No coordinates found for instance_number {instance_number}\")\n\nplt.title('DICOM Image with Label Coordinates (instance_number 20)')\nplt.legend(loc='upper right')\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-23T15:34:12.937162Z","iopub.execute_input":"2024-09-23T15:34:12.937790Z","iopub.status.idle":"2024-09-23T15:34:13.432896Z","shell.execute_reply.started":"2024-09-23T15:34:12.937740Z","shell.execute_reply":"2024-09-23T15:34:13.431716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing DICOMs and Extracting ROIs ","metadata":{}},{"cell_type":"code","source":"\n\n# Load train_label_coordinates.csv\ntrain_label_coords_path = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv'\ntrain_label_coords_df = pd.read_csv(train_label_coords_path)\n\n# Function to preprocess DICOM images (resizing and normalization)\ndef preprocess_dicom_image(dicom_file_path, target_size=(224, 224)):\n    # Load the DICOM image\n    dicom_image = pydicom.dcmread(dicom_file_path).pixel_array\n    original_height, original_width = dicom_image.shape\n    \n    # Normalize the image\n    dicom_image = dicom_image / np.max(dicom_image)  # Normalize to [0, 1]\n    \n    # Resize the image to the target size\n    resized_image = cv2.resize(dicom_image, target_size)\n    \n    return resized_image, original_width, original_height\n\n# Function to adjust coordinates for resized image\ndef adjust_coordinates(x, y, original_width, original_height, target_width, target_height):\n    # Calculate the scaling factors for both width and height\n    x_rescaled = x * (target_width / original_width)\n    y_rescaled = y * (target_height / original_height)\n    \n    return int(x_rescaled), int(y_rescaled)\n\n# Function to extract ROI based on x, y coordinates and a fixed window size\ndef extract_roi(image, x, y, window_size=50):\n    # Ensure coordinates are within bounds\n    x, y = int(x), int(y)\n    height, width = image.shape\n    \n    # Calculate ROI boundaries\n    x_min = max(0, x - window_size // 2)\n    x_max = min(width, x + window_size // 2)\n    y_min = max(0, y - window_size // 2)\n    y_max = min(height, y + window_size // 2)\n    \n    # Extract the ROI\n    roi = image[y_min:y_max, x_min:x_max]\n    \n    return roi\n\n# Function to extract ROIs for a specific study, series, and instance\ndef extract_rois_from_dicom(study_id, series_id, instance_number, base_image_dir, window_size=50, target_size=(224, 224)):\n    # Filter the label coordinates for the specific study_id, series_id, and instance_number\n    coords_df = train_label_coords_df[(train_label_coords_df['study_id'] == study_id) &\n                                      (train_label_coords_df['series_id'] == series_id) &\n                                      (train_label_coords_df['instance_number'] == instance_number)]\n    \n    # Load and preprocess the DICOM image\n    dicom_file_path = f'{base_image_dir}/{study_id}/{series_id}/{instance_number}.dcm'\n    preprocessed_image, original_width, original_height = preprocess_dicom_image(dicom_file_path, target_size)\n\n    # List to store ROIs\n    rois = []\n    \n    # Extract ROIs based on x, y coordinates from the CSV\n    for index, row in coords_df.iterrows():\n        x, y = row['x'], row['y']\n        \n        # Adjust the coordinates based on the resized image dimensions\n        x_rescaled, y_rescaled = adjust_coordinates(x, y, original_width, original_height, target_size[0], target_size[1])\n        \n        # Extract ROI with the adjusted coordinates\n        roi = extract_roi(preprocessed_image, x_rescaled, y_rescaled, window_size)\n        rois.append((roi, row['condition']))  # Append the ROI and condition for reference\n    \n    return rois\n\n# Example Usage\nstudy_id = 100206310\nseries_id = 1012284084\ninstance_number = 20\nbase_image_dir = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images'\n\n# Extract ROIs with adjusted coordinates\nrois = extract_rois_from_dicom(study_id, series_id, instance_number, base_image_dir, window_size=50, target_size=(224, 224))\n\n# Display the first ROI and its condition\nfor roi, condition in rois:\n    plt.figure(figsize=(4, 4))\n    plt.imshow(roi, cmap=plt.cm.bone)\n    plt.title(f'Extracted ROI - {condition}')\n    plt.axis('off')\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-09-23T15:34:57.846010Z","iopub.execute_input":"2024-09-23T15:34:57.846476Z","iopub.status.idle":"2024-09-23T15:34:58.313113Z","shell.execute_reply.started":"2024-09-23T15:34:57.846430Z","shell.execute_reply":"2024-09-23T15:34:58.311843Z"},"trusted":true},"execution_count":null,"outputs":[]}]}