{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"},{"sourceId":34027,"sourceType":"modelInstanceVersion","modelInstanceId":28484,"modelId":40334}],"dockerImageVersionId":30746,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Data Loading and Preprocessing","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom skimage.transform import resize\nimport pydicom\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nfrom tensorflow.keras.applications import ResNet101, EfficientNetB3, ResNet50V2\nfrom tensorflow.keras.layers import GlobalAveragePooling2D, Dense, Dropout\nfrom tensorflow.keras.models import Model","metadata":{"execution":{"iopub.status.busy":"2024-07-25T00:06:37.640616Z","iopub.execute_input":"2024-07-25T00:06:37.640991Z","iopub.status.idle":"2024-07-25T00:06:37.647018Z","shell.execute_reply.started":"2024-07-25T00:06:37.640960Z","shell.execute_reply":"2024-07-25T00:06:37.645950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the data\ndf_train = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train.csv')\ndf_label_coordinates = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv')\ndf_series_descriptions = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_series_descriptions.csv')\n\n# Copy the dataframes to new variables and remove unwanted columns\ndf_train_copy = df_train.copy()\ndf_label_copy = df_label_coordinates.copy()\ndf_series_copy = df_series_descriptions.copy()\n\n# Identify and handle missing data\nprint(\"Missing values in df_train_copy:\")\nprint(df_train_copy.isnull().sum())\n\n# Fill missing values with mean for numeric columns only\ndf_train_copy = df_train_copy.apply(lambda x: x.fillna(x.mean()) if x.dtype.kind in 'biufc' else x)\n\n# Define the severity map\nseverity_map = {'Normal/Mild': 0, 'Moderate': 1, 'Severe': 2}\n\n# Apply the severity map to each relevant column\nseverity_columns = [\n    'spinal_canal_stenosis_l1_l2', 'spinal_canal_stenosis_l2_l3',\n    'spinal_canal_stenosis_l3_l4', 'spinal_canal_stenosis_l4_l5',\n    'spinal_canal_stenosis_l5_s1', 'left_neural_foraminal_narrowing_l1_l2',\n    'left_neural_foraminal_narrowing_l2_l3', 'left_neural_foraminal_narrowing_l3_l4',\n    'left_neural_foraminal_narrowing_l4_l5', 'left_neural_foraminal_narrowing_l5_s1',\n    'right_neural_foraminal_narrowing_l1_l2', 'right_neural_foraminal_narrowing_l2_l3',\n    'right_neural_foraminal_narrowing_l3_l4', 'right_neural_foraminal_narrowing_l4_l5',\n    'right_neural_foraminal_narrowing_l5_s1', 'left_subarticular_stenosis_l1_l2',\n    'left_subarticular_stenosis_l2_l3', 'left_subarticular_stenosis_l3_l4',\n    'left_subarticular_stenosis_l4_l5', 'left_subarticular_stenosis_l5_s1',\n    'right_subarticular_stenosis_l1_l2', 'right_subarticular_stenosis_l2_l3',\n    'right_subarticular_stenosis_l3_l4', 'right_subarticular_stenosis_l4_l5',\n    'right_subarticular_stenosis_l5_s1'\n]\n\nfor column in severity_columns:\n    df_train_copy[column] = df_train_copy[column].map(severity_map)\n\n# Define the image preprocessing function\ndef preprocess_images(row):\n    processed_image = np.random.rand(224, 224, 3)  # Dummy image array\n    print(f'Processing row: {row[\"study_id\"]}')\n    return processed_image\n\n# Apply preprocessing function to each row with a progress bar\ndf_train_copy['image'] = [preprocess_images(row) for _, row in tqdm(df_train_copy.iterrows(), total=df_train_copy.shape[0], mininterval=5)]\n\n# Create a new severity column based on the maximum severity value in each row\ndf_train_copy['severity'] = df_train_copy[severity_columns].max(axis=1)\n\n# Check the results\nprint(\"Unique values in 'severity' column after update:\", df_train_copy['severity'].unique())\nprint(\"Sample data for 'severity' and other columns after update:\")\nprint(df_train_copy[['severity'] + severity_columns].sample(10))\nprint(\"Data prepared successfully\")\n\n# Train-test split\ntrain_data, val_data = train_test_split(df_train_copy, test_size=0.2, random_state=42)\n\n# Prepare the training and validation data\ntrain_images = np.stack(train_data['image'].values)\nval_images = np.stack(val_data['image'].values)\n\ntrain_labels = tf.keras.utils.to_categorical(train_data['severity'], num_classes=3)\nval_labels = tf.keras.utils.to_categorical(val_data['severity'], num_classes=3)","metadata":{"_kg_hide-output":false,"_kg_hide-input":true,"scrolled":true,"execution":{"iopub.status.busy":"2024-07-24T23:53:37.621095Z","iopub.execute_input":"2024-07-24T23:53:37.621754Z","iopub.status.idle":"2024-07-24T23:53:42.174374Z","shell.execute_reply.started":"2024-07-24T23:53:37.621722Z","shell.execute_reply":"2024-07-24T23:53:42.173570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Definition and Training","metadata":{}},{"cell_type":"code","source":"# Load models pre-trained on ImageNet without top layer and use different models based on a parameter\ndef get_model(model_name):\n    if model_name == \"EfficientNetB3\":\n        base_model = EfficientNetB3(weights=None, include_top=False, input_shape=(224, 224, 3))\n        weights_path = '/kaggle/input/efficientnetb3_notop.h5/keras/efficientnetb3/1/efficientnetb3_notop.h5'\n        base_model.load_weights(weights_path)\n    else:\n        raise ValueError(\"Invalid model name\")\n\n    x = base_model.output\n    x = GlobalAveragePooling2D()(x)\n    x = Dense(128, activation='relu')(x)\n    x = Dropout(0.5)(x)\n    predictions = Dense(3, activation='softmax')(x)\n\n    model = Model(inputs=base_model.input, outputs=predictions)\n    return model\n\n# Get the model\nmodel_name = \"EfficientNetB3\"  # Change to \"ResNet50V2\" or \"ResNet101\" as needed\nmodel = get_model(model_name)\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Train the model\nhistory = model.fit(train_images, train_labels, epochs=10, validation_data=(val_images, val_labels))\n\n# Evaluate the model on validation data\nloss, accuracy = model.evaluate(val_images, val_labels)\nprint(f\"Validation Loss: {loss:.2f}\")\nprint(f\"Validation Accuracy: {accuracy:.2f}\")","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-07-24T23:53:47.069863Z","iopub.execute_input":"2024-07-24T23:53:47.070677Z","iopub.status.idle":"2024-07-25T00:00:22.178968Z","shell.execute_reply.started":"2024-07-24T23:53:47.070642Z","shell.execute_reply":"2024-07-25T00:00:22.178059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test Data Processing and Prediction & Submission File Preparation","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport pydicom\nfrom skimage.transform import resize\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\n\n# Load test data\ntest_series_descriptions_df = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/test_series_descriptions.csv')\n\n# Define the directory containing the test images\ntest_images_dir = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/test_images'\n\n# Function to get the image path\ndef get_image_path(base_dir, study_id, series_id, instance_number):\n    return f\"{base_dir}/{study_id}/{series_id}/{instance_number}.dcm\"\n\n# Function to load and preprocess a DICOM image\ndef load_dicom_image(path):\n    dicom = pydicom.dcmread(path)\n    image = dicom.pixel_array\n    image = resize(image, (224, 224), mode='constant')\n    image = np.stack((image,) * 3, axis=-1)  # Convert to 3 channels\n    return image\n\n# Preprocess test images\ndef preprocess_test_images(row):\n    images = []\n    instance_number = 1  # Start with the first instance\n    while True:\n        try:\n            image_path = get_image_path(test_images_dir, row['study_id'], row['series_id'], instance_number)\n            image = load_dicom_image(image_path)\n            images.append(image)\n            instance_number += 1\n        except FileNotFoundError:\n            break  # No more images in the series\n    return np.mean(images, axis=0)  # Average the images if multiple\n\n# Apply the preprocessing function to test data\ntqdm.pandas()  # Enables progress bar for apply\ntest_series_descriptions_df['image'] = test_series_descriptions_df.progress_apply(preprocess_test_images, axis=1)\n\n# Convert images to numpy array\ntest_images = np.stack(test_series_descriptions_df['image'].values)\n\n# Assuming you have a trained model loaded as `model`\n# Predict using the trained model\npredictions = model.predict(test_images)\n\n# Prepare submission file\nsubmission_df = pd.DataFrame({\n    'series_id': test_series_descriptions_df['series_id'],\n    'normal/mild': predictions[:, 0],\n    'moderate': predictions[:, 1],\n    'severe': predictions[:, 2]\n})\n\n# Save submission\nsubmission_df.to_csv('/kaggle/working/submission.csv', index=False)\n\nprint(\"Submission file created successfully\")\n\n# Assuming you have object detection results stored in detected_objects\n# Example format: detected_objects = [[(x1, y1), (x2, y2)], [(x1, y1)], [(x1, y1), (x2, y2), (x3, y3)], ...]\n\ndetected_objects = [\n    [(50, 50), (100, 100)],  # Example coordinates for the first image\n    [(60, 60)],              # Example coordinates for the second image\n    [(70, 70), (120, 120), (150, 150)]  # Example coordinates for the third image\n]\n\n# Function to show DICOM image with prediction and object detection\ndef show_dicom_image_with_prediction(image, prediction, detected_coords):\n    plt.imshow(image[:, :, 0], cmap=plt.cm.gray)  # Show the first channel in grayscale\n    plt.title(f\"Prediction: Normal/Mild: {prediction[0]:.2f}, Moderate: {prediction[1]:.2f}, Severe: {prediction[2]:.2f}\")\n    for (x, y) in detected_coords:\n        plt.scatter(x, y, c='red', s=40, marker='o')  # Draw a red dot\n    plt.axis('off')\n    plt.show()\n\n# Show a few test images with predictions and object detections\nfor i in range(3):  # Show 3 test images\n    show_dicom_image_with_prediction(test_images[i], predictions[i], detected_objects[i])","metadata":{"execution":{"iopub.status.busy":"2024-07-25T00:13:21.202464Z","iopub.execute_input":"2024-07-25T00:13:21.202831Z","iopub.status.idle":"2024-07-25T00:13:24.586972Z","shell.execute_reply.started":"2024-07-25T00:13:21.202805Z","shell.execute_reply":"2024-07-25T00:13:24.586005Z"},"trusted":true},"execution_count":null,"outputs":[]}]}