{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":30748,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install pydicom\n!pip install opencv-python\n!pip install tensorflow  # or pytorch, depending on your preference\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-08-09T12:24:38.659462Z","iopub.execute_input":"2024-08-09T12:24:38.660035Z","iopub.status.idle":"2024-08-09T12:24:48.927686Z","shell.execute_reply.started":"2024-08-09T12:24:38.660002Z","shell.execute_reply":"2024-08-09T12:24:48.926800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Step 2: Data Loading and Preprocessing","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# Load the CSV files\ntrain_df = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train.csv')\ncoordinates_df = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv')\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-08-09T12:24:48.929805Z","iopub.execute_input":"2024-08-09T12:24:48.930069Z","iopub.status.idle":"2024-08-09T12:24:49.004158Z","shell.execute_reply.started":"2024-08-09T12:24:48.930040Z","shell.execute_reply":"2024-08-09T12:24:49.003492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preview the data\ntrain_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-08-09T12:24:49.005001Z","iopub.execute_input":"2024-08-09T12:24:49.005262Z","iopub.status.idle":"2024-08-09T12:24:49.022814Z","shell.execute_reply.started":"2024-08-09T12:24:49.005236Z","shell.execute_reply":"2024-08-09T12:24:49.021940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coordinates_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-09T12:24:49.023990Z","iopub.execute_input":"2024-08-09T12:24:49.024244Z","iopub.status.idle":"2024-08-09T12:24:49.039198Z","shell.execute_reply.started":"2024-08-09T12:24:49.024220Z","shell.execute_reply":"2024-08-09T12:24:49.038295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom\nimport cv2\nimport numpy as np\n\ndef load_dicom_image(path):\n    dicom = pydicom.read_file(path)\n    img = dicom.pixel_array\n    img = cv2.resize(img, (256, 256))  # Resize to a standard size\n    img = img / np.max(img)  # Normalize pixel values\n    return img\n\n# Example of loading an image\nsample_image = load_dicom_image('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images/100206310/1012284084/1.dcm')\nprint(sample_image.shape)\n","metadata":{"execution":{"iopub.status.busy":"2024-08-09T12:24:49.041204Z","iopub.execute_input":"2024-08-09T12:24:49.041440Z","iopub.status.idle":"2024-08-09T12:24:50.487470Z","shell.execute_reply.started":"2024-08-09T12:24:49.041416Z","shell.execute_reply":"2024-08-09T12:24:50.486499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Step 3: Data Augmentation","metadata":{}},{"cell_type":"code","source":"import pydicom\nimport numpy as np\nimport cv2\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport matplotlib.pyplot as plt\n\n# Load the DICOM file\ndicom_path = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images/100206310/1012284084/1.dcm'\ndicom_image = pydicom.dcmread(dicom_path)\n\n# Convert the DICOM image to a Numpy array\nimage_array = dicom_image.pixel_array\n\n# Normalize the image (Optional, depending on your model requirements)\nimage_array = cv2.normalize(image_array, None, 0, 255, cv2.NORM_MINMAX)\n\n# Convert the image to 3 channels if required (e.g., grayscale to RGB)\nif len(image_array.shape) == 2:  # Grayscale\n    image_array = cv2.cvtColor(image_array, cv2.COLOR_GRAY2RGB)\n\n# Expand dimensions if needed (if your model expects 3D inputs like (1, height, width, channels))\nimage_array = np.expand_dims(image_array, axis=0)\n","metadata":{"execution":{"iopub.status.busy":"2024-08-09T12:24:50.488467Z","iopub.execute_input":"2024-08-09T12:24:50.488770Z","iopub.status.idle":"2024-08-09T12:25:05.938168Z","shell.execute_reply.started":"2024-08-09T12:24:50.488737Z","shell.execute_reply":"2024-08-09T12:25:05.936909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the data augmentation generator\ndatagen = ImageDataGenerator(\n    rotation_range=20,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    shear_range=0.1,\n    zoom_range=0.1,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\n# Apply augmentation\naugmented_image = datagen.random_transform(image_array[0])\n\n# Display the augmented image\nplt.imshow(augmented_image.astype('uint8'))\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-08-09T12:25:05.939522Z","iopub.execute_input":"2024-08-09T12:25:05.940066Z","iopub.status.idle":"2024-08-09T12:25:06.321175Z","shell.execute_reply.started":"2024-08-09T12:25:05.940029Z","shell.execute_reply":"2024-08-09T12:25:06.320054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Step 4: Model Development","metadata":{}},{"cell_type":"markdown","source":"**Model Architecture**\nDefine a convolutional neural network (CNN) or a more advanced architecture like EfficientNet.","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers, models\n\ndef build_model(input_shape):\n    model = models.Sequential()\n\n    model.add(layers.Conv2D(32, (3, 3), activation='relu', input_shape=input_shape))\n    model.add(layers.MaxPooling2D((2, 2)))\n\n    model.add(layers.Conv2D(64, (3, 3), activation='relu'))\n    model.add(layers.MaxPooling2D((2, 2)))\n\n    model.add(layers.Conv2D(128, (3, 3), activation='relu'))\n    model.add(layers.MaxPooling2D((2, 2)))\n\n    model.add(layers.Flatten())\n    model.add(layers.Dense(256, activation='relu'))\n    model.add(layers.Dropout(0.5))\n\n    model.add(layers.Dense(3, activation='softmax'))  # Three classes: Normal/Mild, Moderate, Severe\n\n    model.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n    \n    return model\n\nmodel = build_model((256, 256, 1))\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2024-08-09T12:25:06.322313Z","iopub.execute_input":"2024-08-09T12:25:06.322752Z","iopub.status.idle":"2024-08-09T12:25:10.539937Z","shell.execute_reply.started":"2024-08-09T12:25:06.322722Z","shell.execute_reply":"2024-08-09T12:25:10.538934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Step 5: Training the Model\nTrain the model using the labeled data.","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom tensorflow.keras.utils import to_categorical\n\n# Prepare data\nX = []  # List of images\ny = []  # List of labels\n\nfor idx, row in train_df.iterrows():\n    image_path = f\"/kaggle/input/rsna-2024-lumbar-spine/train_images/{row['study_id']}/{coordinates_df['series_id']}.dcm\"\n    image = load_dicom_image(image_path)\n    X.append(image)\n    \n    # Convert label to one-hot encoded format\n    label = [0, 0, 0]\n    if row['severity'] == 'Normal/Mild':\n        label[0] = 1\n    elif row['severity'] == 'Moderate':\n        label[1] = 1\n    elif row['severity'] == 'Severe':\n        label[2] = 1\n    y.append(label)\n\nX = np.array(X).reshape(-1, 256, 256, 1)\ny = np.array(y)\n\n# Train-test split\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Train the model\nhistory = model.fit(\n    datagen.flow(X_train, y_train, batch_size=32),\n    validation_data=(X_val, y_val),\n    epochs=10\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-08-09T12:25:10.541153Z","iopub.execute_input":"2024-08-09T12:25:10.541451Z","iopub.status.idle":"2024-08-09T12:25:12.359391Z","shell.execute_reply.started":"2024-08-09T12:25:10.541422Z","shell.execute_reply":"2024-08-09T12:25:12.358024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coordinates_df['study_id'].head()","metadata":{"execution":{"iopub.status.busy":"2024-08-09T12:25:12.360256Z","iopub.status.idle":"2024-08-09T12:25:12.360609Z","shell.execute_reply.started":"2024-08-09T12:25:12.360433Z","shell.execute_reply":"2024-08-09T12:25:12.360452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coordinates_df['series_id'].head()\n","metadata":{"execution":{"iopub.status.busy":"2024-08-09T12:25:12.361731Z","iopub.status.idle":"2024-08-09T12:25:12.362094Z","shell.execute_reply.started":"2024-08-09T12:25:12.361915Z","shell.execute_reply":"2024-08-09T12:25:12.361931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport pydicom\nimport cv2\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# Define the function to load and preprocess DICOM images\ndef load_dicom_image(dicom_path, img_size=(256, 256)):\n    dicom_image = pydicom.dcmread(dicom_path)\n    image_array = dicom_image.pixel_array\n    image_array = cv2.normalize(image_array, None, 0, 255, cv2.NORM_MINMAX)\n    \n    if len(image_array.shape) == 2:\n        image_array = cv2.cvtColor(image_array, cv2.COLOR_GRAY2RGB)\n    \n    image_array = cv2.resize(image_array, img_size)\n    return image_array\n\n# Assuming train_df is already loaded with columns 'study_id', 'series_id', 'severity'\nX = []  # List of images\ny = []  # List of labels\n\nfor idx, row in train_df.iterrows():\n    image_path = f\"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images/{coordinates_df['series_id']}/{coordinates_df['study_id']}.dcm\"\n    image = load_dicom_image(image_path)\n    X.append(image)\n    \n    # Convert severity to one-hot encoded format\n    label = [0, 0, 0]  # Assuming 3 classes: Normal/Mild, Moderate, Severe\n    if row['severity'] == 'Normal/Mild':\n        label[0] = 1\n    elif row['severity'] == 'Moderate':\n        label[1] = 1\n    elif row['severity'] == 'Severe':\n        label[2] = 1\n    y.append(label)\n\n# Convert lists to numpy arrays and reshape\nX = np.array(X).reshape(-1, 256, 256, 3)  # 3 channels for RGB\ny = np.array(y)\n\n# Split the dataset into training and validation sets\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Define an image data generator with augmentation\ndatagen = ImageDataGenerator(\n    rotation_range=20,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    shear_range=0.1,\n    zoom_range=0.1,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\n# Assuming 'model' is already defined and compiled\nhistory = model.fit(\n    datagen.flow(X_train, y_train, batch_size=32),\n    validation_data=(X_val, y_val),\n    epochs=10\n)\n\n# After training, you can plot the history or save the model as needed.\n","metadata":{"execution":{"iopub.status.busy":"2024-08-09T12:25:12.362984Z","iopub.status.idle":"2024-08-09T12:25:12.363290Z","shell.execute_reply.started":"2024-08-09T12:25:12.363146Z","shell.execute_reply":"2024-08-09T12:25:12.363161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n# Load the CSV files\ntrain_df = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train.csv')\ncoordinates_df = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv')\ndescription_df = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_series_descriptions.csv')\n","metadata":{"execution":{"iopub.status.busy":"2024-08-09T12:25:12.363980Z","iopub.status.idle":"2024-08-09T12:25:12.364266Z","shell.execute_reply.started":"2024-08-09T12:25:12.364128Z","shell.execute_reply":"2024-08-09T12:25:12.364142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-09T12:25:12.365358Z","iopub.status.idle":"2024-08-09T12:25:12.365639Z","shell.execute_reply.started":"2024-08-09T12:25:12.365503Z","shell.execute_reply":"2024-08-09T12:25:12.365517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coordinates_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-09T12:25:12.366425Z","iopub.status.idle":"2024-08-09T12:25:12.366694Z","shell.execute_reply.started":"2024-08-09T12:25:12.366560Z","shell.execute_reply":"2024-08-09T12:25:12.366574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"description_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-09T12:25:12.367378Z","iopub.status.idle":"2024-08-09T12:25:12.367653Z","shell.execute_reply.started":"2024-08-09T12:25:12.367519Z","shell.execute_reply":"2024-08-09T12:25:12.367534Z"},"trusted":true},"execution_count":null,"outputs":[]}]}