{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# About Dataset","metadata":{"execution":{"iopub.status.busy":"2024-06-20T19:24:04.040776Z","iopub.execute_input":"2024-06-20T19:24:04.041214Z"}}},{"cell_type":"markdown","source":"1. Understand the Problem Statement\nObjective: Develop a model that classifies five lumbar spine degenerative conditions (Left Neural Foraminal Narrowing, Right Neural Foraminal Narrowing, Left Subarticular Stenosis, Right Subarticular Stenosis, Spinal Canal Stenosis) across different intervertebral disc levels.\nEvaluation: The model's predictions are evaluated using a weighted log loss metric, with different weights assigned based on the severity of the conditions.\n2. Data Overview\ntrain.csv: Contains labels for the training set, detailing the severity of conditions at various disc levels.\ntrain_label_coordinates.csv: Provides the coordinates of the areas of interest in the images.\nsample_submission.csv: Shows the format for submitting predictions.\ntrain/test_images: Folders containing the MRI scans in DICOM format.\ntrain/test_series_descriptions.csv: Describes the scan orientation.","metadata":{}},{"cell_type":"markdown","source":"# Import Libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport pydicom\nimport numpy as np\nimport pandas as pd\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom PIL import Image\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T15:27:11.685788Z","iopub.execute_input":"2024-06-21T15:27:11.686469Z","iopub.status.idle":"2024-06-21T15:27:11.692452Z","shell.execute_reply.started":"2024-06-21T15:27:11.686436Z","shell.execute_reply":"2024-06-21T15:27:11.691535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize_image(image, size=(128, 128)):\n    img = Image.fromarray(image).convert('L')  # Convert to grayscale\n    return np.array(img.resize(size, Image.ANTIALIAS))\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T15:27:12.844259Z","iopub.execute_input":"2024-06-21T15:27:12.844626Z","iopub.status.idle":"2024-06-21T15:27:12.849666Z","shell.execute_reply.started":"2024-06-21T15:27:12.844590Z","shell.execute_reply":"2024-06-21T15:27:12.848700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_dicom_images(base_path, series_df, batch_size=32, image_size=(128, 128)):\n    images = []\n    labels = []\n    count = 0\n    for _, row in tqdm(series_df.iterrows(), total=series_df.shape[0]):\n        study_id = row['study_id']\n        series_id = row['series_id']\n        instance_number = row['instance_number']\n        condition = row['condition']\n        level = row['level']\n        \n        file_path = os.path.join(base_path, str(study_id), str(series_id), f'{instance_number}.dcm')\n        if os.path.exists(file_path):\n            ds = pydicom.dcmread(file_path)\n            resized_image = resize_image(ds.pixel_array, size=image_size)\n            images.append(resized_image)\n            labels.append(f\"{condition}_{level}\")\n            count += 1\n        \n        if count == batch_size:\n            images = np.array(images) / 255.0\n            images = np.expand_dims(images, axis=-1)  # Add channel dimension\n            yield images, labels\n            images = []\n            labels = []\n            count = 0\n\n    if count > 0:\n        images = np.array(images) / 255.0\n        images = np.expand_dims(images, axis=-1)  # Add channel dimension\n        yield images, labels\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T15:27:13.675828Z","iopub.execute_input":"2024-06-21T15:27:13.676332Z","iopub.status.idle":"2024-06-21T15:27:13.685902Z","shell.execute_reply.started":"2024-06-21T15:27:13.676300Z","shell.execute_reply":"2024-06-21T15:27:13.684975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the path to the dataset\npath = '../input/rsna-2024-lumbar-spine-degenerative-classification/'\n\n# Read the CSV files\ndf_train_main = pd.read_csv(os.path.join(path, 'train.csv'))\ndf_train_label = pd.read_csv(os.path.join(path, 'train_label_coordinates.csv'))\ndf_train_desc = pd.read_csv(os.path.join(path, 'train_series_descriptions.csv'))\ndf_test_desc = pd.read_csv(os.path.join(path, 'test_series_descriptions.csv'))\ndf_sub = pd.read_csv(os.path.join(path, 'sample_submission.csv'))\n\n# Display the first few rows of each dataframe to ensure they loaded correctly\nprint(df_train_main.head())\nprint(df_train_label.head())\nprint(df_train_desc.head())\nprint(df_test_desc.head())\nprint(df_sub.head())\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T15:27:14.941630Z","iopub.execute_input":"2024-06-21T15:27:14.942473Z","iopub.status.idle":"2024-06-21T15:27:15.127732Z","shell.execute_reply.started":"2024-06-21T15:27:14.942440Z","shell.execute_reply":"2024-06-21T15:27:15.126837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define paths and load train images in batches\nbase_path_train = os.path.join(path, 'train_images')\nbatch_size = 32\nimage_size = (128, 128)\nimage_generator = load_dicom_images(base_path_train, df_train_label, batch_size=batch_size, image_size=image_size)\n\nimages_all = []\nlabels_all = []\n\nfor images_batch, labels_batch in image_generator:\n    images_all.append(images_batch)\n    labels_all.append(labels_batch)\n\nimages = np.concatenate(images_all, axis=0)\nlabels = np.concatenate(labels_all, axis=0)\n\nprint(f\"Loaded {images.shape[0]} images.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T15:27:19.271627Z","iopub.execute_input":"2024-06-21T15:27:19.272444Z","iopub.status.idle":"2024-06-21T15:44:17.357504Z","shell.execute_reply.started":"2024-06-21T15:27:19.272402Z","shell.execute_reply":"2024-06-21T15:44:17.356459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encode labels\nlabel_encoder = LabelEncoder()\nencoded_labels = label_encoder.fit_transform(labels)\n\n# Split data into training and validation sets\nX_train, X_val, y_train, y_val = train_test_split(images, encoded_labels, test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T15:44:27.366638Z","iopub.execute_input":"2024-06-21T15:44:27.367001Z","iopub.status.idle":"2024-06-21T15:44:29.347874Z","shell.execute_reply.started":"2024-06-21T15:44:27.366971Z","shell.execute_reply":"2024-06-21T15:44:29.347050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the CNN model\nmodel = Sequential([\n    Conv2D(32, (3, 3), activation='relu', input_shape=(image_size[0], image_size[1], 1)),\n    MaxPooling2D((2, 2)),\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Flatten(),\n    Dense(128, activation='relu'),\n    Dropout(0.5),\n    Dense(len(np.unique(encoded_labels)), activation='softmax')  \n])\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\n# Train the model\nhistory = model.fit(X_train, y_train, epochs=5, validation_data=(X_val, y_val), batch_size=32)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T15:44:34.621012Z","iopub.execute_input":"2024-06-21T15:44:34.621741Z","iopub.status.idle":"2024-06-21T15:46:20.756055Z","shell.execute_reply.started":"2024-06-21T15:44:34.621710Z","shell.execute_reply":"2024-06-21T15:46:20.755198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot training history\nplt.plot(history.history['accuracy'], label='accuracy')\nplt.plot(history.history['val_accuracy'], label='val_accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T15:46:54.142120Z","iopub.execute_input":"2024-06-21T15:46:54.143031Z","iopub.status.idle":"2024-06-21T15:46:54.673011Z","shell.execute_reply.started":"2024-06-21T15:46:54.142991Z","shell.execute_reply":"2024-06-21T15:46:54.672099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_label","metadata":{"execution":{"iopub.status.busy":"2024-06-21T15:47:03.901285Z","iopub.execute_input":"2024-06-21T15:47:03.901633Z","iopub.status.idle":"2024-06-21T15:47:03.920506Z","shell.execute_reply.started":"2024-06-21T15:47:03.901607Z","shell.execute_reply":"2024-06-21T15:47:03.919520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_label.info()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T15:47:04.188929Z","iopub.execute_input":"2024-06-21T15:47:04.189823Z","iopub.status.idle":"2024-06-21T15:47:04.225440Z","shell.execute_reply.started":"2024-06-21T15:47:04.189790Z","shell.execute_reply":"2024-06-21T15:47:04.224482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display rows where condition column has float values\nprint(df_train_label[df_train_label['condition'].apply(lambda x: isinstance(x, float))])\n\n# Filter out rows where 'condition' column contains float values\ndf_train_label = df_train_label[df_train_label['condition'].apply(lambda x: isinstance(x, str))]\n\n# Display the DataFrame to ensure the float values are removed\nprint(df_train_label.info())\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T15:47:05.214218Z","iopub.execute_input":"2024-06-21T15:47:05.214574Z","iopub.status.idle":"2024-06-21T15:47:05.270952Z","shell.execute_reply.started":"2024-06-21T15:47:05.214545Z","shell.execute_reply":"2024-06-21T15:47:05.270105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check unique values in the 'condition' column\nunique_conditions = df_train_label['condition'].unique()\n\n# Print unique values\nprint(unique_conditions)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T15:47:06.171617Z","iopub.execute_input":"2024-06-21T15:47:06.172602Z","iopub.status.idle":"2024-06-21T15:47:06.181724Z","shell.execute_reply.started":"2024-06-21T15:47:06.172558Z","shell.execute_reply":"2024-06-21T15:47:06.180785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# Display unique values in the 'condition' column before encoding\nprint(\"Unique values in 'condition' column before encoding:\")\nprint(df_train_label['condition'].unique())\n\n# Initialize LabelEncoder\nlabel_encoder = LabelEncoder()\n\n# Fit label encoder and transform 'condition' column\ndf_train_label['condition_encoded'] = label_encoder.fit_transform(df_train_label['condition'])\n\n# Display unique values in the encoded 'condition' column\nprint(\"\\nUnique encoded values in 'condition' column:\")\nprint(df_train_label['condition_encoded'].unique())\n\n# Optionally, you can create a mapping of original labels to encoded values\nlabel_mapping = dict(zip(label_encoder.classes_, label_encoder.transform(label_encoder.classes_)))\nprint(\"\\nLabel mapping:\")\nprint(label_mapping)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T15:47:06.839033Z","iopub.execute_input":"2024-06-21T15:47:06.839890Z","iopub.status.idle":"2024-06-21T15:47:06.866292Z","shell.execute_reply.started":"2024-06-21T15:47:06.839856Z","shell.execute_reply":"2024-06-21T15:47:06.865425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_label","metadata":{"execution":{"iopub.status.busy":"2024-06-21T15:47:07.243175Z","iopub.execute_input":"2024-06-21T15:47:07.243535Z","iopub.status.idle":"2024-06-21T15:47:07.258961Z","shell.execute_reply.started":"2024-06-21T15:47:07.243506Z","shell.execute_reply":"2024-06-21T15:47:07.257979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop the original 'condition' column\ndf_train_label.drop(columns=['condition'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T15:47:08.253091Z","iopub.execute_input":"2024-06-21T15:47:08.254061Z","iopub.status.idle":"2024-06-21T15:47:08.261342Z","shell.execute_reply.started":"2024-06-21T15:47:08.254027Z","shell.execute_reply":"2024-06-21T15:47:08.260546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_label","metadata":{"execution":{"iopub.status.busy":"2024-06-21T15:47:09.353978Z","iopub.execute_input":"2024-06-21T15:47:09.354613Z","iopub.status.idle":"2024-06-21T15:47:09.368024Z","shell.execute_reply.started":"2024-06-21T15:47:09.354584Z","shell.execute_reply":"2024-06-21T15:47:09.367108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check unique values in the 'condition' column\nunique_conditions = df_train_label['level'].unique()\n\n# Print unique values\nprint(unique_conditions)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T15:47:13.037836Z","iopub.execute_input":"2024-06-21T15:47:13.038493Z","iopub.status.idle":"2024-06-21T15:47:13.046726Z","shell.execute_reply.started":"2024-06-21T15:47:13.038460Z","shell.execute_reply":"2024-06-21T15:47:13.045792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n\n# Initialize LabelEncoder\nlabel_encoder = LabelEncoder()\n\n# Fit label encoder and transform 'condition' column\ndf_train_label['level_encoded'] = label_encoder.fit_transform(df_train_label['level'])\n\nprint(df_train_label['level_encoded'].unique())\n\n# Optionally, you can create a mapping of original labels to encoded values\nlabel_mapping = dict(zip(label_encoder.classes_, label_encoder.transform(label_encoder.classes_)))\nprint(\"\\nLabel mapping:\")\nprint(label_mapping)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T15:47:13.547724Z","iopub.execute_input":"2024-06-21T15:47:13.548288Z","iopub.status.idle":"2024-06-21T15:47:13.566829Z","shell.execute_reply.started":"2024-06-21T15:47:13.548261Z","shell.execute_reply":"2024-06-21T15:47:13.565948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_label","metadata":{"execution":{"iopub.status.busy":"2024-06-21T15:47:14.527224Z","iopub.execute_input":"2024-06-21T15:47:14.527956Z","iopub.status.idle":"2024-06-21T15:47:14.542242Z","shell.execute_reply.started":"2024-06-21T15:47:14.527923Z","shell.execute_reply":"2024-06-21T15:47:14.541343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop the original 'condition' column\ndf_train_label.drop(columns=['level'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T15:47:15.593896Z","iopub.execute_input":"2024-06-21T15:47:15.594735Z","iopub.status.idle":"2024-06-21T15:47:15.600662Z","shell.execute_reply.started":"2024-06-21T15:47:15.594698Z","shell.execute_reply":"2024-06-21T15:47:15.599760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_label","metadata":{"execution":{"iopub.status.busy":"2024-06-21T15:47:17.301653Z","iopub.execute_input":"2024-06-21T15:47:17.302246Z","iopub.status.idle":"2024-06-21T15:47:17.316062Z","shell.execute_reply.started":"2024-06-21T15:47:17.302209Z","shell.execute_reply":"2024-06-21T15:47:17.315070Z"},"trusted":true},"execution_count":null,"outputs":[]}]}