{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import os\nimport pydicom\nimport cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras import layers, models\nfrom sklearn.metrics import confusion_matrix, classification_report, roc_auc_score, roc_curve, precision_score, recall_score\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.layers import Dense, Flatten\nfrom tensorflow.keras.models import Model","metadata":{"execution":{"iopub.status.busy":"2024-11-18T11:35:02.752031Z","iopub.execute_input":"2024-11-18T11:35:02.752941Z","iopub.status.idle":"2024-11-18T11:35:24.299856Z","shell.execute_reply.started":"2024-11-18T11:35:02.752877Z","shell.execute_reply":"2024-11-18T11:35:24.298722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Read data","metadata":{}},{"cell_type":"code","source":"train_path = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/'\ntest_path = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/test_images/'\ntrain_images = train_path + \"train_images/\"\ntrain  = pd.read_csv(train_path + 'train.csv')\nlabel = pd.read_csv(train_path + 'train_label_coordinates.csv')\ntrain_desc  = pd.read_csv(train_path + 'train_series_descriptions.csv')\ntest_desc   = pd.read_csv(train_path + 'test_series_descriptions.csv')\nsub         = pd.read_csv(train_path + 'sample_submission.csv')\nstudy_id = 52695609","metadata":{"execution":{"iopub.status.busy":"2024-11-18T11:35:24.302080Z","iopub.execute_input":"2024-11-18T11:35:24.302796Z","iopub.status.idle":"2024-11-18T11:35:24.523126Z","shell.execute_reply.started":"2024-11-18T11:35:24.302749Z","shell.execute_reply":"2024-11-18T11:35:24.521855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Define a Function to Load and Preprocess DICOM Images","metadata":{}},{"cell_type":"code","source":"def load_dicom_image(file_path, img_size=(128, 128)):\n    # Read the DICOM file\n    dicom = pydicom.dcmread(file_path)\n    # dicom = pydicom.read_file(file_path)\n    # Convert the DICOM file to a NumPy array\n    img = dicom.pixel_array\n    # Normalize the image (to 0-1 range)\n    img = img / np.max(img)\n    # Resize the image\n    img_resized = cv2.resize(img, img_size)\n    # Convert to 3-channel image (optional, as some CNN architectures require 3 channels)\n    img_resized = np.stack((img_resized,)*3, axis=-1)\n    return img_resized","metadata":{"execution":{"iopub.status.busy":"2024-11-18T11:35:24.524872Z","iopub.execute_input":"2024-11-18T11:35:24.525397Z","iopub.status.idle":"2024-11-18T11:35:24.533783Z","shell.execute_reply.started":"2024-11-18T11:35:24.525339Z","shell.execute_reply":"2024-11-18T11:35:24.532382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Show images ","metadata":{}},{"cell_type":"code","source":"import math\n\nfrom mpl_toolkits.axes_grid1 import ImageGrid\n\nroot_dir = train_images + str(study_id)\npath_file_name = ''\nimg_list = []\n\nfor path, subdirs, files in os.walk(root_dir):\n    for name in files:\n        # print(os.path.join(path, name))\n        path_file_name = os.path.join(path, name)\n        img = load_dicom_image(path_file_name, img_size=(128, 128))\n        img_list.append(img)\n\n# calculate images to know cols and rows number\n\ntotal_images = len(img_list)\nprint('Total imamges = ', total_images)\ncols = 10\nnum_rows_ = math.ceil(total_images / cols)\n\nfig = plt.figure(figsize=(20., 20))  # adjust figure size as needed\ngrid = ImageGrid(fig, 111,  # similar to subplot(111)\n                 nrows_ncols=(num_rows_, cols),  # create a 2x2 grid of Axes\n                 axes_pad=0.1,  # pad between Axes in inch\n                 )\nfor ax, im in zip(grid, img_list):\n       ax.imshow(im)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-18T11:35:24.536597Z","iopub.execute_input":"2024-11-18T11:35:24.537064Z","iopub.status.idle":"2024-11-18T11:35:54.119287Z","shell.execute_reply.started":"2024-11-18T11:35:24.537008Z","shell.execute_reply":"2024-11-18T11:35:54.117651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Show images with some points found at coordinates dataset","metadata":{}},{"cell_type":"code","source":"def load_dicom_image_local_(study_id, series_id, instance_number):\n    image_path = os.path.join(train_images, str(study_id), str(series_id), f\"{instance_number}.dcm\")\n    if os.path.exists(image_path):\n        dicom = pydicom.dcmread(image_path)\n        img = dicom.pixel_array  # Extract the pixel data\n        return img\n    else:\n        print(f\"Image {image_path} not found.\")\n        return None\n\n# Function to plot the image and overlay the key points\ndef plot_image_with_points_(study_id, series_id, instance_number, points):\n    img = load_dicom_image_local_(study_id, series_id, instance_number)\n    \n    if img is not None:\n        plt.imshow(img, cmap='gray')\n        plt.scatter(points['x'], points['y'], c='red', s=40, label=\"Key Points\")\n        plt.legend()\n        plt.title(f\"Study ID: {study_id}, Series ID: {series_id}, Instance: {instance_number}\")\n        plt.axis('off')\n        plt.figure(figsize=(5., 5))\n        plt.show()\n    else:\n        print(f\"Could not load the image for Study ID: {study_id}, Series ID: {series_id}, Instance: {instance_number}\")\n\n# Sample visualization for the first few rows of the CSV\nfor index, row in label.iterrows():\n    study_id = row['study_id']\n    series_id = row['series_id']\n    instance_number = row['instance_number']\n    points = {'x': row['x'], 'y': row['y']}    \n    # Plot the image with points\n    plot_image_with_points_(study_id, series_id, instance_number, points)\n    \n    # Limit the number of visualizations for demonstration (optional)\n    if index >= 5:\n        break","metadata":{"execution":{"iopub.status.busy":"2024-11-18T11:35:54.121547Z","iopub.execute_input":"2024-11-18T11:35:54.122028Z","iopub.status.idle":"2024-11-18T11:35:55.826835Z","shell.execute_reply.started":"2024-11-18T11:35:54.121978Z","shell.execute_reply":"2024-11-18T11:35:55.825735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Data from Multiple Folders","metadata":{}},{"cell_type":"code","source":"def load_data_from_folders(root_dir, img_size=(128, 128)):\n    X = []  # Image data\n    y = []  # Labels\n    class_names = os.listdir(root_dir)\n    \n    for class_label, class_name in enumerate(class_names):\n        class_dir = os.path.join(root_dir, class_name)\n\n        if os.path.isdir(class_dir):\n            for file_name in os.listdir(class_dir):\n                file_path = os.path.join(class_dir, file_name)\n                print(file_path)\n                # print(file_name)\n                if file_name.endswith('.dcm'):\n                    try:\n                        # Load and preprocess the image\n                        img = load_dicom_image(file_path, img_size)\n                        print('image = ', img)\n                        X.append(img)\n                        y.append(class_label)\n                    except Exception as e:\n                        print(f\"Error loading {file_path}: {e}\")\n    \n    X = np.array(X)\n    y = np.array(y)\n    return X, y, class_names\n","metadata":{"execution":{"iopub.status.busy":"2024-11-18T11:35:55.828688Z","iopub.execute_input":"2024-11-18T11:35:55.829314Z","iopub.status.idle":"2024-11-18T11:35:55.841623Z","shell.execute_reply.started":"2024-11-18T11:35:55.829234Z","shell.execute_reply":"2024-11-18T11:35:55.840448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train Test Split","metadata":{}},{"cell_type":"code","source":"root_dir = train_images + str(study_id)\nX, y, class_names = load_data_from_folders(root_dir)\n\nn_samples = 100  \ntrain_size = 0.6  #  for training\ntest_size = 0.4  #  for testing\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=test_size, train_size=train_size, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-18T11:35:55.843535Z","iopub.execute_input":"2024-11-18T11:35:55.844571Z","iopub.status.idle":"2024-11-18T11:35:57.783814Z","shell.execute_reply.started":"2024-11-18T11:35:55.844508Z","shell.execute_reply":"2024-11-18T11:35:57.782345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"####  Define the CNN Model","metadata":{}},{"cell_type":"code","source":"def create_cnn_model(input_shape, num_classes):\n    model = models.Sequential([\n        layers.Conv2D(32, (3, 3), activation='relu', input_shape=input_shape),\n        layers.MaxPooling2D((2, 2)),\n        layers.Conv2D(64, (3, 3), activation='relu'),\n        layers.MaxPooling2D((2, 2)),\n        layers.Conv2D(128, (3, 3), activation='relu'),\n        layers.MaxPooling2D((2, 2)),\n        layers.Flatten(),\n        layers.Dense(128, activation='relu'),\n        layers.Dense(num_classes, activation='softmax')\n    ])\n    \n    model.compile(optimizer='adam',\n                  loss='sparse_categorical_crossentropy',\n                  metrics=['accuracy'])\n    return model","metadata":{"execution":{"iopub.status.busy":"2024-11-18T11:35:57.785395Z","iopub.execute_input":"2024-11-18T11:35:57.785909Z","iopub.status.idle":"2024-11-18T11:35:57.795377Z","shell.execute_reply.started":"2024-11-18T11:35:57.785865Z","shell.execute_reply":"2024-11-18T11:35:57.794088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create the CNN model","metadata":{}},{"cell_type":"code","source":"input_shape = (128, 128, 3)\nnum_classes = len(class_names)\nmodel = create_cnn_model(input_shape, num_classes)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-11-18T11:35:57.796900Z","iopub.execute_input":"2024-11-18T11:35:57.797293Z","iopub.status.idle":"2024-11-18T11:35:57.990648Z","shell.execute_reply.started":"2024-11-18T11:35:57.797251Z","shell.execute_reply":"2024-11-18T11:35:57.989509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train the model","metadata":{}},{"cell_type":"code","source":"early_stopping = EarlyStopping(monitor='val_loss', patience=3)\nhistory = model.fit(X_train, y_train, epochs=10, batch_size=32, validation_data=(X_test, y_test), callbacks=[early_stopping])","metadata":{"execution":{"iopub.status.busy":"2024-11-18T11:35:57.994830Z","iopub.execute_input":"2024-11-18T11:35:57.995290Z","iopub.status.idle":"2024-11-18T11:36:06.581944Z","shell.execute_reply.started":"2024-11-18T11:35:57.995246Z","shell.execute_reply":"2024-11-18T11:36:06.580621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Plot the Training Graphs","metadata":{}},{"cell_type":"code","source":"def plot_training_history(history):\n    acc = history.history['accuracy']\n    val_acc = history.history['val_accuracy']\n    loss = history.history['loss']\n    val_loss = history.history['val_loss']\n\n    epochs = range(1, len(acc) + 1)\n\n    # Plot accuracy\n    plt.figure(figsize=(12, 4))\n    \n    plt.subplot(1, 2, 1)\n    plt.plot(epochs, acc, 'b', label='Training Accuracy')\n    plt.plot(epochs, val_acc, 'r', label='Validation Accuracy')\n    plt.title('Training and Validation Accuracy')\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy')\n    plt.legend()\n\n    # Plot loss\n    plt.subplot(1, 2, 2)\n    plt.plot(epochs, loss, 'b', label='Training Loss')\n    plt.plot(epochs, val_loss, 'r', label='Validation Loss')\n    plt.title('Training and Validation Loss')\n    plt.xlabel('Epochs')\n    plt.ylabel('Loss')\n    plt.legend()\n\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-18T11:36:06.583582Z","iopub.execute_input":"2024-11-18T11:36:06.583980Z","iopub.status.idle":"2024-11-18T11:36:06.594629Z","shell.execute_reply.started":"2024-11-18T11:36:06.583939Z","shell.execute_reply":"2024-11-18T11:36:06.593386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluate the Model","metadata":{}},{"cell_type":"code","source":"test_loss, test_accuracy = model.evaluate(X_test, y_test)\nprint(f'Test accuracy: {test_accuracy:.4f}')","metadata":{"execution":{"iopub.status.busy":"2024-11-18T11:36:06.596155Z","iopub.execute_input":"2024-11-18T11:36:06.596617Z","iopub.status.idle":"2024-11-18T11:36:06.861722Z","shell.execute_reply.started":"2024-11-18T11:36:06.596559Z","shell.execute_reply":"2024-11-18T11:36:06.860559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predict","metadata":{}},{"cell_type":"code","source":"sub_study_id = '44036939/3844393089/'\ndef preprocess_dicom_image(file_path, img_size=(128, 128)):\n    img = load_dicom_image(file_path, img_size)\n    img = np.expand_dims(img, axis=0)  # Add batch dimension\n    return img\n# Load test DICOM images and make predictions\ntest_dir = test_path + sub_study_id #  Path to the test folder\ntest_files = [f for f in os.listdir(test_dir) if f.endswith('.dcm')]\npredictions = []\n\nfor test_file in test_files:\n    file_path = os.path.join(test_dir, test_file)\n    img = preprocess_dicom_image(file_path)\n    pred = model.predict(img)\n    predicted_class = np.argmax(pred, axis=1)\n    predictions.append((test_file, predicted_class[0]))","metadata":{"execution":{"iopub.status.busy":"2024-11-18T11:36:06.863596Z","iopub.execute_input":"2024-11-18T11:36:06.864083Z","iopub.status.idle":"2024-11-18T11:36:10.985061Z","shell.execute_reply.started":"2024-11-18T11:36:06.864029Z","shell.execute_reply":"2024-11-18T11:36:10.983841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"predictions","metadata":{"execution":{"iopub.status.busy":"2024-11-18T11:36:10.987077Z","iopub.execute_input":"2024-11-18T11:36:10.987615Z","iopub.status.idle":"2024-11-18T11:36:11.000674Z","shell.execute_reply.started":"2024-11-18T11:36:10.987556Z","shell.execute_reply":"2024-11-18T11:36:10.999239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_labels = [class_names[p[1]] for p in predictions]\n#  Creating a Submission File\nresult = pd.DataFrame({\n    'image_id': [p[0] for p in predictions],  # The test image file names\n    'label': predicted_labels                 # The predicted class labels\n})\nresult_file = 'result.csv'\nresult.to_csv('/kaggle/working/' + result_file, index=False)\n\nprint(f'Submission file saved to {result_file}')","metadata":{"execution":{"iopub.status.busy":"2024-11-18T11:36:11.002253Z","iopub.execute_input":"2024-11-18T11:36:11.002788Z","iopub.status.idle":"2024-11-18T11:36:11.016927Z","shell.execute_reply.started":"2024-11-18T11:36:11.002726Z","shell.execute_reply":"2024-11-18T11:36:11.015670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result.head() #","metadata":{"execution":{"iopub.status.busy":"2024-11-18T11:41:34.020300Z","iopub.execute_input":"2024-11-18T11:41:34.021499Z","iopub.status.idle":"2024-11-18T11:41:34.039435Z","shell.execute_reply.started":"2024-11-18T11:41:34.021416Z","shell.execute_reply":"2024-11-18T11:41:34.037913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_training_history(history)","metadata":{"execution":{"iopub.status.busy":"2024-11-18T11:46:15.626658Z","iopub.execute_input":"2024-11-18T11:46:15.627948Z","iopub.status.idle":"2024-11-18T11:46:16.216913Z","shell.execute_reply.started":"2024-11-18T11:46:15.627894Z","shell.execute_reply":"2024-11-18T11:46:16.215552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}