{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Now define a the input directoy path:\nstart_dir_path=\"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-15T23:24:36.466319Z","iopub.execute_input":"2024-06-15T23:24:36.466710Z","iopub.status.idle":"2024-06-15T23:24:36.504096Z","shell.execute_reply.started":"2024-06-15T23:24:36.466678Z","shell.execute_reply":"2024-06-15T23:24:36.502760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#First import relevant libraries:\n\nimport pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt #For ploting and working with graphs\nimport seaborn as sns\nimport cv2 #Computer vision version 2 library for loading and reading images\nimport pydicom #for working wiht dicom images (MRI images are dicom images with .dcm extension)\nimport os # Already imported above. This Operating system library is to work with computer kernel.usefyl to interact with computer kernel and run commands like terminal. \nimport glob #glob library helps finding global patterns in file names. Not sure how it will be useful !\nfrom tqdm import tqdm #Taqadum is arabic word meanng progress and hence tqdm is ibrary which help in showing progress bars.\nimport warnings #lbraary helps in rasing warning messages where needed. \nimport ipywidgets as widgets\nfrom IPython.display import display","metadata":{"execution":{"iopub.status.busy":"2024-06-15T23:24:36.506009Z","iopub.execute_input":"2024-06-15T23:24:36.506400Z","iopub.status.idle":"2024-06-15T23:24:38.029104Z","shell.execute_reply.started":"2024-06-15T23:24:36.506354Z","shell.execute_reply":"2024-06-15T23:24:38.027966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Lets check the directory structure. We will take help of os library:\n\n## Define a function to pring the files and filder overview:\ndef print_dir_str(dir_path):\n    \"\"\"A function to print the files and folders in the root input folder\"\"\"\n    \n    #creates a tupple containing root(main folder), dirs(subfolder) and files.\n    directories=os.walk(dir_path) \n    \n    #make a list of of all files and folder in input root folder.\n    start_dir_contents=os.listdir(dir_path) \n    print(f\"Total number of files and folder in root directory are = {len(start_dir_contents)}\\n---------------------------------------------------\\n\")\n    \n    #loop through tupple to print file/folder path:\n    for root,dirs,files in directories: \n        print('\\n[Root dir]\\n')\n        print(f\"{root}\")\n        print('\\n[Folders]\\n')\n        for directory in dirs:\n            print (f'{\" \"*4}{os.path.join(dir_path,directory)}')\n        print('\\n[Files]\\n')\n        for file in files:\n            print(f'{\" \"*8}{os.path.join(dir_path,file)}')\n        return(start_dir_contents)\n        break\n\n# call the function print_dir_str()\nprint_dir_str(start_dir_path)","metadata":{"execution":{"iopub.status.busy":"2024-06-15T23:24:38.032106Z","iopub.execute_input":"2024-06-15T23:24:38.032783Z","iopub.status.idle":"2024-06-15T23:24:38.046419Z","shell.execute_reply.started":"2024-06-15T23:24:38.032735Z","shell.execute_reply":"2024-06-15T23:24:38.045146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assign file path to simple variables:\n# CSV fiels provided and thier paths:\ntrain_data_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train.csv'\nsample_csv_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/sample_submission.csv'\ntrain_series_descriptions_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_series_descriptions.csv'\ntrain_label_coordinates_data_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv'\ntest_series_descriptions_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/test_series_descriptions.csv'\n\n# Image folder paths:\ntrain_images_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images'\ntest_images_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/test_images'","metadata":{"execution":{"iopub.status.busy":"2024-06-15T23:24:38.049252Z","iopub.execute_input":"2024-06-15T23:24:38.050105Z","iopub.status.idle":"2024-06-15T23:24:38.057299Z","shell.execute_reply.started":"2024-06-15T23:24:38.050069Z","shell.execute_reply":"2024-06-15T23:24:38.056036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read csv files:\nsample_csv=pd.read_csv(sample_csv_path)\ntrain_data=pd.read_csv(train_data_path)\ntrain_series_description=pd.read_csv(train_series_descriptions_path)\ntest_series_description=pd.read_csv(test_series_descriptions_path)\ntrain_label_coordinates_data=pd.read_csv(train_label_coordinates_data_path)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-15T23:24:38.058555Z","iopub.execute_input":"2024-06-15T23:24:38.058921Z","iopub.status.idle":"2024-06-15T23:24:38.248336Z","shell.execute_reply.started":"2024-06-15T23:24:38.058891Z","shell.execute_reply":"2024-06-15T23:24:38.247118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#visualise the sample csv file:\nsample_csv.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-15T23:24:38.249771Z","iopub.execute_input":"2024-06-15T23:24:38.250187Z","iopub.status.idle":"2024-06-15T23:24:38.276294Z","shell.execute_reply.started":"2024-06-15T23:24:38.250158Z","shell.execute_reply":"2024-06-15T23:24:38.275176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"You can see from above sample submission file, that our otput file should be a csv file with following features-:\n    1.Index column \n    2. headers: row_id and thre possibe diagnostic categories (4 columns in totol)\n    3. The first column- row_id= is a id generated by combination of study id + site of abnormality + level of abnormality \n    4.Reemember there are 5 sites per level and total of 5 levels per patient. so total of 25 sites per patient to be diagnosed.\n   ","metadata":{}},{"cell_type":"code","source":"# Visualise the training data csv:\ntrain_data.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-15T23:24:38.277962Z","iopub.execute_input":"2024-06-15T23:24:38.278423Z","iopub.status.idle":"2024-06-15T23:24:38.306960Z","shell.execute_reply.started":"2024-06-15T23:24:38.278366Z","shell.execute_reply":"2024-06-15T23:24:38.305754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualise train_series_description\ntrain_series_description.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-15T23:24:38.308341Z","iopub.execute_input":"2024-06-15T23:24:38.308716Z","iopub.status.idle":"2024-06-15T23:24:38.320513Z","shell.execute_reply.started":"2024-06-15T23:24:38.308678Z","shell.execute_reply":"2024-06-15T23:24:38.319293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**So we can see that each patient (study) has three series, one T1 sagittal, one T2 Sagittal and one T2 axial.**","metadata":{}},{"cell_type":"code","source":"# Visualise test_series_description\ntest_series_description.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-15T23:24:38.321864Z","iopub.execute_input":"2024-06-15T23:24:38.322213Z","iopub.status.idle":"2024-06-15T23:24:38.336513Z","shell.execute_reply.started":"2024-06-15T23:24:38.322182Z","shell.execute_reply":"2024-06-15T23:24:38.335258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Drop all missing values fromt train_data:\ntrain=train_data.dropna()\n\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2024-06-15T23:24:38.339887Z","iopub.execute_input":"2024-06-15T23:24:38.340715Z","iopub.status.idle":"2024-06-15T23:24:38.359256Z","shell.execute_reply.started":"2024-06-15T23:24:38.340670Z","shell.execute_reply":"2024-06-15T23:24:38.358009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.shape","metadata":{"execution":{"iopub.status.busy":"2024-06-15T23:24:38.360759Z","iopub.execute_input":"2024-06-15T23:24:38.361110Z","iopub.status.idle":"2024-06-15T23:24:38.367928Z","shell.execute_reply.started":"2024-06-15T23:24:38.361081Z","shell.execute_reply":"2024-06-15T23:24:38.366815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_label_coordinates_data.shape","metadata":{"execution":{"iopub.status.busy":"2024-06-15T23:24:38.369777Z","iopub.execute_input":"2024-06-15T23:24:38.370128Z","iopub.status.idle":"2024-06-15T23:24:38.381887Z","shell.execute_reply.started":"2024-06-15T23:24:38.370097Z","shell.execute_reply":"2024-06-15T23:24:38.380861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-15T23:24:38.383321Z","iopub.execute_input":"2024-06-15T23:24:38.383714Z","iopub.status.idle":"2024-06-15T23:24:38.410554Z","shell.execute_reply.started":"2024-06-15T23:24:38.383683Z","shell.execute_reply":"2024-06-15T23:24:38.409252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add a new column to store the categories\ntrain_label_coordinates_data[\"category\"] = None\n\n# Iterate through the rows of train_label_coordinates_data\nfor idx, row in train_label_coordinates_data.iterrows():\n    r = row['study_id']\n    col = (row['condition'].lower().replace(' ', '_')) + '_' + (row['level'].lower().replace('/', '_'))\n    \n    # Ensure the column exists in train_data to avoid KeyError\n    if col in train_data.columns:\n        # Check if there is a matching study_id and the column exists\n        value = train_data.loc[train_data['study_id'] == r, col].values\n        if len(value) > 0:\n            train_label_coordinates_data.at[idx, \"category\"] = value[0]\n        else:\n            train_label_coordinates_data.at[idx, \"category\"] = None\n    else:\n        train_label_coordinates_data.at[idx, \"category\"] = None\n\n# Display the first 5 rows of the updated DataFrame\ntrain_label_coordinates_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-15T23:24:38.411655Z","iopub.execute_input":"2024-06-15T23:24:38.411966Z","iopub.status.idle":"2024-06-15T23:25:00.360362Z","shell.execute_reply.started":"2024-06-15T23:24:38.411935Z","shell.execute_reply":"2024-06-15T23:25:00.359326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that there are 25 sites for each study (as expected). The data above gives exact slice number (instance_numner column) and series number where the abnormality is seen best along with an anotated cooridinate which markes the exact site of abnormality.\nWe can see from the overview provided by competition host that :\nimage file name is labelled as : \n*** [train/test]_images/[study_id]/[series_id]/[instance_number].dcm ***\ntherefore in the first row, the image containing abnormality will have following path:\n*** /kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images/4003253/702807833/8.dcm ***\n**Lets visualise one of this image showing the abnormality.**","metadata":{}},{"cell_type":"code","source":"# Now we need to create a list of image file paths :\n# For this, let us first create a new df containing only the first three columns. All we need to do is to drop rest of the columns.\nfirst_three_cols=train_label_coordinates_data.iloc[:,:3]\nfirst_three_cols.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-15T23:25:00.362053Z","iopub.execute_input":"2024-06-15T23:25:00.362501Z","iopub.status.idle":"2024-06-15T23:25:00.373982Z","shell.execute_reply.started":"2024-06-15T23:25:00.362459Z","shell.execute_reply":"2024-06-15T23:25:00.372840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now all we have to do is to conver this df into a new df that contains combined values of the each row.\njoin_row_as_str=lambda row: train_images_path+\"/\"+\"/\".join(row.astype(str))+\".dcm\"\ntrain_label_images_path=first_three_cols.apply(join_row_as_str,axis=1)\nprint(f'The number of rows in train_label_images_path and train_label_coordinates_data are equal? {train_label_images_path.shape[0]==train_label_coordinates_data.shape[0]}')\ntrain_label_coordinates_data.shape","metadata":{"execution":{"iopub.status.busy":"2024-06-15T23:25:00.375754Z","iopub.execute_input":"2024-06-15T23:25:00.376482Z","iopub.status.idle":"2024-06-15T23:25:03.695087Z","shell.execute_reply.started":"2024-06-15T23:25:00.376436Z","shell.execute_reply":"2024-06-15T23:25:03.693949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add img file patth as column in train_label_coordinates_data\n# Add a new column to store the categories\ntrain_label_coordinates_data[\"img_file_path\"] = None\n\n# Iterate through the rows of train_label_coordinates_data\nfor idx, row in train_label_coordinates_data.iterrows():\n    file_path = train_label_images_path.loc[idx]\n    train_label_coordinates_data.at[idx, \"img_file_path\"] = file_path\n# Display the first 5 rows of the updated DataFrame\ntrain_label_coordinates_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-15T23:25:03.696430Z","iopub.execute_input":"2024-06-15T23:25:03.696734Z","iopub.status.idle":"2024-06-15T23:25:08.027171Z","shell.execute_reply.started":"2024-06-15T23:25:03.696706Z","shell.execute_reply":"2024-06-15T23:25:08.026113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_label_coordinates_data.shape","metadata":{"execution":{"iopub.status.busy":"2024-06-15T23:25:08.028371Z","iopub.execute_input":"2024-06-15T23:25:08.028682Z","iopub.status.idle":"2024-06-15T23:25:08.035252Z","shell.execute_reply.started":"2024-06-15T23:25:08.028656Z","shell.execute_reply":"2024-06-15T23:25:08.034119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the list of DICOM file paths\nimg_file_paths = train_label_coordinates_data['img_file_path'].tolist()\nx_coordinates= train_label_coordinates_data['x'].tolist()\ny_coordinates= train_label_coordinates_data['y'].tolist()\n\n# Define a function to display DICOM images for a given file path\ndef display_dicom_images(idx):\n    \"\"\"A function to display aDICOM images for a given index.\n    The function takes idx , a integer whihc represent index of the image.\n    The images are annoted by a transparent circle to marke the x and y coordinates marking \n    the epicentre of the site being analysed\"\"\"\n    file_path = img_file_paths[idx]\n    # Import DICOM object\n    dicom_obj = pydicom.dcmread(file_path)\n    # Read DICOM object as image\n    img = dicom_obj.pixel_array\n    # Display the image using matplotlib.pyplot, use grayscale color map to display image\n    plt.imshow(img, cmap='gray')\n    plt.title(f'{train_label_coordinates_data.loc[idx, \"condition\"]} at {train_label_coordinates_data.loc[idx, \"level\"]}\\n-------------\\n{train_label_coordinates_data.loc[idx, \"category\"]}')\n    plt.axis('off')\n    \n    #overlay annotation on images:\n      # Overlay annotation (colored circle)\n    x = x_coordinates[idx]\n    y = y_coordinates[idx]\n    #circle = plt.patches.Circle((x, y), radius=20, color='yellow', fill=False, linestyle='dashed')\n    plt.plot(x_coordinates[idx],y_coordinates[idx],'o', markersize=20, linestyle=\"solid\",alpha=0.5)  # 'o' for circle, alpha for transparency\n    \n    plt.show()\n\n# Create a widget for scrolling through images\nindex_slider = widgets.IntSlider(value=0, min=0, max=len(img_file_paths) - 1, step=1, description='Image Index:')\n\n# Use the widget to interactively scroll through images\nwidgets.interact(display_dicom_images, idx=index_slider)\n\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2024-06-15T23:25:08.036925Z","iopub.execute_input":"2024-06-15T23:25:08.037380Z","iopub.status.idle":"2024-06-15T23:25:08.383890Z","shell.execute_reply.started":"2024-06-15T23:25:08.037341Z","shell.execute_reply":"2024-06-15T23:25:08.382646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Global variable to keep track of the current image index\ncurrent_img_index=0\ndef display_next_img():\n    \"\"\"A function to display images one after another each time the cell is run.\"\"\"\n    global current_img_index\n    \n    # Check if current_img_index is within bounds\n    if current_img_index < len(img_file_paths):\n        display_dicom_images(current_img_index)  # Assuming display_dicom_images is defined elsewhere\n        current_img_index += 1  # Increment the index for the next image\n    else:\n        print(\"End of image list reached. Resetting to the beginning.\")\n        current_img_index = 0  # Reset index to start from the beginning\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-15T23:25:08.385221Z","iopub.execute_input":"2024-06-15T23:25:08.385552Z","iopub.status.idle":"2024-06-15T23:25:08.392337Z","shell.execute_reply.started":"2024-06-15T23:25:08.385523Z","shell.execute_reply":"2024-06-15T23:25:08.391014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Call the function to display the next image\ndisplay_next_img()","metadata":{"execution":{"iopub.status.busy":"2024-06-15T23:25:08.393916Z","iopub.execute_input":"2024-06-15T23:25:08.394351Z","iopub.status.idle":"2024-06-15T23:25:08.660147Z","shell.execute_reply.started":"2024-06-15T23:25:08.394309Z","shell.execute_reply":"2024-06-15T23:25:08.658858Z"},"trusted":true},"execution_count":null,"outputs":[]}]}