{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":30715,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#** Original code whihc came with notebook was displaying all the files (more 45,000 files !). This is cumbersome. I have commented this code and replaced it with new code\n#whihc ensures that only first n files name and paths are displayed where n is the any desired numer. ALos wroe a fucntion print_drectory_strucutre whihc displays directory sturcutre.\n#See below code cell for the same. \n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-06T21:26:25.867943Z","iopub.execute_input":"2024-06-06T21:26:25.868467Z","iopub.status.idle":"2024-06-06T21:26:25.873626Z","shell.execute_reply.started":"2024-06-06T21:26:25.868425Z","shell.execute_reply":"2024-06-06T21:26:25.872395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Now define a the input directoy path:\nstart_dir_path=\"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification\"","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:25.875785Z","iopub.execute_input":"2024-06-06T21:26:25.876213Z","iopub.status.idle":"2024-06-06T21:26:25.889951Z","shell.execute_reply.started":"2024-06-06T21:26:25.876175Z","shell.execute_reply":"2024-06-06T21:26:25.888811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Lets check the directory structure. We will take help of os library:\nimport os\ndef print_dir_str(dir_path):\n    \"\"\"A function to print the files and folders in the root input folder\"\"\"\n    \n    #creates a tupple containing root(main folder), dirs(subfolder) and files.\n    directories=os.walk(dir_path) \n    \n    #make a list of of all files and folder in input root folder.\n    start_dir_contents=os.listdir(dir_path) \n    print(f\"Total number of files and folder in root directory are = {len(start_dir_contents)}\\n---------------------------------------------------\\n\")\n    \n    #loop through tupple to print file/folder path:\n    for root,dirs,files in directories: \n        print('\\n[Root dir]\\n')\n        print(f\"{root}\")\n        print('\\n[Folders]\\n')\n        for directory in dirs:\n            print (f'{\" \"*4}{os.path.join(dir_path,directory)}')\n        print('\\n[Files]\\n')\n        for file in files:\n            print(f'{\" \"*8}{os.path.join(dir_path,file)}')\n        return(start_dir_contents,directories)\n        break\nprint_dir_str(start_dir_path)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:25.891422Z","iopub.execute_input":"2024-06-06T21:26:25.891737Z","iopub.status.idle":"2024-06-06T21:26:25.908542Z","shell.execute_reply.started":"2024-06-06T21:26:25.891710Z","shell.execute_reply":"2024-06-06T21:26:25.907469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Here is more relevant information.\n# There are two folders cotaining images in the data set provided. One folder is for training images and other is for valdiation images.\n#Thte stenosis is graded into mild (same as normal) , moderate or severe with wieght of 1,2 and 4 rrespectively.\n#Below are images depicting this classification for Neural foramina stenosis, subarticular stenosis (also known as lateral recess stenosis) and spinal stenosis respectively. ","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:25.911249Z","iopub.execute_input":"2024-06-06T21:26:25.911552Z","iopub.status.idle":"2024-06-06T21:26:25.916639Z","shell.execute_reply.started":"2024-06-06T21:26:25.911527Z","shell.execute_reply":"2024-06-06T21:26:25.915307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**NEURAL FORAMINA STENOSIS CLASSIFICATION**\n![Classification of nerual foramina stenosis](https://i.imgur.com/6c7erNM.png)\n![Classification of nerual foramina stenosis](https://i.imgur.com/b1VGiN5.png)\n**SUBARTICUALR RECESS STENOSIS CLASSIFICATION**\n![Classification of nerual foramina stenosis](https://files.miamineurosciencecenter.com/media/filer_public_thumbnails/filer_public/d5/08/d508ae6a-a4f2-4796-be9f-455f8df45fe1/herniation_zones.jpg__1700.0x1308.0_q85_subject_location-850%2C656_subsampling-2.jpg)\n![Classification of nerual foramina stenosis](https://i.imgur.com/Usuxgge.png)\n**SPINAL CANAL STENOSIS CLASSIFICATION**\n![Classification of nerual foramina stenosis](https://prod-images-static.radiopaedia.org/images/940993/f7a8adca63efae788f621869cc21e8_big_gallery.jpg)\n![Classification of nerual foramina stenosis](https://i.imgur.com/opjnAwl.png)\n  ","metadata":{}},{"cell_type":"code","source":"#First import relevant libraries:\n\nimport pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt #For ploating and working with graphs\nimport seaborn as sns\nimport cv2 #Computer vision version 2 library for loading and reading images\nimport pydicom #for working wiht dicom images (MRI images are dicom images with .dcm extension)\n#import os # Already imported above. This Operating system library is to work with computer kernel.usefyl to interact with computer kernel and run commands like terminal. \nimport glob #glob library helps finding global patterns in file names. Not sure how it will be useful !\nfrom tqdm import tqdm #Taqadum is arabic word meanng progress and hence tqdm is ibrary which help in showing progress bars.\nimport warnings #lbraary helps in rasing warning messages where needed. ","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:25.917880Z","iopub.execute_input":"2024-06-06T21:26:25.918247Z","iopub.status.idle":"2024-06-06T21:26:27.421756Z","shell.execute_reply.started":"2024-06-06T21:26:25.918220Z","shell.execute_reply":"2024-06-06T21:26:27.420737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Exciting times ! We beging with reading our source data-:\ntrain_data_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train.csv'\ntrain = pd.read_csv(train_data_path)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:27.422988Z","iopub.execute_input":"2024-06-06T21:26:27.423466Z","iopub.status.idle":"2024-06-06T21:26:27.456766Z","shell.execute_reply.started":"2024-06-06T21:26:27.423439Z","shell.execute_reply":"2024-06-06T21:26:27.455705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#Visualise first 4 rows of traning data-: \n\n#summerise and study the data:\n#First gettng total number of cases:\nprint(\"The total number of cases are : {}\".format(len(train)))\n#How many columns:\nprint(\"Total number of columns are : {}\".format(len(train.columns)))\n#printing the shape of data frame :\ntrain_shape = train.shape\nprint(\"The shape of our data frame is  ..... \\nNumber of rows are : {}.\\nNumber of columns are: {}\".format(train_shape[0], train_shape[1]))\nprint(\"-------------------------------------------\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:27.459988Z","iopub.execute_input":"2024-06-06T21:26:27.460358Z","iopub.status.idle":"2024-06-06T21:26:27.500317Z","shell.execute_reply.started":"2024-06-06T21:26:27.460328Z","shell.execute_reply":"2024-06-06T21:26:27.499241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So we have roughly **2,000 cases** for our model buidling, its not a large number and I would like to have more images !\n","metadata":{}},{"cell_type":"markdown","source":"The data tells us that:\n**1**There are **5 potential sites of stenosis** in spine (left and right foraminal,left and rigth subarticular and central) at each of the five spinal levels. \n**2**The stenosis at each site can be uniquely categorised into either mild(/normal) moderate or severe stenosis.\nWe will call these different sites as **sites**.\nand we will each of the daignostic categories as **categories**","metadata":{}},{"cell_type":"code","source":"# We need to extract sites from the columns (be removing study_id):\ncolumns=train.columns\nsites=[col for col in columns if col!=\"study_id\"] #sttudy is is not a site !\n#he diagnostic categgories can be put ina list named cateogories\n#As a prepartion to use pd.unique() function extract diagnostic catefories , frist flattern the 2d array into 1d array using ravel()\nflattened_train=train.dropna()[sites].values.ravel()\ncategories=pd.unique(flattened_train)\n\nprint(f'{categories} \\n---------------- \\n Total number of categories : {len(categories)}')\nprint(f'Potential sites of stenosis are :\\n{sites}')","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:27.501633Z","iopub.execute_input":"2024-06-06T21:26:27.502019Z","iopub.status.idle":"2024-06-06T21:26:27.527005Z","shell.execute_reply.started":"2024-06-06T21:26:27.501985Z","shell.execute_reply":"2024-06-06T21:26:27.525913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n## To understand data and trends better we will create a dataframe (frequency table)\n#this dataframe/frequency table will cntain three columns - each representing one diagnostic category, \n#Tge rows will be indexed as one of the each 25 site of stenosis\n## each row will give frequnect of the diagnostic category occuring at a particual site\n\n## create empty data frame call it counts_df\n\ncounts_df=pd.DataFrame(index=sites,columns=categories)\n\n##define a function to make frequenct table. \ndef make_frequency_table(train, sites, categories):\n    for site in sites:\n        counts = train[site].value_counts()\n        for category in categories:\n            counts_df.at[site, category] = counts.get(category, 0)\n    print(\"Congratulations-frequencty table is now ready:\\nHere\\'s a preview of the generated Frequency table\\nFREQUENCY TABLE\\n------------------\\n\")\n    return counts_df\n\n# Generate the frequency table\ncounts_df = make_frequency_table(train, sites, categories)\n\ncounts_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:27.528435Z","iopub.execute_input":"2024-06-06T21:26:27.528819Z","iopub.status.idle":"2024-06-06T21:26:27.566574Z","shell.execute_reply.started":"2024-06-06T21:26:27.528784Z","shell.execute_reply":"2024-06-06T21:26:27.565297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use('seaborn-whitegrid')\n# Set Matplotlib defaults\nplt.rc('figure', autolayout=True)\nplt.rc('axes', labelweight='bold', labelsize='large',\n       titleweight='bold', titlesize=18, titlepad=10)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:27.567827Z","iopub.execute_input":"2024-06-06T21:26:27.568171Z","iopub.status.idle":"2024-06-06T21:26:27.574666Z","shell.execute_reply.started":"2024-06-06T21:26:27.568122Z","shell.execute_reply":"2024-06-06T21:26:27.573554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counts_df=counts_df.T\n#Lets visualise this data: we will create 25 plots, one for each site,where the x-axis shows the diagnostic categories \n#and the y-axis shows the counts,\n## we will use followig colors lsit to ensure taht each daignostic category has a unique colour.\n\ncolors=['#1f77b4','#2ca02c', '#ff7f0e' ]\n\n#The function plot_graph (below) can accept desired number of row and column, hieght and widht of the graph. \n\ndef plot_graph(data_set, features, grid_col, grid_row, canvas_width, canvas_height, colors):\n    # Define plot area/canvas.\n    fig, axes = plt.subplots(grid_col, grid_row, figsize=(canvas_width, canvas_height), sharey=True)\n    \n    # Flatten the 2D array axes to 1D array.\n    axes = axes.flatten()\n    \n    #Create colors series using matplotlib\n    cmap=plt.get_cmap('hsv')\n    n_colors=len(features)\n    colors=[cmap(i/n_colors) for i in range(n_colors)]\n    # Loop through each site and plot the data for each site.\n    for idx, feature in enumerate(features):\n        data_set[feature].plot(kind='bar', ax=axes[idx], color=colors)\n        axes[idx].set_title(feature.replace(\"_\", \" \").title())\n        axes[idx].set_xlabel(\"Diagnostic category\")  # Fixed typo in 'xlabel'\n        axes[idx].set_ylabel(\"Count\")\n    \n    plt.tight_layout()  # Fixed typo in 'tight_layout'\n    plt.show()\n\nplot_graph(counts_df,sites,5,5,20,30,colors)\n\ncounts_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:45:31.464443Z","iopub.execute_input":"2024-06-06T21:45:31.465635Z","iopub.status.idle":"2024-06-06T21:45:38.014774Z","shell.execute_reply.started":"2024-06-06T21:45:31.465597Z","shell.execute_reply":"2024-06-06T21:45:38.013735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Lets visualize above data in pie chart instead-:\ndef plot_pie_chart(data_set, features, grid_col, grid_row, canvas_width, canvas_height, colors):\n    #Define plot area/canvas.\n    fig, axes = plt.subplots(grid_col, grid_row, figsize=(canvas_width, canvas_height))\n    \n    # Flatten the 2D array axes to 1D array.\n    axes = axes.flatten()\n    #Create colors series using matplotlib\n    cmap=plt.get_cmap('hsv')\n    n_colors=len(features)\n    colors=[cmap(i/n_colors) for i in range(n_colors)]\n    \n    for idx, feature in enumerate(features):\n        counts = data_set[feature].values\n        labels = data_set[feature].index\n        axes[idx].pie(counts, labels=labels, colors=colors, autopct='%1.1f%%', startangle=140)\n        axes[idx].set_title(feature.replace(\"_\", \" \").title())\n    plt.tight_layout()\n    plt.show()\nplot_pie_chart(counts_df,sites,5,5,20,30,colors)\n    ","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:34.120258Z","iopub.execute_input":"2024-06-06T21:26:34.120594Z","iopub.status.idle":"2024-06-06T21:26:37.263360Z","shell.execute_reply.started":"2024-06-06T21:26:34.120564Z","shell.execute_reply":"2024-06-06T21:26:37.262336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## We have till depicted data to compare the degeree of stenosis at various sites; Now lets try\n##and depict data such that there are total of three outputs- each graph/output depicts distrubution of\n## a specific degree of stenossi at various sites. For example first graph will be titled as Mild/Normal\n## The X axis will depict the potential sites of stenosis (a total of 25 sites) and Y axis will depict \n## the number of patients at each sites having mild/moderate stenosis.","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:37.264656Z","iopub.execute_input":"2024-06-06T21:26:37.265625Z","iopub.status.idle":"2024-06-06T21:26:37.270591Z","shell.execute_reply.started":"2024-06-06T21:26:37.265583Z","shell.execute_reply":"2024-06-06T21:26:37.269471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#rcheck current status of counts_df (remember we trasnposed it!)\ncounts_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:37.271868Z","iopub.execute_input":"2024-06-06T21:26:37.272303Z","iopub.status.idle":"2024-06-06T21:26:37.306651Z","shell.execute_reply.started":"2024-06-06T21:26:37.272268Z","shell.execute_reply":"2024-06-06T21:26:37.305422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#We will have to change columns to categories and rows to sites.\ncounts_df=counts_df.T\n#we can now simply run previously defined function but features will now change to categories instead of sites\nplot_graph(counts_df,categories,3,1,20,30,colors)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:37.312481Z","iopub.execute_input":"2024-06-06T21:26:37.313265Z","iopub.status.idle":"2024-06-06T21:26:39.013575Z","shell.execute_reply.started":"2024-06-06T21:26:37.313222Z","shell.execute_reply":"2024-06-06T21:26:39.012223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_pie_chart(counts_df,categories,3,1,20,30,colors)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:39.015414Z","iopub.execute_input":"2024-06-06T21:26:39.016106Z","iopub.status.idle":"2024-06-06T21:26:40.761790Z","shell.execute_reply.started":"2024-06-06T21:26:39.016070Z","shell.execute_reply":"2024-06-06T21:26:40.760494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Obvious patterns visible from above visualisations -:**\n1. Severe stenosis is much less common than moderate and even more less common then normal patients. \n2. Amnong 25 sites, the absense of stenosis is more or less uniformly distrubuted.\n3. But L4-L5 level is much more likey to demonstrate severe or moderate stenosis then higher levels. Second most common level with severe or moderate stenosis is L5-S1.","metadata":{}},{"cell_type":"markdown","source":"**Lets compare the distribution degree of stenosis on right and left sides (applicable ony for formainal and subarticular stenosis.**\n**Lets also analyse each site seprately**\n**For this we will first create seprate dataframe for each site. We will then cfreate a dataframe for containing date for right and left foraminal as subarticual sites only.","metadata":{}},{"cell_type":"code","source":"# we can iterate though the dataset though column names. Since we want to make new dataframe according to the site,\n#we will have to transpose our dataframe once again.\ncounts_df=counts_df.T\ncounts_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:40.763372Z","iopub.execute_input":"2024-06-06T21:26:40.763797Z","iopub.status.idle":"2024-06-06T21:26:40.796282Z","shell.execute_reply.started":"2024-06-06T21:26:40.763761Z","shell.execute_reply":"2024-06-06T21:26:40.794963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Lets create dataframe for each site seperately -:\n# lets first create empty data frames for each sites:\n\nspecific_sites=['canal','foraminal','subarticular']\ncanal_sites=[]\nforaminal_sites=[]\nsubarticular_sites=[]\ncanal_df= pd.DataFrame()\nforaminal_df=pd.DataFrame()\nsubarticular_df= pd.DataFrame()\ndef create_site_specific_df(data_set):\n    for col_name in data_set:\n        if 'canal' in col_name:\n            canal_sites.append(col_name)\n        if \"foraminal\" in col_name:\n                \n            foraminal_sites.append(col_name)\n        if \"subarticular\" in col_name:\n            subarticular_sites.append(col_name)\n    \n    for canal_site in canal_sites:\n        for category in categories:\n            canal_df.at[canal_site, category] = data_set[canal_site].get(category, 0)\n    for foraminal_site in foraminal_sites:\n        for category in categories:\n            foraminal_df.at[foraminal_site, category] = data_set[foraminal_site].get(category, 0)\n    for subarticular_site in subarticular_sites:\n        for category in categories:\n            subarticular_df.at[subarticular_site, category] = data_set[subarticular_site].get(category, 0)\n    return canal_df,foraminal_df,subarticular_df,canal_sites,foraminal_sites,subarticular_sites\n                \n                \n\ncreate_site_specific_df(counts_df)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:40.798636Z","iopub.execute_input":"2024-06-06T21:26:40.799083Z","iopub.status.idle":"2024-06-06T21:26:40.866756Z","shell.execute_reply.started":"2024-06-06T21:26:40.799042Z","shell.execute_reply":"2024-06-06T21:26:40.865477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"canal_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:40.868257Z","iopub.execute_input":"2024-06-06T21:26:40.868600Z","iopub.status.idle":"2024-06-06T21:26:40.881745Z","shell.execute_reply.started":"2024-06-06T21:26:40.868572Z","shell.execute_reply":"2024-06-06T21:26:40.880472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"foraminal_df","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:40.883293Z","iopub.execute_input":"2024-06-06T21:26:40.883711Z","iopub.status.idle":"2024-06-06T21:26:40.907980Z","shell.execute_reply.started":"2024-06-06T21:26:40.883674Z","shell.execute_reply":"2024-06-06T21:26:40.906691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subarticular_df","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:40.909290Z","iopub.execute_input":"2024-06-06T21:26:40.909826Z","iopub.status.idle":"2024-06-06T21:26:40.926610Z","shell.execute_reply.started":"2024-06-06T21:26:40.909783Z","shell.execute_reply":"2024-06-06T21:26:40.925519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Now we can create seperate pie charts and bar charts:\n# First we will ahve to get features into to index and so transpose the dataframes.\ncanal_df=canal_df.T\nsubarticular_df=subarticular_df.T\nforaminal_df=foraminal_df.T","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:40.927841Z","iopub.execute_input":"2024-06-06T21:26:40.928187Z","iopub.status.idle":"2024-06-06T21:26:40.940131Z","shell.execute_reply.started":"2024-06-06T21:26:40.928137Z","shell.execute_reply":"2024-06-06T21:26:40.938918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create graphs:\nplot_graph(canal_df,canal_sites,1,5,20,5,colors)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:40.941651Z","iopub.execute_input":"2024-06-06T21:26:40.942090Z","iopub.status.idle":"2024-06-06T21:26:41.947459Z","shell.execute_reply.started":"2024-06-06T21:26:40.942052Z","shell.execute_reply":"2024-06-06T21:26:41.946358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_pie_chart(canal_df,canal_sites,1,5,20,5,colors)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:41.948823Z","iopub.execute_input":"2024-06-06T21:26:41.949179Z","iopub.status.idle":"2024-06-06T21:26:42.538105Z","shell.execute_reply.started":"2024-06-06T21:26:41.949138Z","shell.execute_reply":"2024-06-06T21:26:42.536932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"foraminal_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:42.539518Z","iopub.execute_input":"2024-06-06T21:26:42.539889Z","iopub.status.idle":"2024-06-06T21:26:42.564397Z","shell.execute_reply.started":"2024-06-06T21:26:42.539856Z","shell.execute_reply":"2024-06-06T21:26:42.563205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_pie_chart(foraminal_df,foraminal_sites,5,2,20,30,colors)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:42.565792Z","iopub.execute_input":"2024-06-06T21:26:42.566206Z","iopub.status.idle":"2024-06-06T21:26:43.979827Z","shell.execute_reply.started":"2024-06-06T21:26:42.566169Z","shell.execute_reply":"2024-06-06T21:26:43.978746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_pie_chart(subarticular_df,subarticular_sites,5,2,20,30,colors)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:43.981010Z","iopub.execute_input":"2024-06-06T21:26:43.981361Z","iopub.status.idle":"2024-06-06T21:26:45.683962Z","shell.execute_reply.started":"2024-06-06T21:26:43.981331Z","shell.execute_reply":"2024-06-06T21:26:45.682754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_graph(foraminal_df,foraminal_sites,5,2,20,30,colors)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:45.685356Z","iopub.execute_input":"2024-06-06T21:26:45.685754Z","iopub.status.idle":"2024-06-06T21:26:48.243718Z","shell.execute_reply.started":"2024-06-06T21:26:45.685716Z","shell.execute_reply":"2024-06-06T21:26:48.242430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_graph(subarticular_df,subarticular_sites,5,2,20,30,colors)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:48.245106Z","iopub.execute_input":"2024-06-06T21:26:48.245451Z","iopub.status.idle":"2024-06-06T21:26:50.958569Z","shell.execute_reply.started":"2024-06-06T21:26:48.245424Z","shell.execute_reply":"2024-06-06T21:26:50.957453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counts_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:32:21.983900Z","iopub.execute_input":"2024-06-06T21:32:21.984345Z","iopub.status.idle":"2024-06-06T21:32:22.006414Z","shell.execute_reply.started":"2024-06-06T21:32:21.984311Z","shell.execute_reply":"2024-06-06T21:32:22.005282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Before we proceed to analysing images, lets do one more analyzes.\nLets visualise what happens to incidence of mild , moderate and severe stenosis as we move down the different spnal levels.","metadata":{}},{"cell_type":"markdown","source":"*Plotting graph to anlyse relation between spinal level and degree of stenosis*","metadata":{}},{"cell_type":"code","source":"#plot line graphs to see trend of incidence of mild , moderate and severe stenosis at different levels:\n# we will have to create new df - call it level_counts_df - the columns will be various level and rows will be diagnostic catefories.\n# let levels be a list with levels:\nlevels=[\"l1_l2\",\"l2_l3\",\"l3_l4\",\"l4_l5\",\"l5_s1\"]\nlevels_dict={}\nlevels_counts_df= pd.DataFrame(index=levels,columns=categories)\nfor category in categories:\n    for level in levels:\n        level_value=0\n        for col in columns:\n            if level in col:\n                level_value=level_value+(counts_df[col].get(category,0))\n                levels_dict[level]=level_value\n        levels_counts_df.loc[level,category]=level_value\nlevels_counts_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-07T00:28:30.898164Z","iopub.execute_input":"2024-06-07T00:28:30.898659Z","iopub.status.idle":"2024-06-07T00:28:30.918541Z","shell.execute_reply.started":"2024-06-07T00:28:30.898616Z","shell.execute_reply":"2024-06-07T00:28:30.917244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# Define fig and axes object\nfig, axes = plt.subplots(1,1,figsize=(20, 10))\ncolors=['green','orange','red']\ni=-1\n# Loop through DataFrame columns and plot on each subplot\nfor idx, col in enumerate(levels_counts_df.columns):\n    i+=1\n    levels_counts_df[col].plot(ax=axes, label=col,color=colors[i])\n\n    # Set title, xlabel, and ylabel for the current subplot\n    axes.set_title('Severity of Stenosis Across Spinal Levels')\n    axes.set_xlabel('Spinal Level')\n    axes.set_ylabel('Number of Patients')\n    axes.legend()\n\n# Show the plot\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-07T00:59:06.857785Z","iopub.execute_input":"2024-06-07T00:59:06.858203Z","iopub.status.idle":"2024-06-07T00:59:07.304630Z","shell.execute_reply.started":"2024-06-07T00:59:06.858170Z","shell.execute_reply":"2024-06-07T00:59:07.303492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Working with DICOM images**","metadata":{}},{"cell_type":"code","source":"#Find the list of files in train images data set. first once again visualise dir structure\n#use print_dir_str(dir_path) fucntion.\nprint_dir_str(start_dir_path)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:50.959980Z","iopub.execute_input":"2024-06-06T21:26:50.960339Z","iopub.status.idle":"2024-06-06T21:26:50.969263Z","shell.execute_reply.started":"2024-06-06T21:26:50.960301Z","shell.execute_reply":"2024-06-06T21:26:50.968215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define train and test images paths\ntrain_images_path = os.path.join(start_dir_path, 'train_images')\ntest_images_path = os.path.join(start_dir_path, 'test_images')","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:50.970784Z","iopub.execute_input":"2024-06-06T21:26:50.971169Z","iopub.status.idle":"2024-06-06T21:26:50.982069Z","shell.execute_reply.started":"2024-06-06T21:26:50.971119Z","shell.execute_reply":"2024-06-06T21:26:50.980673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nimport os\n\n# Define the base directory path\nstart_dir_path = \"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification\"  # Ensure this path is correct\n\n# Define train and test images paths\ntrain_images_path = os.path.join(start_dir_path, 'train_images')\ntest_images_path = os.path.join(start_dir_path, 'test_images')\n\n# Construct the pattern with ** to match files at any depth\ntrain_images_path_pattern = os.path.join(train_images_path, '**', '*.dcm')\nprint(f\"Pattern: {train_images_path_pattern}\")  # Debugging step\n\n# Use glob with recursive=True to find all .dcm files\ntrain_image_files_list = glob.glob(train_images_path_pattern, recursive=True)\n\n# Print the total number of files found\nprint(f\"Total number of files: {len(train_image_files_list)}\")\n\n# Print the type of train_image_files_list to ensure it is a list\nprint(f\"Type of train_image_files_list: {type(train_image_files_list)}\")\n\n# Optionally, print the first few file paths to verify correct matching\nprint(f\"Sample file paths: {train_image_files_list[:2]}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-06-06T21:26:50.983584Z","iopub.execute_input":"2024-06-06T21:26:50.983958Z","iopub.status.idle":"2024-06-06T21:27:31.332041Z","shell.execute_reply.started":"2024-06-06T21:26:50.983918Z","shell.execute_reply":"2024-06-06T21:27:31.330960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}