{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"},{"sourceId":184917418,"sourceType":"kernelVersion"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Configure notebook to use GP**\n\nEnsure that tensorflow uses GPU\n\n> Check if tensorflow is having access to GPU","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nif (len(tf.config.list_physical_devices('GPU'))==1):\n    print(\"TensorFlow has access to the GPU.\")","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:16:22.694444Z","iopub.execute_input":"2024-06-23T17:16:22.694911Z","iopub.status.idle":"2024-06-23T17:16:22.700741Z","shell.execute_reply.started":"2024-06-23T17:16:22.694865Z","shell.execute_reply":"2024-06-23T17:16:22.699519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove warnings !\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:16:22.702911Z","iopub.execute_input":"2024-06-23T17:16:22.7034Z","iopub.status.idle":"2024-06-23T17:16:22.720803Z","shell.execute_reply.started":"2024-06-23T17:16:22.70336Z","shell.execute_reply":"2024-06-23T17:16:22.719592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Ensure that Tensorflow uses GPU","metadata":{}},{"cell_type":"code","source":"gpus=tf.config.list_physical_devices('GPU')\nif gpus:\n    try:\n        for gpu in gpus:\n            tf.config.experimental.set_memory_growth(gpu, True)\n            print(\"GPU is available and configured sucesssfully\")\n    except RuntimeError as e:\n        print(e)\nelse:\n    print('No GPUS found!')","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:16:22.733155Z","iopub.execute_input":"2024-06-23T17:16:22.733582Z","iopub.status.idle":"2024-06-23T17:16:22.741518Z","shell.execute_reply.started":"2024-06-23T17:16:22.733543Z","shell.execute_reply":"2024-06-23T17:16:22.740246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Run a testcode to see if Tensorflow is using GPU:","metadata":{}},{"cell_type":"code","source":"with tf.device('/GPU:0'):\n    a = tf.constant([[1.0, 2.0, 3.0], [4.0, 5.0, 6.0]])\n    b = tf.constant([[1.0, 2.0, 3.0], [4.0, 5.0, 6.0]])\n    c = tf.matmul(a, b, transpose_b=True)  # Correct the matrix dimensions for matmul\n    print(c.numpy())","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:16:22.743314Z","iopub.execute_input":"2024-06-23T17:16:22.743722Z","iopub.status.idle":"2024-06-23T17:16:22.757511Z","shell.execute_reply.started":"2024-06-23T17:16:22.743689Z","shell.execute_reply":"2024-06-23T17:16:22.756289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import modules and libraries:","metadata":{}},{"cell_type":"code","source":"#Now define a the input directoy path:\nstart_dir_path=\"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification\"\n\n#import relevant libraries:\n\nimport pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt #For ploting and working with graphs\nimport seaborn as sns\nimport cv2 #Computer vision version 2 library for loading and reading images\nimport pydicom #for working wiht dicom images (MRI images are dicom images with .dcm extension)\nimport os # Already imported above. This Operating system library is to work with computer kernel.usefyl to interact with computer kernel and run commands like terminal. \nimport glob #glob library helps finding global patterns in file names. Not sure how it will be useful !\nfrom tqdm import tqdm #Taqadum is arabic word meanng progress and hence tqdm is ibrary which help in showing progress bars.\nimport warnings #lbraary helps in rasing warning messages where needed. \nimport ipywidgets as widgets\nfrom IPython.display import display\nimport logging\nfrom multiprocessing import Pool\nimport plotly.express as px #useful to display dicom image.\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:16:22.759972Z","iopub.execute_input":"2024-06-23T17:16:22.760467Z","iopub.status.idle":"2024-06-23T17:16:22.770043Z","shell.execute_reply.started":"2024-06-23T17:16:22.760425Z","shell.execute_reply":"2024-06-23T17:16:22.768871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Imports functions from rsna_functions","metadata":{}},{"cell_type":"code","source":"%run \"/kaggle/usr/lib/rsna_functons/rsna_functons.py\"","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:16:22.771818Z","iopub.execute_input":"2024-06-23T17:16:22.772801Z","iopub.status.idle":"2024-06-23T17:16:22.799046Z","shell.execute_reply.started":"2024-06-23T17:16:22.772754Z","shell.execute_reply":"2024-06-23T17:16:22.797488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define/Asign file paths -:","metadata":{}},{"cell_type":"code","source":"\n# Assign file path to simple variables:\n# CSV fiels provided and thier paths:\ntrain_data_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train.csv'\nsample_csv_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/sample_submission.csv'\ntrain_series_descriptions_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_series_descriptions.csv'\ntrain_label_coordinates_data_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv'\ntest_series_descriptions_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/test_series_descriptions.csv'\n\n# Image folder paths:\ntrain_images_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images'\ntest_images_path='/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/test_images'\n\n\n# Read csv files:\nsample_csv=pd.read_csv(sample_csv_path)\ntrain_data=pd.read_csv(train_data_path)\ntrain_series_description=pd.read_csv(train_series_descriptions_path)\ntest_series_description=pd.read_csv(test_series_descriptions_path)\ntrain_label_coordinates_data=pd.read_csv(train_label_coordinates_data_path)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:16:22.801054Z","iopub.execute_input":"2024-06-23T17:16:22.8014Z","iopub.status.idle":"2024-06-23T17:16:22.922411Z","shell.execute_reply.started":"2024-06-23T17:16:22.801368Z","shell.execute_reply":"2024-06-23T17:16:22.921279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data prepertion-: Preparing the labelled data-: ","metadata":{}},{"cell_type":"code","source":"# How does labelled data look-:\ndisplay(train_label_coordinates_data.head(2))\nprint(train_label_coordinates_data.shape)","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:16:22.92377Z","iopub.execute_input":"2024-06-23T17:16:22.924221Z","iopub.status.idle":"2024-06-23T17:16:22.93945Z","shell.execute_reply.started":"2024-06-23T17:16:22.924181Z","shell.execute_reply":"2024-06-23T17:16:22.938211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How does non labelled dataa looks-:\ndisplay(train_data.head(2))\nprint(train_data.shape)","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:16:22.94175Z","iopub.execute_input":"2024-06-23T17:16:22.942229Z","iopub.status.idle":"2024-06-23T17:16:22.967298Z","shell.execute_reply.started":"2024-06-23T17:16:22.942191Z","shell.execute_reply":"2024-06-23T17:16:22.966047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How are images stored in image dir-:\n\ndef read_dir_stepwise(dir_patj):\n    for root,folders,files in os.walk(train_images_path):\n        user_input = input('Do you want to print this folder? If yes, enter y; otherwise, enter n: ')\n        if user_input.lower()==\"y\":\n            if len(root)>0:\n                print(root)\n            else:\n                print(f\"There is no parent folder\")\n            if len(folders)>0:\n                print(f\"The area {len(folders)} folders in root dir\")\n            else:\n                print(f\"There is no folders in root folder\")\n            if len(files)>0:\n                print(f\"\\n....There are {len(files)} files in root folder\")\n            else:\n                print(f\"There are {len(folders)} folders in root folder but there are no files in the root folder.\")\n        else:\n            print(\"The process was aborted by the user\")\n            break\n\n    ","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:16:22.969042Z","iopub.execute_input":"2024-06-23T17:16:22.969493Z","iopub.status.idle":"2024-06-23T17:16:22.979539Z","shell.execute_reply.started":"2024-06-23T17:16:22.969451Z","shell.execute_reply":"2024-06-23T17:16:22.978095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Call above function to understand the image dir structure-:\nread_dir_stepwise(train_images_path)","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:16:22.982935Z","iopub.execute_input":"2024-06-23T17:16:22.98351Z","iopub.status.idle":"2024-06-23T17:16:27.86648Z","shell.execute_reply.started":"2024-06-23T17:16:22.983464Z","shell.execute_reply":"2024-06-23T17:16:27.865306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Primary observation on data-:\n\n<li>There are 1975 rows in non labelled data- which corresponds to 1975 patients each with unique study ID.\n<li>The image dir contains 1975 folder, that obviously correspond to each unique study ID.\n<li>Each patient image folder has three subfolder which correspond to the series IDs(T1 and T2 sagittal and T1 axial)\n<li>Each series folder has 20-50 images.\n<li>The labelled data tells us which of these images is having abnormality (series number and instance number) and what are coordinates of the anormality.\n<li>But labeled data do not have information as to how much is the degree of stenosis (category) and what is the file path of the labelled image.\n    ","metadata":{}},{"cell_type":"markdown","source":"# Startegy","metadata":{}},{"cell_type":"markdown","source":"> What have been done till now-:","metadata":{}},{"cell_type":"markdown","source":"We have already processed the labelled data in RSNA_data_handling where we did the following -:\n\n<li>We have removed the empty cells (7 patients effected).\n<li> We have created a unified dataset which contains category of diagnosis and also the serialised and pickled labelled image along with the study id. The dataset was called processed_train_data. We also labelencoded the cateogories. In addition we resizes the images and also normalized the images.The processed data was called processed_data.\n<li>This was then split into train_data and val_data, pickled and then zipped before\n    upoloading as privagte datasets. ","metadata":{}},{"cell_type":"markdown","source":"# Further strategy-:\nI think we should try and see the full image stack rather then just annotated and labelled images and see what insights we get.\nWe will try to achive this now.","metadata":{}},{"cell_type":"code","source":"# Create a df which contains-: all information of labelled_coordinates data but also contain the a column containing the path to image dir of the patient.\n# we start with train_labelled_coordinates_data:\n\ntrain_label_coordinates_data.shape\ndisplay(train_label_coordinates_data.head(2))","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:16:27.867833Z","iopub.execute_input":"2024-06-23T17:16:27.868219Z","iopub.status.idle":"2024-06-23T17:16:27.882747Z","shell.execute_reply.started":"2024-06-23T17:16:27.868178Z","shell.execute_reply":"2024-06-23T17:16:27.881567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#lets add series description to train_label_coordinates_data:\n\ntrain_temp_df=pd.merge(train_label_coordinates_data,train_series_description , on='series_id',how=\"left\")","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:16:27.884043Z","iopub.execute_input":"2024-06-23T17:16:27.884396Z","iopub.status.idle":"2024-06-23T17:16:27.913338Z","shell.execute_reply.started":"2024-06-23T17:16:27.884365Z","shell.execute_reply":"2024-06-23T17:16:27.911993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_temp_df.head(1))\n#You will notice pandas has duplicated study_id column and added suffices to it- this happened as there was same named column in the two df being merged.\n#Therefore, we will drop of this columna and rename the other back to origing.\n#Both below command can be run only once. So you wil have to uncomment it when the notebook started for the firstime.\n\ntrain_temp_df.drop('study_id_y',axis=1,inplace=True)\ntrain_temp_df.rename(columns={\"study_id_x\":\"study_id\"},inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:17:02.795526Z","iopub.execute_input":"2024-06-23T17:17:02.795969Z","iopub.status.idle":"2024-06-23T17:17:02.816326Z","shell.execute_reply.started":"2024-06-23T17:17:02.795935Z","shell.execute_reply":"2024-06-23T17:17:02.81504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_temp_df.head())","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:17:06.26353Z","iopub.execute_input":"2024-06-23T17:17:06.263952Z","iopub.status.idle":"2024-06-23T17:17:06.280577Z","shell.execute_reply.started":"2024-06-23T17:17:06.263916Z","shell.execute_reply":"2024-06-23T17:17:06.279366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Now lets add a column containing the path the image dir of the specific patient in the row-::\n\n\ndef write_folder_path(row):\n    return os.path.join(train_images_path,str(row['study_id']))\ntrain_temp_df['pt_image_dir_path']=train_temp_df.apply((lambda row: write_folder_path(row)),axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:17:08.852829Z","iopub.execute_input":"2024-06-23T17:17:08.853269Z","iopub.status.idle":"2024-06-23T17:17:09.647613Z","shell.execute_reply.started":"2024-06-23T17:17:08.853234Z","shell.execute_reply":"2024-06-23T17:17:09.646598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check:\ndisplay(train_temp_df.head(2))\nprint(train_temp_df.loc[0,'pt_image_dir_path'])","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:17:11.453385Z","iopub.execute_input":"2024-06-23T17:17:11.45379Z","iopub.status.idle":"2024-06-23T17:17:11.471238Z","shell.execute_reply.started":"2024-06-23T17:17:11.453761Z","shell.execute_reply":"2024-06-23T17:17:11.469993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Lets now add the path for the labelled images in each row. This we did before also but at that time we had instead replaced the column withb serialized pickled images,\n#This increased the dataset size immensely, this time we not direclty add images to dataset.\n#The label image is indicated by instance_number in the image stack.\ndef write_file_path(row,folder_col=\"pt_image_dir_path\",instance_col=\"instance_number\"):\n    pt_image_folder=row[folder_col]\n    series=str(row['series_id'])\n    image_number=row[instance_col]\n    label_file_path=os.path.join(pt_image_folder,series,f\"{image_number}.dcm\")\n    return label_file_path","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:17:13.955414Z","iopub.execute_input":"2024-06-23T17:17:13.955832Z","iopub.status.idle":"2024-06-23T17:17:13.962851Z","shell.execute_reply.started":"2024-06-23T17:17:13.955799Z","shell.execute_reply":"2024-06-23T17:17:13.961487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Aplly lmabda function-:\ntrain_temp_df['labelled_image_path']=train_temp_df.apply(lambda row: write_file_path(row,\"pt_image_dir_path\",\"instance_number\"),axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:17:16.045411Z","iopub.execute_input":"2024-06-23T17:17:16.045802Z","iopub.status.idle":"2024-06-23T17:17:17.399824Z","shell.execute_reply.started":"2024-06-23T17:17:16.045771Z","shell.execute_reply":"2024-06-23T17:17:17.398791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_temp_df.head(2))\nprint(train_temp_df.loc[0,'labelled_image_path'])","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:17:18.501793Z","iopub.execute_input":"2024-06-23T17:17:18.502239Z","iopub.status.idle":"2024-06-23T17:17:18.518737Z","shell.execute_reply.started":"2024-06-23T17:17:18.502205Z","shell.execute_reply":"2024-06-23T17:17:18.517502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#lets add the category/daignosis to this dataframe:\n# Add a new column to store the categories in the train_labels_coordinates_data dataset.\ntrain_temp_df[\"category\"] = None\n\n# Iterate through the rows of train_label_coordinates_data\nfor idx, row in train_temp_df.iterrows():\n    r = row['study_id']\n    col = (row['condition'].lower().replace(' ', '_')) + '_' + (row['level'].lower().replace('/', '_'))\n    \n    # Ensure the column exists in train_data to avoid KeyError\n    if col in train_data.columns:\n        # Check if there is a matching study_id and the column exists\n        value = train_data.loc[train_data['study_id'] == r, col].values\n        if len(value) > 0:\n            train_temp_df.at[idx, \"category\"] = value[0]\n        else:\n            train_temp_df.at[idx, \"category\"] = None\n    else:\n        train_temp_df.at[idx, \"category\"] = None\n","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:17:22.207947Z","iopub.execute_input":"2024-06-23T17:17:22.20842Z","iopub.status.idle":"2024-06-23T17:17:46.18381Z","shell.execute_reply.started":"2024-06-23T17:17:22.208385Z","shell.execute_reply":"2024-06-23T17:17:46.182695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display the and check first 10 rows of the updated DataFrame\ntrain_temp_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:17:46.186102Z","iopub.execute_input":"2024-06-23T17:17:46.186568Z","iopub.status.idle":"2024-06-23T17:17:46.203242Z","shell.execute_reply.started":"2024-06-23T17:17:46.186528Z","shell.execute_reply":"2024-06-23T17:17:46.202067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for missing values\ndisplay(train_temp_df.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:17:49.783273Z","iopub.execute_input":"2024-06-23T17:17:49.783676Z","iopub.status.idle":"2024-06-23T17:17:49.828753Z","shell.execute_reply.started":"2024-06-23T17:17:49.783644Z","shell.execute_reply":"2024-06-23T17:17:49.827561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display rows where the \"category\" column has missing values\nmissing_category_entries = train_temp_df[train_temp_df['category'].isnull()]\n\ndisplay(missing_category_entries.head(2))\nprint(f\"Number of rows with missing 'category' entries: {len(missing_category_entries)}\\n----\\n\")\n\n# Get unique study IDs where 'category' is null\nunique_study_ids_missing_category_entries = missing_category_entries['study_id'].unique()\n\n# Count the number of unique study IDs\nnum_unique_study_ids_missing_category_entries = len(unique_study_ids_missing_category_entries)\n\n# Print the number of unique study IDs with missing category entries\nprint(f\"Number of unique study IDs with missing 'category' entries: {num_unique_study_ids_missing_category_entries}\")\ndisplay(unique_study_ids_missing_category_entries)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:17:52.649632Z","iopub.execute_input":"2024-06-23T17:17:52.650065Z","iopub.status.idle":"2024-06-23T17:17:52.678764Z","shell.execute_reply.started":"2024-06-23T17:17:52.650026Z","shell.execute_reply":"2024-06-23T17:17:52.677423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> > We will loose only 7 patients if we drop those withb missing category enteries.So its worth dropping them.","metadata":{}},{"cell_type":"code","source":"# Drop rows with missing values without modifiying original df (hence inplace=False) and store this in new df names train:\ntrain_temp_df.dropna(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:17:55.707619Z","iopub.execute_input":"2024-06-23T17:17:55.708081Z","iopub.status.idle":"2024-06-23T17:17:55.763316Z","shell.execute_reply.started":"2024-06-23T17:17:55.708046Z","shell.execute_reply":"2024-06-23T17:17:55.762082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Confirm if there are still any missing valyes in new df:\nnum_of_missing_values=train_temp_df.isnull().sum()\nprint(num_of_missing_values)","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:17:57.398637Z","iopub.execute_input":"2024-06-23T17:17:57.399079Z","iopub.status.idle":"2024-06-23T17:17:57.440722Z","shell.execute_reply.started":"2024-06-23T17:17:57.399045Z","shell.execute_reply":"2024-06-23T17:17:57.439356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_temp_df.head())","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:40:48.859462Z","iopub.execute_input":"2024-06-23T17:40:48.859885Z","iopub.status.idle":"2024-06-23T17:40:48.879078Z","shell.execute_reply.started":"2024-06-23T17:40:48.859851Z","shell.execute_reply":"2024-06-23T17:40:48.877763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#making sure every numerical data is int:\ntrain_temp_df['series_id']=train_temp_df['series_id'].astype(int)\nprint(type(train_temp_df['series_id'][0]))\nprint(type(train_temp_df['study_id'][0]))\nprint(type(train_temp_df['instance_number'][0]))\nprint(type(train_temp_df['x'][0]))","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:43:33.76572Z","iopub.execute_input":"2024-06-23T17:43:33.766184Z","iopub.status.idle":"2024-06-23T17:43:33.78532Z","shell.execute_reply.started":"2024-06-23T17:43:33.76615Z","shell.execute_reply":"2024-06-23T17:43:33.784188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Viewing and analysing images- Advance view","metadata":{}},{"cell_type":"code","source":"import pydicom\nimport matplotlib.pyplot as plt\n\n# Function to display a DICOM image and its annotations as well its metadata ina df format:\ndef show_dicom_image(df, fpath_col, idx):\n    \"\"\"\n    Displays a DICOM image along with its annotations and metadata.\n\n    Parameters:\n    df (pd.DataFrame): The DataFrame containing the DICOM file paths and annotation data.\n    fpath_col (str): The column name in the DataFrame that contains the file paths to the DICOM files.\n    idx (int): The index of the row in the DataFrame to process.\n\n    This function performs the following steps:\n    1. Reads the DICOM file from the specified file path using pydicom.\n    2. Extracts the pixel array from the DICOM file to obtain the image.\n    3. Displays the image using matplotlib with a grayscale colormap.\n    4. Retrieves the x and y coordinates for annotation from the DataFrame.\n    5. Plots the coordinates as red semi-transparent circles on the image.\n    6. Extracts the metadata from the DICOM file and stores it in a dictionary, excluding the pixel data.\n    7. Displays the metadata and additional information (study ID, series description, level, and category) in the title of the plot.\n\n    Example usage:\n    show_dicom_image(dataframe, 'file_path_column', 0)\n    \"\"\"\n    # Read the DICOM file using pydicom\n    dicom = pydicom.dcmread(df.loc[idx, fpath_col])\n    \n    # Extract the pixel data to obtain the image\n    image = dicom.pixel_array\n    \n    # Display the image using matplotlib with a grayscale colormap\n    plt.imshow(image, cmap=plt.cm.gray)\n    \n    # Retrieve the x and y coordinates for annotation from the DataFrame\n    coordinates = {\n        \"x\": df.loc[idx, \"x\"],\n        \"y\": df.loc[idx, \"y\"]\n    }\n    \n    # Plot the coordinates as red semi-transparent circles on the image\n    plt.scatter(coordinates['x'], coordinates['y'], c='r', s=100, alpha=0.3)\n    \n    # Extract the metadata from the DICOM file and store it in a dictionary, excluding the pixel data\n    metadata_dict = {}\n    for atribute in dicom:\n        if atribute.name != \"Pixel Data\":  # Exclude pixel data due to its large size\n            metadata_dict[atribute.name] = atribute.value  # Store the metadata name and value in the dictionary\n    #Display metadata as df:\n    meta=pd.DataFrame(list(metadata_dict.items()),columns=[\"Metadata\",\"Value\"])\n    display(meta)\n   \n    # Display the metadata and additional information in the title of the plot\n    plt.title(f\"Patient ID: {df.loc[idx, 'study_id']}\\n\"\n              f\"{df.loc[idx, 'series_description']}\\n\"\n              f\"{df.loc[idx, 'level']}: {df.loc[idx, 'category']}\")\n    \n    # Show the plot\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:43:58.620103Z","iopub.execute_input":"2024-06-23T17:43:58.620493Z","iopub.status.idle":"2024-06-23T17:43:58.633293Z","shell.execute_reply.started":"2024-06-23T17:43:58.620462Z","shell.execute_reply":"2024-06-23T17:43:58.631602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read and display the DICOM image with annotations and metadata\nshow_dicom_image(train_temp_df,\"labelled_image_path\",1000)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:44:00.796873Z","iopub.execute_input":"2024-06-23T17:44:00.797298Z","iopub.status.idle":"2024-06-23T17:44:01.119955Z","shell.execute_reply.started":"2024-06-23T17:44:00.797261Z","shell.execute_reply":"2024-06-23T17:44:01.118644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_temp_df.head(25)","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:44:05.950031Z","iopub.execute_input":"2024-06-23T17:44:05.950454Z","iopub.status.idle":"2024-06-23T17:44:05.976755Z","shell.execute_reply.started":"2024-06-23T17:44:05.95042Z","shell.execute_reply":"2024-06-23T17:44:05.975579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Conclusions-Part 1:\n\nThus we succesfully -:\nCreated a df named train_temp_df which contains all the information needed in one place.\nWe also wrote a function that display the image with annoatation and also displays all the emtadata associated with the image ina dataframe format for easy visibilty. ","metadata":{}},{"cell_type":"markdown","source":"# Advanced interactive dicom viewer","metadata":{}},{"cell_type":"code","source":"import os\nimport pydicom\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport ipywidgets as widgets\nfrom IPython.display import display\n\n\n\ndef load_dicom_series():\n    \"\"\"\n    Load a DICOM series based on the study ID provided by the user.\n    \"\"\"\n    ptid = input('Please enter the ID of the study you want to load. It should be an integer: ')\n    if not ptid.isdigit():\n        print(\"Invalid input. Please enter a valid integer or 'e' to exit.\")\n        return\n\n    print(f\"Loading study for patient with ID: {ptid} .....\")\n    pt_folder = os.path.join(train_images_path, ptid)\n    \n    if not os.path.exists(pt_folder):\n        print(f\"Folder for patient ID {ptid} does not exist.\")\n        return\n    \n    s_ids = os.listdir(pt_folder)\n    s_paths = {s_id: os.path.join(pt_folder, s_id) for s_id in s_ids}\n    \n    sagt2_paths = []\n    sagt1_paths = []\n    axt2_paths = []\n    \n    for s_id, s_path in s_paths.items():\n        df = train_temp_df\n        s_id=int(s_id) # ensuring s_id are intiger to mach df types\n        s_description = df.loc[df['series_id'] == s_id, \"series_description\"].values\n        if s_description.size == 0:\n            print(f\"No description found for series ID: {s_id}\")\n            continue\n        s_description = s_description[0]\n        \n        image_paths = [os.path.join(s_path, image) for image in os.listdir(s_path)]\n        \n        if s_description == \"Sagittal T2/STIR\":\n            sagt2_paths.extend(image_paths)\n        elif s_description == \"Sagittal T1\":\n            sagt1_paths.extend(image_paths)\n        else:\n            axt2_paths.extend(image_paths)\n    \n    def plot_images(index):\n        fig, ax = plt.subplots(1, 3, figsize=(15, 5))\n\n        paths_list = [sagt2_paths, sagt1_paths, axt2_paths]\n        descriptions = [\"Sagittal T2\", \"Sagittal T1\", \"Axial T2\"]\n        \n        for position, (paths, desc) in enumerate(zip(paths_list, descriptions)):\n            if index >= len(paths):\n                continue\n            dicom = pydicom.dcmread(paths[index])\n            image = dicom.pixel_array\n            ax[position].imshow(image, cmap='gray')\n\n            slice_loc = getattr(dicom, 'SliceLocation', None)\n           \n            image_position_patient = getattr(dicom, 'ImagePositionPatient', [None, None, None])\n            \n            y_pos = image_position_patient[1] if desc in [\"Sagittal T2\", \"Sagittal T1\"] else None\n            \n            z_pos = image_position_patient[2] if desc == \"Axial T2\" else None\n            instance = getattr(dicom, 'InstanceNumber', None)\n\n            if instance is not None:\n                x_annot = df.loc[(df['study_id'] == int(ptid)) & (df['series_id'] == int(s_id)) & (df['instance_number'] == int(instance)), \"x\"].values\n                print(x_annot)\n                y_annot = df.loc[(df['study_id'] == ptid) & (df['series_id'] == s_id) & (df['instance_number'] == instance), \"y\"].values\n                level = df.loc[(df['study_id'] == ptid) & (df['series_id'] == s_id) & (df['instance_number'] == instance), \"level\"].values\n                severity = df.loc[(df['study_id'] == ptid) & (df['series_id'] == s_id) & (df['instance_number'] == instance), \"category\"].values\n                dx = df.loc[(df['study_id'] == ptid) & (df['series_id'] == s_id) & (df['instance_number'] == instance), \"condition\"].values\n\n                if all([x_annot.size, y_annot.size]):\n                    ax[position].scatter(x_annot[0], y_annot[0], c='r', s=100, alpha=0.4)\n                if y_pos is not None:\n                    ax[position].axhline(y=y_pos, color='b', linestyle='--', linewidth=1)\n                if z_pos is not None:\n                    ax[position].axhline(y=z_pos, color='g', linestyle='--', linewidth=1)\n                if slice_loc is not None:\n                    ax[position].legend([f\"Slice Location: {slice_loc}\"], loc='upper right')\n\n                if all([level.size, severity.size, dx.size]):\n                    title = (f\"Patient ID: {ptid}\\n\"\n                             f\"{desc}\\n\"\n                             f\"Showing: {severity[0]} {level[0]} {dx[0]}\")\n                    ax[position].set_title(title)\n        \n        plt.show()\n\n    index_slider = widgets.IntSlider(value=0, min=0, max=max(len(sagt2_paths), len(sagt1_paths), len(axt2_paths))-1, step=1, description='Image Index')\n    widgets.interact(plot_images, index=index_slider)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:51:20.312623Z","iopub.execute_input":"2024-06-23T17:51:20.313123Z","iopub.status.idle":"2024-06-23T17:51:20.341531Z","shell.execute_reply.started":"2024-06-23T17:51:20.313086Z","shell.execute_reply":"2024-06-23T17:51:20.340158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(type(train_temp_df['series_id'][0]))","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:44:23.946095Z","iopub.execute_input":"2024-06-23T17:44:23.94653Z","iopub.status.idle":"2024-06-23T17:44:23.953204Z","shell.execute_reply.started":"2024-06-23T17:44:23.946501Z","shell.execute_reply":"2024-06-23T17:44:23.951685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = train_temp_df.loc[(train_temp_df['study_id'] == 4003253) & (train_temp_df['series_id'] == 2448190387) & (train_temp_df['instance_number'] == 4), \"x\"].values\nprint(x)","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:44:44.965074Z","iopub.execute_input":"2024-06-23T17:44:44.965503Z","iopub.status.idle":"2024-06-23T17:44:44.974923Z","shell.execute_reply.started":"2024-06-23T17:44:44.96547Z","shell.execute_reply":"2024-06-23T17:44:44.973683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"load_dicom_series()","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:51:24.230489Z","iopub.execute_input":"2024-06-23T17:51:24.230918Z","iopub.status.idle":"2024-06-23T17:51:27.142232Z","shell.execute_reply.started":"2024-06-23T17:51:24.230884Z","shell.execute_reply":"2024-06-23T17:51:27.141071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"series_paths=os.listdir(dir)\n    files = [pydicom.dcmread(os.path.join(directory, f)) for f in os.listdir(directory)]\n    if orientation == 'sagittal':\n        files.sort(key=lambda x: x.ImagePositionPatient[1])  # Sort by y-coordinate\n    elif orientation == 'axial':\n        files.sort(key=lambda x: x.ImagePositionPatient[2])  # Sort by z-coordinate\n    slices = np.stack([f.pixel_array for f in files])\n    return slices, files\n\ndef dicom_viewer(df, ptid):\n    for folder in os.listdir(os.path.join(train_image_path,ptid)):\n        mask_sagt2 = (df['study_id'] == ptid) & (df['series_id'] == \"folder\")\n        mask_sagt1 = (df['study_id'] == ptid) & (df['series_description'] == \"Sagittal T1\")\n    mask_axt2 = (df['study_id'] == ptid) & (df['series_description'] == \"Axial T2\")\n    if main_image_folder=df.loc[mask_sagt2, \"image_dir\"].values[0]\n    seriesid=df.loc[mask_sagt2, \"series_id\"].values[0]\n    sagt2_folder = os.path.join(main_image_folder,seriesid)\n    sagt1_folder = df.loc[mask_sagt1, \"image_dir\"].values[0]\n    axt2_folder = df.loc[mask_axt2, \"image_dir\"].values[0]\n    \n    sag1_slices, sag1_files = load_dicom_series(sagt1_folder, 'sagittal')\n    sag2_slices, sag2_files = load_dicom_series(sagt2_folder, 'sagittal')\n    axial_slices, axial_files = load_dicom_series(axt2_folder, 'axial')\n    \n    def show_mri_series(slice_index):\n        fig, axes = plt.subplots(1, 3, figsize=(15, 5))\n        \n        axes[0].imshow(sag1_slices[slice_index], cmap='gray')\n        axes[0].set_title('Sagittal T1')\n        \n        axes[1].imshow(sag2_slices[slice_index], cmap='gray')\n        axes[1].set_title('Sagittal T2/STIR')\n        \n        axes[2].imshow(axial_slices[slice_index], cmap='gray')\n        axes[2].set_title('Axial T2')\n        \n        z_coord = axial_files[slice_index].ImagePositionPatient[2]\n        y_coord = sag1_files[slice_index].ImagePositionPatient[1]\n        \n        for ax in axes[:2]:\n            ax.axhline(y=y_coord, color='r')\n        axes[2].axvline(x=z_coord, color='r')\n        \n        plt.show()\n    \n    interact(show_mri_series, slice_index=IntSlider(min=0, max=len(sag1_slices)-1, step=1, value=0))\n\n# Example usage\ndf = train_temp_df\nptid = 4003253\ndicom_viewer(df, ptid)","metadata":{"execution":{"iopub.status.busy":"2024-06-23T17:16:30.503043Z","iopub.status.idle":"2024-06-23T17:16:30.503564Z","shell.execute_reply.started":"2024-06-23T17:16:30.503297Z","shell.execute_reply":"2024-06-23T17:16:30.503319Z"},"trusted":true},"execution_count":null,"outputs":[]}]}