{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt # for visualization\nimport seaborn as sns\nimport warnings\n\nimport pydicom # library for handling images in dicom format\nimport cv2 # Open CV library for image handling\n\nfrom tqdm import tqdm # Library to monitor the progress of code execution\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#   for filename in filenames:\n#       print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\nsns.set() # to use the beautification of seaborn by default","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-10T16:35:54.321003Z","iopub.execute_input":"2024-06-10T16:35:54.321502Z","iopub.status.idle":"2024-06-10T16:35:56.285745Z","shell.execute_reply.started":"2024-06-10T16:35:54.321464Z","shell.execute_reply":"2024-06-10T16:35:56.284389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/'\ntrain_df = pd.read_csv(os.path.join(ROOT, 'train.csv'))\ntrain_label_coord_df = pd.read_csv(os.path.join(ROOT, 'train_label_coordinates.csv'))\ntrain_series_desc_df = pd.read_csv(os.path.join(ROOT, 'train_series_descriptions.csv'))\n\nprint(f\"Shape of train_df : {train_df.shape}\")\nprint(f\"Shape of train_label_coord_df : {train_label_coord_df.shape}\")\nprint(f\"Shape of train_series_desc_df : {train_series_desc_df.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-10T16:35:56.290052Z","iopub.execute_input":"2024-06-10T16:35:56.291433Z","iopub.status.idle":"2024-06-10T16:35:56.490551Z","shell.execute_reply.started":"2024-06-10T16:35:56.291394Z","shell.execute_reply":"2024-06-10T16:35:56.489367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Description","metadata":{}},{"cell_type":"code","source":"train_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-10T16:35:56.492245Z","iopub.execute_input":"2024-06-10T16:35:56.493351Z","iopub.status.idle":"2024-06-10T16:35:56.538607Z","shell.execute_reply.started":"2024-06-10T16:35:56.493309Z","shell.execute_reply":"2024-06-10T16:35:56.537365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_label_coord_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-10T16:35:56.541808Z","iopub.execute_input":"2024-06-10T16:35:56.542196Z","iopub.status.idle":"2024-06-10T16:35:56.558693Z","shell.execute_reply.started":"2024-06-10T16:35:56.542149Z","shell.execute_reply":"2024-06-10T16:35:56.557195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check data for a single study\ntrain_label_coord_df[train_label_coord_df['study_id'] == train_df['study_id'][0]] # first study","metadata":{"execution":{"iopub.status.busy":"2024-06-10T16:35:56.560357Z","iopub.execute_input":"2024-06-10T16:35:56.560822Z","iopub.status.idle":"2024-06-10T16:35:56.590802Z","shell.execute_reply.started":"2024-06-10T16:35:56.560782Z","shell.execute_reply":"2024-06-10T16:35:56.589350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In terms of background. Each patient has a Study (called a StudyInstanceUID). That study has multiple series called a (SeriesInstanceUID). Within a series you have multiple images with a unique SOPInstanceUID.\n\nAll of this data is tied together via the csv from earlier that listed diagnoses types, but also a separate dicom metadata file that contains information about the particular series descriptions (containing useful series descriptions such as whether the image is sagital or axial and t1 vs t2). Though the names for the series descriptions are not standardized.","metadata":{}},{"cell_type":"code","source":"train_series_desc_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-10T16:35:56.592489Z","iopub.execute_input":"2024-06-10T16:35:56.595550Z","iopub.status.idle":"2024-06-10T16:35:56.609139Z","shell.execute_reply.started":"2024-06-10T16:35:56.595469Z","shell.execute_reply":"2024-06-10T16:35:56.607757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data for a single study\ntrain_series_desc_df[train_series_desc_df['study_id'] == train_df['study_id'][0]] # first study","metadata":{"execution":{"iopub.status.busy":"2024-06-10T16:35:56.610701Z","iopub.execute_input":"2024-06-10T16:35:56.611173Z","iopub.status.idle":"2024-06-10T16:35:56.626583Z","shell.execute_reply.started":"2024-06-10T16:35:56.611135Z","shell.execute_reply":"2024-06-10T16:35:56.625392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualize the Study Data\n- How many abnormalities?\n- Which are the most prevalant abnormalities?","metadata":{}},{"cell_type":"code","source":"# copied from https://www.kaggle.com/code/abhinavsuri/anatomy-image-visualization-overview-rsna-raids\nfigure, axis = plt.subplots(1,3, figsize=(20,5)) \nfor idx, d in enumerate(['foraminal', 'subarticular', 'canal']):\n    diagnosis = list(filter(lambda x: x.find(d) > -1, train_df.columns))\n    dff = train_df[diagnosis]\n    with warnings.catch_warnings():\n        warnings.simplefilter(action='ignore', category=FutureWarning)\n        value_counts = dff.apply(pd.value_counts).fillna(0).T\n    value_counts.plot(kind='bar', stacked=True, ax=axis[idx])\n    axis[idx].set_title(f'{d} distribution')","metadata":{"execution":{"iopub.status.busy":"2024-06-10T16:35:56.628040Z","iopub.execute_input":"2024-06-10T16:35:56.628522Z","iopub.status.idle":"2024-06-10T16:35:58.409799Z","shell.execute_reply.started":"2024-06-10T16:35:56.628491Z","shell.execute_reply":"2024-06-10T16:35:58.408573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Display Image\nWe shall try to grab the images from data set for a given patient and display it. \n- Get the study ID\n- Get the Series IDs for the given study\n- For each series get the available images\n- Display the images in appropriate format","metadata":{}},{"cell_type":"code","source":"list(train_series_desc_df[train_series_desc_df['study_id'] == 4003253]['series_id']) # first study)","metadata":{"execution":{"iopub.status.busy":"2024-06-10T16:35:58.411060Z","iopub.execute_input":"2024-06-10T16:35:58.411402Z","iopub.status.idle":"2024-06-10T16:35:58.420665Z","shell.execute_reply.started":"2024-06-10T16:35:58.411373Z","shell.execute_reply":"2024-06-10T16:35:58.419459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for (path, dirnames, filenames) in os.walk(os.path.join(ROOT, 'train_images', str(100206310), str(1012284084))):\n    print(filenames)","metadata":{"execution":{"iopub.status.busy":"2024-06-10T16:35:58.424646Z","iopub.execute_input":"2024-06-10T16:35:58.425433Z","iopub.status.idle":"2024-06-10T16:35:58.464099Z","shell.execute_reply.started":"2024-06-10T16:35:58.425388Z","shell.execute_reply":"2024-06-10T16:35:58.462887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs = next(os.walk(os.path.join(ROOT, 'train_images', str(100206310), str(1012284084))))[2]\nfor img in imgs:\n    print(img)","metadata":{"execution":{"iopub.status.busy":"2024-06-10T16:35:58.465641Z","iopub.execute_input":"2024-06-10T16:35:58.465979Z","iopub.status.idle":"2024-06-10T16:35:58.473820Z","shell.execute_reply.started":"2024-06-10T16:35:58.465952Z","shell.execute_reply":"2024-06-10T16:35:58.472532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def search_files(directory = None):\n    \"\"\"\n    Method to search the files in a given directory.\n    \"\"\"\n    assert os.path.isdir(directory)\n    #walk_iterator = next(os.walk(directory)) # walk in the current directory\n    imgs = next(os.walk(os.path.join(directory)))[2]\n    file_paths = []  # Initialize list to store file paths\n    for img in imgs:\n        file_paths.append(os.path.join(directory, img))  # Append full path of each file to list\n    return file_paths  # Return list of full paths of all files","metadata":{"execution":{"iopub.status.busy":"2024-06-10T16:35:58.475561Z","iopub.execute_input":"2024-06-10T16:35:58.475973Z","iopub.status.idle":"2024-06-10T16:35:58.484551Z","shell.execute_reply.started":"2024-06-10T16:35:58.475943Z","shell.execute_reply":"2024-06-10T16:35:58.483379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_images_paths(study_id: int):\n    series_ids = list(train_series_desc_df[train_series_desc_df['study_id'] == study_id]['series_id']) # Get the series ids for the study\n    \n    images = {} # dictionary to store image ids for each series\n    for series in series_ids:\n        images_dir = os.path.join(ROOT, 'train_images', str(study_id), str(series))\n        images[series] = search_files(images_dir)\n        \n    return images","metadata":{"execution":{"iopub.status.busy":"2024-06-10T16:35:58.486046Z","iopub.execute_input":"2024-06-10T16:35:58.486490Z","iopub.status.idle":"2024-06-10T16:35:58.498434Z","shell.execute_reply.started":"2024-06-10T16:35:58.486453Z","shell.execute_reply":"2024-06-10T16:35:58.497038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_images(study_id):\n    \"\"\"\n    Display the images from obtained paths in appropriate format.\n    \"\"\"\n    image_paths = get_images_paths(study_id)\n    ncols = 6\n    for series in image_paths.keys():\n        instances = train_label_coord_df[train_label_coord_df['series_id'] == series][['instance_number', 'x', 'y']]\n        fig = plt.figure(figsize=(15, 15))\n        plt.suptitle(f\"IMAGES OF SERIES {series}\")\n        nrows = (len(image_paths[series]) + ncols - 1) // ncols  # Ensure enough rows to cover all images\n        for i, img_path in enumerate(image_paths[series]):\n            ds = pydicom.dcmread(img_path)\n            ax = fig.add_subplot(nrows, ncols, i + 1)\n            ax.imshow(ds.pixel_array, cmap=plt.cm.gray)\n            ax.set_title(f\"Instance {i + 1}\")\n            ax.set_xticks([])\n            ax.set_yticks([])\n            # Plot the pathology on the image\n            instance = instances[instances['instance_number'] == i + 1]\n            if not instance.empty:\n                ax.scatter(x=instance['x'], y=instance['y'], facecolors='none', edgecolors='red', s=100, linewidths=1.5)\n            ax.grid(False)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-10T17:39:11.989071Z","iopub.execute_input":"2024-06-10T17:39:11.989608Z","iopub.status.idle":"2024-06-10T17:39:12.003324Z","shell.execute_reply.started":"2024-06-10T17:39:11.989574Z","shell.execute_reply":"2024-06-10T17:39:12.001286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_images(4003253)","metadata":{"execution":{"iopub.status.busy":"2024-06-10T17:39:12.600266Z","iopub.execute_input":"2024-06-10T17:39:12.600725Z","iopub.status.idle":"2024-06-10T17:39:20.106624Z","shell.execute_reply.started":"2024-06-10T17:39:12.600691Z","shell.execute_reply":"2024-06-10T17:39:20.105487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Number of Classes\n25 classes for each of the abnormality, and one class for normal. \n$$\n    Number of Classes = 25 \\times 2 + 1 = 51\n$$","metadata":{}},{"cell_type":"code","source":"conditions = list(train_df.columns)\nconditions.pop(0)\nconditions","metadata":{"execution":{"iopub.status.busy":"2024-06-10T16:42:29.241266Z","iopub.execute_input":"2024-06-10T16:42:29.241732Z","iopub.status.idle":"2024-06-10T16:42:29.252244Z","shell.execute_reply.started":"2024-06-10T16:42:29.241700Z","shell.execute_reply":"2024-06-10T16:42:29.250397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate classes from column names\nclasses = [f'{condition}_{severity}' for condition in conditions for severity in ('moderate', 'severe')]\nclasses.append('mild_normal')\nclasses","metadata":{"execution":{"iopub.status.busy":"2024-06-10T16:47:07.180200Z","iopub.execute_input":"2024-06-10T16:47:07.180705Z","iopub.status.idle":"2024-06-10T16:47:07.192385Z","shell.execute_reply.started":"2024-06-10T16:47:07.180670Z","shell.execute_reply":"2024-06-10T16:47:07.190296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(classes)","metadata":{"execution":{"iopub.status.busy":"2024-06-10T16:46:49.145267Z","iopub.execute_input":"2024-06-10T16:46:49.145944Z","iopub.status.idle":"2024-06-10T16:46:49.155958Z","shell.execute_reply.started":"2024-06-10T16:46:49.145896Z","shell.execute_reply":"2024-06-10T16:46:49.154538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}