{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport cv2\nimport pydicom\nimport numpy as np\nimport os\nimport glob\nfrom tqdm import tqdm\nimport warnings","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train.csv')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Total Cases: \", len(train))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.columns","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Some of the diagnoses, especially subarticular stenosis category has some missing data. This is due to the fact that some of the images do not have those regions visualized (more specifically the most superior vertebral bodies were less likely to make it into the imaging field)","metadata":{}},{"cell_type":"code","source":"figure, axis = plt.subplots(1,3, figsize=(20,5)) \nfor idx, d in enumerate(['foraminal', 'subarticular', 'canal']):\n    diagnosis = list(filter(lambda x: x.find(d) > -1, train.columns))\n    dff = train[diagnosis]\n    with warnings.catch_warnings():\n        warnings.simplefilter(action='ignore', category=FutureWarning)\n        value_counts = dff.apply(pd.value_counts).fillna(0).T\n    value_counts.plot(kind='bar', stacked=True, ax=axis[idx])\n    axis[idx].set_title(f'{d} distribution')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"As expected, most patients have normal/mild or moderate grade for every diagonistic category. ","metadata":{}},{"cell_type":"markdown","source":"# Loading the image\n","metadata":{}},{"cell_type":"code","source":"part_1 = os.listdir('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images')\npart_1 = list(filter(lambda x: x.find('.DS') == -1, part_1))\ndf_meta_f = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_series_descriptions.csv')\np1 = [(x, f\"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images/{x}\") for x in part_1]\nmeta_obj = { p[0]: { 'folder_path': p[1], \n                    'SeriesInstanceUIDs': [] \n                   } \n            for p in p1 }\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for m in meta_obj:\n    meta_obj[m]['SeriesInstanceUIDs'] = list(\n        filter(lambda x: x.find('.DS') == -1, \n               os.listdir(meta_obj[m]['folder_path'])\n              )\n    )\n# grabs the correspoding series descriptions\nfor k in tqdm(meta_obj):\n    for s in meta_obj[k]['SeriesInstanceUIDs']:\n        if 'SeriesDescriptions' not in meta_obj[k]:\n            meta_obj[k]['SeriesDescriptions'] = []\n        try:\n            meta_obj[k]['SeriesDescriptions'].append(\n                df_meta_f[(df_meta_f['study_id'] == int(k)) & \n                (df_meta_f['series_id'] == int(s))]['series_description'].iloc[0])\n        except:\n            print(\"Failed on\", s, k)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta_obj[list(meta_obj.keys())[1]]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"patient = train.iloc[1]\nptobj = meta_obj[str(patient['study_id'])]\nprint(ptobj)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get data into the format\n\"\"\"\nim_list_dcm = {\n    '{SeriesInstanceUID}': {\n        'images': [\n            {'SOPInstanceUID': ...,\n             'dicom': PyDicom object\n            },\n            ...,\n        ],\n        'description': # SeriesDescription\n    },\n    ...\n}\n\"\"\"\nim_list_dcm = {}\nfor idx, i in enumerate(ptobj['SeriesInstanceUIDs']):\n    im_list_dcm[i] = {'images': [], 'description': ptobj['SeriesDescriptions'][idx]}\n    images = glob.glob(f\"{ptobj['folder_path']}/{ptobj['SeriesInstanceUIDs'][idx]}/*.dcm\")\n    for j in sorted(images, key=lambda x: int(x.split('/')[-1].replace('.dcm', ''))):\n        im_list_dcm[i]['images'].append({\n            'SOPInstanceUID': j.split('/')[-1].replace('.dcm', ''), \n            'dicom': pydicom.dcmread(j) })\n# Function to display images","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to display images\ndef display_images(images, title, max_images_per_row=4):\n    # Calculate the number of rows needed\n    num_images = len(images)\n    num_rows = (num_images + max_images_per_row - 1) // max_images_per_row  # Ceiling division\n\n    # Create a subplot grid\n    fig, axes = plt.subplots(num_rows, max_images_per_row, figsize=(5, 1.5 * num_rows))\n    \n    # Flatten axes array for easier looping if there are multiple rows\n    if num_rows > 1:\n        axes = axes.flatten()\n    else:\n        axes = [axes]  # Make it iterable for consistency\n\n    # Plot each image\n    for idx, image in enumerate(images):\n        ax = axes[idx]\n        ax.imshow(image, cmap='gray')  # Assuming grayscale for simplicity, change cmap as needed\n        ax.axis('off')  # Hide axes\n\n    # Turn off unused subplots\n    for idx in range(num_images, len(axes)):\n        axes[idx].axis('off')\n    fig.suptitle(title, fontsize=16)\n\n    plt.tight_layout()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in im_list_dcm:\n    display_images([x['dicom'].pixel_array for x in im_list_dcm[i]['images']], \n                   im_list_dcm[i]['description'])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Looking at the co-ordinate of the pathologies","metadata":{}},{"cell_type":"code","source":"df_coor = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_coor.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def display_coor_on_img(c, i, title):\n    center_coordinates = (int(c['x']), int(c['y']))\n    radius = 10\n    color = (255, 0, 0)  # Red color in BGR\n    thickness = 2\n    IMG = i['dicom'].pixel_array\n    IMG_normalized = cv2.normalize(IMG, None, alpha=0, beta=255, norm_type=cv2.NORM_MINMAX, dtype=cv2.CV_8U)\n    \n    IMG_with_circle = cv2.circle(IMG_normalized.copy(), center_coordinates, radius, color, thickness)\n    \n    # Convert the image from BGR to RGB for correct color display in matplotlib\n    IMG_with_circle = cv2.cvtColor(IMG_with_circle, cv2.COLOR_BGR2RGB)\n    \n    # Display the image\n    plt.imshow(IMG_with_circle)\n    plt.axis('off')  # Turn off axis numbers and ticks\n    plt.title(title)\n    plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"coor_entries = df_coor[df_coor['study_id'] == int(patient['study_id'])]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Only showing severe cases for this patient\")\nfor idc, c in coor_entries.iterrows():\n    for i in im_list_dcm[str(c['series_id'])]['images']:\n        if int(i['SOPInstanceUID']) == int(c['instance_number']):\n            try:\n                patient_severity = patient[\n                    f\"{c['condition'].lower().replace(' ', '_')}_{c['level'].lower().replace('/', '_')}\"\n                ]\n            except Exception as e:\n                patient_severity = \"unknown severity\"\n            title = f\"{i['SOPInstanceUID']} \\n{c['level']}, {c['condition']}: {patient_severity} \\n{c['x']}, {c['y']}\"\n            if patient_severity == 'Severe':\n                display_coor_on_img(c, i, title)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}