{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"},{"sourceId":9203608,"sourceType":"datasetVersion","datasetId":5564657}],"dockerImageVersionId":30746,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# WORKING WITH DICOM FILE - LUMBAR SPINE DATA","metadata":{}},{"cell_type":"markdown","source":"## Imports\n* pathlib for easy path handling\n* pydicom to handle dicom files\n* matplotlib for visualization\n* numpy to create the 3D container","metadata":{}},{"cell_type":"code","source":"from pathlib import Path\nimport pydicom\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:09:54.449821Z","iopub.execute_input":"2024-08-19T13:09:54.450112Z","iopub.status.idle":"2024-08-19T13:09:54.637327Z","shell.execute_reply.started":"2024-08-19T13:09:54.450087Z","shell.execute_reply":"2024-08-19T13:09:54.636568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"At first we try to read a single dcm file <br />\nTo this end we use the **read_file(path)** function provided by pydicom","metadata":{}},{"cell_type":"code","source":"dicom_file = pydicom.read_file('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images/100206310/1012284084/1.dcm')","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:09:59.734921Z","iopub.execute_input":"2024-08-19T13:09:59.735703Z","iopub.status.idle":"2024-08-19T13:09:59.748711Z","shell.execute_reply.started":"2024-08-19T13:09:59.735671Z","shell.execute_reply":"2024-08-19T13:09:59.747809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's take a look what this file contains. <br />\nYou can *print* the dicom file to get a bulk of information, containing e.g the company which built the scanner (RSNA in this case), the shape of the image (*Rows, Columns*, 320x320 in this case), table height all information about the patient (of course the personal information is anonymized here), and of large importance, the **image orientation**\n\nFeel free to scroll through the available information","metadata":{}},{"cell_type":"code","source":"print(dicom_file)","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:10:01.733574Z","iopub.execute_input":"2024-08-19T13:10:01.734201Z","iopub.status.idle":"2024-08-19T13:10:01.740454Z","shell.execute_reply.started":"2024-08-19T13:10:01.734168Z","shell.execute_reply":"2024-08-19T13:10:01.739572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Accessing DICOM **header** information:\nYou can access the dicom tags by using the hexadecimal encoded identifiers at the start of each line.\nAs an example, if you want to get the shape of the image you can use those two identifiers\n\n* (0028, 0010) Rows\n* (0028, 0011) Columns\n\nThe 0x in front of the identifier tells the python interpreter that it should interpret this value as hexadecimal","metadata":{}},{"cell_type":"code","source":"print(dicom_file[0x0028, 0x0010])\nprint(dicom_file[0x0028, 0x0011])","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:10:04.063649Z","iopub.execute_input":"2024-08-19T13:10:04.064001Z","iopub.status.idle":"2024-08-19T13:10:04.068888Z","shell.execute_reply.started":"2024-08-19T13:10:04.063975Z","shell.execute_reply":"2024-08-19T13:10:04.068026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There is an alternative, more direct way to access the values of the DICOM header tags using the tag descriptions\n    <br /> ","metadata":{}},{"cell_type":"code","source":"print('Rows: ', dicom_file.Rows)\nprint('Columns: ', dicom_file.Columns)","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:10:07.389736Z","iopub.execute_input":"2024-08-19T13:10:07.390345Z","iopub.status.idle":"2024-08-19T13:10:07.395298Z","shell.execute_reply.started":"2024-08-19T13:10:07.390316Z","shell.execute_reply":"2024-08-19T13:10:07.394440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Accessing DICOM **body** information (the actual image):\n\nThe **pixel_array** contains the actual image data:","metadata":{}},{"cell_type":"code","source":"dicom_file = pydicom.read_file('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images/100206310/1012284084/1.dcm')\n\n# Extract image from dcm file\nct = dicom_file.pixel_array\nplt.figure()\nplt.imshow(ct)","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:10:10.997395Z","iopub.execute_input":"2024-08-19T13:10:10.997801Z","iopub.status.idle":"2024-08-19T13:10:11.357937Z","shell.execute_reply.started":"2024-08-19T13:10:10.997768Z","shell.execute_reply":"2024-08-19T13:10:11.357055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUG_PROB = 0.75\nIMG_SIZE = [512, 512]\nimport albumentations as A\n\ntransforms_train = A.Compose([\n    A.RandomBrightnessContrast(brightness_limit=(-0.3, 0.3), contrast_limit=(-0.3, 0.3), p=AUG_PROB),\n    A.OneOf([\n        A.MotionBlur(blur_limit=7),\n        A.MedianBlur(blur_limit=7),\n        A.GaussianBlur(blur_limit=7),\n        A.GaussNoise(var_limit=(5.0, 40.0)),\n    ], p=AUG_PROB),\n\n    A.OneOf([\n        A.OpticalDistortion(distort_limit=0.6),\n        A.GridDistortion(num_steps=5, distort_limit=1.),\n        A.ElasticTransform(alpha=4),\n    ], p=AUG_PROB),\n\n    A.ShiftScaleRotate(shift_limit=0.1, scale_limit=0.1, rotate_limit=15, border_mode=0, p=AUG_PROB),\n    A.Resize(IMG_SIZE[0], IMG_SIZE[1]),\n    A.CoarseDropout(max_holes=100, max_height=50, max_width=50, min_holes=1, min_height=8, min_width=8, p=AUG_PROB),  \n    A.Normalize(mean=0.5, std=0.5)\n])\n\ntransforms_val = A.Compose([\n    A.Resize(IMG_SIZE[0], IMG_SIZE[1]),\n    A.Normalize(mean=0.5, std=0.5)\n])","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:10:12.910740Z","iopub.execute_input":"2024-08-19T13:10:12.911098Z","iopub.status.idle":"2024-08-19T13:10:13.837310Z","shell.execute_reply.started":"2024-08-19T13:10:12.911071Z","shell.execute_reply":"2024-08-19T13:10:13.836353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom\nimport matplotlib.pyplot as plt\nimport albumentations as A\nimport cv2\nimport numpy as np\n\n# Load the DICOM file\ndicom_file = pydicom.dcmread('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images/2386338288/3384576679/8.dcm')\n# Extract image from DICOM file\nct = dicom_file.pixel_array\n# Normalize the image\nct = ct / ct.max()\nct = (ct * 255).astype(np.uint8)\n\n\n\n\n# Apply the augmentation\naugmented = transforms_train(image=ct)\naugmented_ct = augmented['image']\n# Display the original and augmented images\nfig, axs = plt.subplots(1, 2, figsize=(12, 6))\naxs[0].imshow(ct)\naxs[0].set_title('Original CT Image')\naxs[0].axis('off')\naxs[1].imshow(augmented_ct)\naxs[1].set_title('Augmented CT Image')\naxs[1].axis('off')\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:10:14.124883Z","iopub.execute_input":"2024-08-19T13:10:14.125339Z","iopub.status.idle":"2024-08-19T13:10:14.666386Z","shell.execute_reply.started":"2024-08-19T13:10:14.125308Z","shell.execute_reply":"2024-08-19T13:10:14.665445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can perform a quick sanity check and make sure that the image shape corresponds to the Rows and Columns we saw earlier in the header information","metadata":{}},{"cell_type":"code","source":"print(ct.shape)","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:10:17.478463Z","iopub.execute_input":"2024-08-19T13:10:17.479303Z","iopub.status.idle":"2024-08-19T13:10:17.483534Z","shell.execute_reply.started":"2024-08-19T13:10:17.479269Z","shell.execute_reply":"2024-08-19T13:10:17.482710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visuals","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport pydicom as dicom\nimport matplotlib.patches as patches\n\n# read data\npath = '../input/rsna-2024-lumbar-spine-degenerative-classification/'\ndf_train_main  = pd.read_csv(path + 'train.csv')\ndf_train_label = pd.read_csv(path + 'train_label_coordinates.csv')\ndf_train_desc  = pd.read_csv(path + 'train_series_descriptions.csv')\ndf_test_desc   = pd.read_csv(path + 'test_series_descriptions.csv')\ndf_sub         = pd.read_csv(path + 'sample_submission.csv')\n# join first two tables\ndf_train_step_1 = pd.merge(left=df_train_label, right=df_train_main, how='left', on='study_id').reset_index(drop=True)\n# join with third table\ndf_train = pd.merge(left=df_train_step_1, right=df_train_desc, how='left', on=['study_id', 'series_id']).reset_index(drop=True)\n# convert study_id's to categorical\ndf_train.study_id = df_train.study_id.astype('category')\ndf_train.series_id = df_train.series_id.astype('category')\n\ntrain_label_coordinates=df_train_label\ntrain_label_coordinates['series_description'] = df_train.series_description","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:10:20.317967Z","iopub.execute_input":"2024-08-19T13:10:20.318808Z","iopub.status.idle":"2024-08-19T13:10:21.038898Z","shell.execute_reply.started":"2024-08-19T13:10:20.318775Z","shell.execute_reply":"2024-08-19T13:10:21.037577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:10:24.178083Z","iopub.execute_input":"2024-08-19T13:10:24.179020Z","iopub.status.idle":"2024-08-19T13:10:24.203073Z","shell.execute_reply.started":"2024-08-19T13:10:24.178987Z","shell.execute_reply":"2024-08-19T13:10:24.202271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport pydicom as dicom\nimport matplotlib.patches as patches\n\n# read data\npath = '../input/rsna-2024-lumbar-spine-degenerative-classification/'\ndf_train_main  = pd.read_csv(path + 'train.csv')\ndf_train_label = pd.read_csv(path + 'train_label_coordinates.csv')\ndf_train_desc  = pd.read_csv(path + 'train_series_descriptions.csv')\ndf_test_desc   = pd.read_csv(path + 'test_series_descriptions.csv')\ndf_sub         = pd.read_csv(path + 'sample_submission.csv')\n# join first two tables\ndf_train_step_1 = pd.merge(left=df_train_label, right=df_train_main, how='left', on='study_id').reset_index(drop=True)\n# join with third table\ndf_train = pd.merge(left=df_train_step_1, right=df_train_desc, how='left', on=['study_id', 'series_id']).reset_index(drop=True)\n# convert study_id's to categorical\ndf_train.study_id = df_train.study_id.astype('category')\ndf_train.series_id = df_train.series_id.astype('category')\n\ntrain_label_coordinates=df_train_label\ntrain_label_coordinates['series_description'] = df_train.series_description\n\n\ndef mrt(id, ser, inst):\n    lag=20\n    path2 = path+'train_images/' + str(id) +'/' + str(ser)+'/' + str(inst) + '.dcm'\n\n    ds = dicom.dcmread(path2)\n    fig, ax = plt.subplots(figsize=(16, 8))\n    from matplotlib.colors import LogNorm \n\n    ax.imshow(ds.pixel_array)     # Display the image\n\n    # Create a legend\n#     legend_elements = []\n\n    # Plot the coordinates for the current condition\n    ab = train_label_coordinates[(train_label_coordinates.study_id==id) & \n                                          (train_label_coordinates.instance_number==inst)&\n                                         (train_label_coordinates.series_id==ser)]\n\n    a = 25 * max(ds.pixel_array.shape)/512\n    for _, row in ab.iterrows():\n        x, y = row['x'], row['y']\n\n        rect2 = patches.Rectangle((x - a, y - a), 2*a, 2*a, linewidth=1, edgecolor='white', facecolor='none')\n        rect1 = patches.Rectangle((x - a, y - a), 2*a, 2*a, linewidth=1, facecolor='none', alpha = 0.25)\n\n        ax.add_patch(rect2)\n        ax.add_patch(rect1)\n\n        # Add the condition to the legend\n#         legend_elements.append(patches.Patch(facecolor='none', edgecolor='r', ))\n\n    # Add title\n#     title = f\"{ab.series_description.unique()}, Study: {id}, Series: {ser}, Instance: {inst}\"\n#     ax.set_title(title, fontsize=20)\n\n   # Display additional columns:\n    for _, row in ab.iterrows():\n        text = f\" {row['level']}, {row['condition']}\"\n        ax.text(row['x'] + lag, row['y']+np.random.randint(-15, 15), text, fontsize=10, color='white', verticalalignment='center_baseline')\n    \n    plt.show() ","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:10:25.558309Z","iopub.execute_input":"2024-08-19T13:10:25.559123Z","iopub.status.idle":"2024-08-19T13:10:25.797766Z","shell.execute_reply.started":"2024-08-19T13:10:25.559091Z","shell.execute_reply":"2024-08-19T13:10:25.796955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for idx, row in df_train.iterrows(): \n    id, ser, inst = row['study_id'], row['series_id'], row['instance_number']\n    mrt(id, ser, inst)\n    path+'train_images/' + str(id) +'/' + str(ser)+'/' + str(inst) + '.dcm'\n    break","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:10:27.119505Z","iopub.execute_input":"2024-08-19T13:10:27.120159Z","iopub.status.idle":"2024-08-19T13:10:27.594007Z","shell.execute_reply.started":"2024-08-19T13:10:27.120130Z","shell.execute_reply":"2024-08-19T13:10:27.593158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom as dicom\nimport matplotlib.patches as patches\n\n#Case #1\nid = 100206310\nser= 2092806862\ninst=12\n\nmrt(id, ser, inst)","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:10:30.713469Z","iopub.execute_input":"2024-08-19T13:10:30.714271Z","iopub.status.idle":"2024-08-19T13:10:31.158906Z","shell.execute_reply.started":"2024-08-19T13:10:30.714233Z","shell.execute_reply":"2024-08-19T13:10:31.157990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Case #2\nid = 4290709089\nser= 4237840455\ninst=11\n\nmrt(id, ser, inst)","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:10:34.966201Z","iopub.execute_input":"2024-08-19T13:10:34.966574Z","iopub.status.idle":"2024-08-19T13:10:35.371279Z","shell.execute_reply.started":"2024-08-19T13:10:34.966545Z","shell.execute_reply":"2024-08-19T13:10:35.370448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3D Data\nIn this section we will take a look at a MRI scan, so that you learn how to handle 3D Data stored as multiple 2D DICOM files.\n\nTypically there is one file for each slice, thus we need to read all files and append the slices to a list","metadata":{}},{"cell_type":"code","source":"path_to_mri = Path('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images')","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:22:45.423864Z","iopub.execute_input":"2024-08-19T13:22:45.424498Z","iopub.status.idle":"2024-08-19T13:22:45.428688Z","shell.execute_reply.started":"2024-08-19T13:22:45.424468Z","shell.execute_reply":"2024-08-19T13:22:45.427770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can use the glob function to return all items in a directory which correspond to the provided pattern. <br />\nAs in this case, the directory only contains the DICOM files, we can return all files in it (\"*\")","metadata":{}},{"cell_type":"code","source":"all_files = list(path_to_mri.glob('*/*/*')) # as glob returns a generator, we convert it to a list","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:10:43.053591Z","iopub.execute_input":"2024-08-19T13:10:43.053955Z","iopub.status.idle":"2024-08-19T13:11:24.406305Z","shell.execute_reply.started":"2024-08-19T13:10:43.053925Z","shell.execute_reply":"2024-08-19T13:11:24.405317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_files[:20] # make sure that all files are present in the list","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:11:24.408053Z","iopub.execute_input":"2024-08-19T13:11:24.408776Z","iopub.status.idle":"2024-08-19T13:11:24.415586Z","shell.execute_reply.started":"2024-08-19T13:11:24.408741Z","shell.execute_reply":"2024-08-19T13:11:24.414577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we will read these files by using the read_file method and append them to a list\n- It will take some time to read all the paths.","metadata":{}},{"cell_type":"code","source":"mri_data = []\n\nfor path in all_files[:20]:\n    data = pydicom.read_file(path)\n    mri_data.append(data)","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:11:24.416709Z","iopub.execute_input":"2024-08-19T13:11:24.416963Z","iopub.status.idle":"2024-08-19T13:11:24.594626Z","shell.execute_reply.started":"2024-08-19T13:11:24.416941Z","shell.execute_reply":"2024-08-19T13:11:24.593913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As you can see from the printed paths above, it is possible that the DICOM files are not ordered according to their actual image position! <br />\nThis can be verified by inspecting the SliceLocation","metadata":{}},{"cell_type":"code","source":"### unordered slices ###\nfor slice in mri_data:\n    print(slice.SliceLocation)","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:11:49.469030Z","iopub.execute_input":"2024-08-19T13:11:49.469379Z","iopub.status.idle":"2024-08-19T13:11:49.475106Z","shell.execute_reply.started":"2024-08-19T13:11:49.469352Z","shell.execute_reply":"2024-08-19T13:11:49.474133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It crucial to order them, as otherwise your complete scan would be shuffeled and useless\n\nWe can use the \"SliceLocation\" attribute passed to the *sorted* function to identify the 2D slice position and thus order the slices","metadata":{}},{"cell_type":"code","source":"# this sorts the slices according to their location\nmri_data_ordered = sorted(mri_data, key = lambda slice: slice.SliceLocation)\n\n### Ordered slices ###\nfor slice in mri_data_ordered:\n    print(slice.SliceLocation)","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:11:55.478928Z","iopub.execute_input":"2024-08-19T13:11:55.479747Z","iopub.status.idle":"2024-08-19T13:11:55.485204Z","shell.execute_reply.started":"2024-08-19T13:11:55.479713Z","shell.execute_reply":"2024-08-19T13:11:55.484278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we extract the actual data (pixel_arrays) from the Dicom files and store them in a list","metadata":{}},{"cell_type":"code","source":"full_volume = []\nfor slice in mri_data_ordered:\n    full_volume.append(slice.pixel_array)","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:12:04.416176Z","iopub.execute_input":"2024-08-19T13:12:04.416973Z","iopub.status.idle":"2024-08-19T13:12:04.579420Z","shell.execute_reply.started":"2024-08-19T13:12:04.416942Z","shell.execute_reply":"2024-08-19T13:12:04.578683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\nshutil.copy('/kaggle/input/dc-full-volume/full_volume.txt', '/kaggle/working/')","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:20:47.951055Z","iopub.execute_input":"2024-08-19T13:20:47.951713Z","iopub.status.idle":"2024-08-19T13:20:47.959647Z","shell.execute_reply.started":"2024-08-19T13:20:47.951682Z","shell.execute_reply":"2024-08-19T13:20:47.958596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Write and Read from File\nfile = open('/kaggle/working/full_volume.txt','w')\nfor item in full_volume:\n    file.write(str(item)+\"\\n\")\nfile.close()","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:21:26.475296Z","iopub.execute_input":"2024-08-19T13:21:26.476714Z","iopub.status.idle":"2024-08-19T13:21:26.486319Z","shell.execute_reply.started":"2024-08-19T13:21:26.476684Z","shell.execute_reply":"2024-08-19T13:21:26.485496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"And now we can take a look at some slices of the ordered 3D volume:","metadata":{}},{"cell_type":"code","source":"fig, axis = plt.subplots(3,3, figsize = (10,10))\n\nslice_counter = 0\nfor i in range (3):\n    for j in range(3):\n        axis[i][j].imshow(full_volume[slice_counter], cmap ='gray')\n        slice_counter +=1","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:21:39.920041Z","iopub.execute_input":"2024-08-19T13:21:39.920682Z","iopub.status.idle":"2024-08-19T13:21:41.440241Z","shell.execute_reply.started":"2024-08-19T13:21:39.920649Z","shell.execute_reply":"2024-08-19T13:21:41.439321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Perfect!** <br />\nWe now have a way to handle 2D and 3D data stored in the DICOM format\n\nBut (as you will have noticed), manual file reading and ordering seems kind of tedious, it would be great if there was a tool which handles this for us.\n\nThere is, and its name is SimpleITK https://pypi.org/project/SimpleITK/\n\nSimpleITK provides functionality to automatically detect and read all dicom files without you managing the file reading or slice ordering\n\nThe overall routine is always identical\n\n1. Get Series Ids of all files in the directory. This is important as there also might be multiple scans in the same directory and we do not want to mix them. *ImageSeriesReader.GetGDCMSeriesIDs(path)* handles this and returns all Ids it can find\n2. Then we return all file names in the directory which have our desired Id *ImageSeriesReader.GetGDCMSeriesFileNames(path, ID)* provides this functionality\n3. We then define the image reader called *ImageSeriesReader()* and feed it the file names using *SetFileNames(file_names)*\n4. Finally we execute the reader in order to get our desired data by calling *Execute()*\n\nLet's go:<br />\nAt first we import the necessary libraries:","metadata":{}},{"cell_type":"markdown","source":"## Imports\n* **SimpleITK** to read 3D volumes in dcm format","metadata":{}},{"cell_type":"code","source":"path_to_mri = Path('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_images')\nall_files = list(path_to_mri.glob('*/*'))","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:39:51.711483Z","iopub.execute_input":"2024-08-19T13:39:51.712224Z","iopub.status.idle":"2024-08-19T13:39:52.686763Z","shell.execute_reply.started":"2024-08-19T13:39:51.712191Z","shell.execute_reply":"2024-08-19T13:39:52.685962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import SimpleITK as sitk\n# Simple library to read dcm file from path, extract, and sort\n#series_ids = sitk.ImageSeriesReader.GetGDCMSeriesIDs()\n#print (series_ids)\n\n\n# Iterate through each subdirectory and get the GDCM Series IDs\nfor subdir in all_files[:10]:\n    series_ids = sitk.ImageSeriesReader.GetGDCMSeriesIDs(str(subdir))\n    series_file_names = sitk.ImageSeriesReader.GetGDCMSeriesFileNames(str(subdir), series_ids[0])\n    if series_ids:\n        print(f\"Series IDs in directory {subdir}: {series_ids}\")\n    else:\n        print(f\"No DICOM series found in directory {subdir}\")","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:42:02.708718Z","iopub.execute_input":"2024-08-19T13:42:02.709456Z","iopub.status.idle":"2024-08-19T13:42:10.619585Z","shell.execute_reply.started":"2024-08-19T13:42:02.709421Z","shell.execute_reply":"2024-08-19T13:42:10.618663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"series_file_names","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:42:15.195108Z","iopub.execute_input":"2024-08-19T13:42:15.195781Z","iopub.status.idle":"2024-08-19T13:42:15.201500Z","shell.execute_reply.started":"2024-08-19T13:42:15.195748Z","shell.execute_reply":"2024-08-19T13:42:15.200540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sorted Data Files\n#series_file_names = sitk.ImageSeriesReader.GetGDCMSeriesFileNames(str(all_files), series_ids[0])\nseries_file_names","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:42:29.479493Z","iopub.execute_input":"2024-08-19T13:42:29.480461Z","iopub.status.idle":"2024-08-19T13:42:29.486463Z","shell.execute_reply.started":"2024-08-19T13:42:29.480431Z","shell.execute_reply":"2024-08-19T13:42:29.485536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"series_reader = sitk.ImageSeriesReader()\nseries_reader.SetFileNames(series_file_names)","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:42:32.423490Z","iopub.execute_input":"2024-08-19T13:42:32.424118Z","iopub.status.idle":"2024-08-19T13:42:32.431109Z","shell.execute_reply.started":"2024-08-19T13:42:32.424087Z","shell.execute_reply":"2024-08-19T13:42:32.430262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_data = series_reader.Execute()","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:42:35.031555Z","iopub.execute_input":"2024-08-19T13:42:35.032366Z","iopub.status.idle":"2024-08-19T13:42:37.235309Z","shell.execute_reply.started":"2024-08-19T13:42:35.032335Z","shell.execute_reply":"2024-08-19T13:42:37.234295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"head_mri = sitk.GetArrayFromImage(image_data)","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:42:39.447342Z","iopub.execute_input":"2024-08-19T13:42:39.448185Z","iopub.status.idle":"2024-08-19T13:42:39.464356Z","shell.execute_reply.started":"2024-08-19T13:42:39.448151Z","shell.execute_reply":"2024-08-19T13:42:39.463421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"head_mri.shape","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:42:40.919181Z","iopub.execute_input":"2024-08-19T13:42:40.919560Z","iopub.status.idle":"2024-08-19T13:42:40.925553Z","shell.execute_reply.started":"2024-08-19T13:42:40.919531Z","shell.execute_reply":"2024-08-19T13:42:40.924657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### This is all you have to do, to get your full volumetric data\nAs you can see, the shape is (320, 320, 60), whereas above the shape was (320, 320, 60). \nThis is just due to a different order of image dimensions. <br />\n\nThe final step we have to perform is the conversion of the sitk image object to a numpy array. This can be done by calling *GetArrayFromImage(image_data)*","metadata":{}},{"cell_type":"code","source":"fig, axis = plt.subplots(3,3, figsize = (10,10))\n\nslice_counter = 10\nfor i in range (3):\n    for j in range(3):\n        axis[i][j].imshow(head_mri[slice_counter], cmap ='gray')\n        slice_counter +=1","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:42:46.167454Z","iopub.execute_input":"2024-08-19T13:42:46.168099Z","iopub.status.idle":"2024-08-19T13:42:48.078092Z","shell.execute_reply.started":"2024-08-19T13:42:46.168064Z","shell.execute_reply":"2024-08-19T13:42:48.077119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"head_mri = sitk.GetArrayFromImage(image_data)\nprint(type(head_mri))\nprint(head_mri.shape)","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:42:52.073073Z","iopub.execute_input":"2024-08-19T13:42:52.073929Z","iopub.status.idle":"2024-08-19T13:42:52.088578Z","shell.execute_reply.started":"2024-08-19T13:42:52.073894Z","shell.execute_reply":"2024-08-19T13:42:52.087862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"As you can see, it also directly moved the slice channel to the front - Great!\nNow we can take a look at our images and as you can see the result is identical to the one above!","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axis = plt.subplots(3, 3, figsize=(10, 10))\n\nslice_counter = 0\nfor i in range(3):\n    for j in range(3):\n        axis[i][j].imshow(head_mri[slice_counter], cmap=\"gray\")\n        slice_counter+=1","metadata":{"execution":{"iopub.status.busy":"2024-08-19T13:42:58.982337Z","iopub.execute_input":"2024-08-19T13:42:58.982712Z","iopub.status.idle":"2024-08-19T13:43:00.841300Z","shell.execute_reply.started":"2024-08-19T13:42:58.982681Z","shell.execute_reply":"2024-08-19T13:43:00.840449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"You now have all the tools to work with files stored in the DICOM format\n","metadata":{}}]}