{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:08:20.303312Z","iopub.execute_input":"2025-03-06T16:08:20.303673Z","iopub.status.idle":"2025-03-06T16:08:20.308674Z","shell.execute_reply.started":"2025-03-06T16:08:20.303636Z","shell.execute_reply":"2025-03-06T16:08:20.307573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install pydicom -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:25:56.710195Z","iopub.execute_input":"2025-03-06T16:25:56.710635Z","iopub.status.idle":"2025-03-06T16:26:01.019691Z","shell.execute_reply.started":"2025-03-06T16:25:56.710596Z","shell.execute_reply":"2025-03-06T16:26:01.018202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nfrom collections import Counter\n\nimport matplotlib.pyplot as plt\nimport os\nimport time\nimport numpy as np\nimport collections\n\nimport pydicom as dicom\nimport matplotlib.patches as patches\n\nfrom matplotlib import animation, rc\nimport pandas as pd\n\nimport pydicom as dicom # dicom\nimport pydicom\n# from pydicom.pixel_data_handlers.util import apply_voi_lut","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:01.021887Z","iopub.execute_input":"2025-03-06T16:26:01.022389Z","iopub.status.idle":"2025-03-06T16:26:01.028682Z","shell.execute_reply.started":"2025-03-06T16:26:01.022344Z","shell.execute_reply":"2025-03-06T16:26:01.027603Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pwd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:01.030974Z","iopub.execute_input":"2025-03-06T16:26:01.031328Z","iopub.status.idle":"2025-03-06T16:26:01.171728Z","shell.execute_reply.started":"2025-03-06T16:26:01.0313Z","shell.execute_reply":"2025-03-06T16:26:01.170362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PROJECT_DIR = '/kaggle/input'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:01.173978Z","iopub.execute_input":"2025-03-06T16:26:01.174347Z","iopub.status.idle":"2025-03-06T16:26:01.179518Z","shell.execute_reply.started":"2025-03-06T16:26:01.174318Z","shell.execute_reply":"2025-03-06T16:26:01.178197Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Read, Inspect and Load Metadata","metadata":{}},{"cell_type":"code","source":"DATA_DIR = os.path.join(PROJECT_DIR, 'rsna-2024-lumbar-spine-degenerative-classification')\n\ntrain       = pd.read_csv(os.path.join(DATA_DIR, 'train.csv'))\ntrain_desc  = pd.read_csv(os.path.join(DATA_DIR, 'train_series_descriptions.csv'))\ntrain_label = pd.read_csv(os.path.join(DATA_DIR, 'train_label_coordinates.csv'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:01.180764Z","iopub.execute_input":"2025-03-06T16:26:01.181141Z","iopub.status.idle":"2025-03-06T16:26:01.349732Z","shell.execute_reply.started":"2025-03-06T16:26:01.181114Z","shell.execute_reply":"2025-03-06T16:26:01.34868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define function to reshape a single row of the DataFrame\ndef reshape_row(row):\n    data = {'study_id': [], 'condition': [], 'level': [], 'severity': []}\n    \n    for column, value in row.items():\n        if column not in ['study_id', 'series_id', 'instance_number', 'x', 'y', 'series_description']:\n            parts = column.split('_')\n            condition = ' '.join([word.capitalize() for word in parts[:-2]])\n            level = parts[-2].capitalize() + '/' + parts[-1].capitalize()\n            data['study_id'].append(row['study_id'])\n            data['condition'].append(condition)\n            data['level'].append(level)\n            data['severity'].append(value)\n    \n    return pd.DataFrame(data)\n\n# Reshape the DataFrame for all rows\nnew_train_df = pd.concat([reshape_row(row) for _, row in train.iterrows()], ignore_index=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:01.350859Z","iopub.execute_input":"2025-03-06T16:26:01.351235Z","iopub.status.idle":"2025-03-06T16:26:02.554664Z","shell.execute_reply.started":"2025-03-06T16:26:01.351208Z","shell.execute_reply":"2025-03-06T16:26:02.553763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"merged_df = pd.merge(new_train_df, train_label, on=['study_id', 'condition', 'level'], how='inner')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:02.555745Z","iopub.execute_input":"2025-03-06T16:26:02.556149Z","iopub.status.idle":"2025-03-06T16:26:02.608228Z","shell.execute_reply.started":"2025-03-06T16:26:02.556112Z","shell.execute_reply":"2025-03-06T16:26:02.607347Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_merged_df = pd.merge(merged_df, train_desc, on=['series_id','study_id'], how='inner')\nfinal_merged_df['image_path'] = [os.path.join(DATA_DIR, 'train_images',\n                                             str(final_merged_df.study_id[i]), \n                                             str(final_merged_df.series_id[i]), \n                                             str(final_merged_df.instance_number[i]) + '.dcm') for i in range(final_merged_df.shape[0])]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:02.610338Z","iopub.execute_input":"2025-03-06T16:26:02.610628Z","iopub.status.idle":"2025-03-06T16:26:04.055046Z","shell.execute_reply.started":"2025-03-06T16:26:02.610605Z","shell.execute_reply":"2025-03-06T16:26:04.054165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_merged_df.sample(n = 10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:04.056378Z","iopub.execute_input":"2025-03-06T16:26:04.056746Z","iopub.status.idle":"2025-03-06T16:26:04.073287Z","shell.execute_reply.started":"2025-03-06T16:26:04.056711Z","shell.execute_reply":"2025-03-06T16:26:04.072217Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"na_info = pd.DataFrame({\n    'Missing Values': final_merged_df.isna().sum(),\n    'Percentage': (final_merged_df.isna().sum() / len(final_merged_df) * 100).round(2),\n    'Total Rows': len(final_merged_df),\n    'Dtype': final_merged_df.dtypes\n})\nprint(na_info)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:04.074624Z","iopub.execute_input":"2025-03-06T16:26:04.074931Z","iopub.status.idle":"2025-03-06T16:26:04.140967Z","shell.execute_reply.started":"2025-03-06T16:26:04.074906Z","shell.execute_reply":"2025-03-06T16:26:04.139704Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_merged_df = final_merged_df.dropna()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:05.139825Z","iopub.execute_input":"2025-03-06T16:26:05.14022Z","iopub.status.idle":"2025-03-06T16:26:05.181765Z","shell.execute_reply.started":"2025-03-06T16:26:05.140189Z","shell.execute_reply":"2025-03-06T16:26:05.180817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"duplicates = final_merged_df.duplicated(subset=['study_id', 'series_id', 'instance_number', 'level'])\nduplicate_rows = final_merged_df[duplicates]\nduplicate_rows.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:06.59154Z","iopub.execute_input":"2025-03-06T16:26:06.592016Z","iopub.status.idle":"2025-03-06T16:26:06.620236Z","shell.execute_reply.started":"2025-03-06T16:26:06.591975Z","shell.execute_reply":"2025-03-06T16:26:06.619066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"duplicate_rows['condition'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:22.0752Z","iopub.execute_input":"2025-03-06T16:26:22.075526Z","iopub.status.idle":"2025-03-06T16:26:22.08407Z","shell.execute_reply.started":"2025-03-06T16:26:22.075501Z","shell.execute_reply":"2025-03-06T16:26:22.082894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"duplicate_rows['series_description'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:22.582562Z","iopub.execute_input":"2025-03-06T16:26:22.58299Z","iopub.status.idle":"2025-03-06T16:26:22.591344Z","shell.execute_reply.started":"2025-03-06T16:26:22.582951Z","shell.execute_reply":"2025-03-06T16:26:22.590181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"duplicate_rows['severity'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:23.189041Z","iopub.execute_input":"2025-03-06T16:26:23.189387Z","iopub.status.idle":"2025-03-06T16:26:23.197214Z","shell.execute_reply.started":"2025-03-06T16:26:23.189357Z","shell.execute_reply":"2025-03-06T16:26:23.196105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_merged_df = final_merged_df[~duplicates]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:25.773461Z","iopub.execute_input":"2025-03-06T16:26:25.773784Z","iopub.status.idle":"2025-03-06T16:26:25.785978Z","shell.execute_reply.started":"2025-03-06T16:26:25.773758Z","shell.execute_reply":"2025-03-06T16:26:25.784743Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Enocde Categories","metadata":{}},{"cell_type":"code","source":"LEVEL_LABELS = {\n    \"L1/L2\": 1,\n    \"L2/L3\": 2,\n    \"L3/L4\": 3,\n    \"L4/L5\": 4,\n    \"L5/S1\": 5\n}\nSEVERITY_LABELS = {\n    \"Normal/Mild\": 0,\n    \"Moderate\": 1,\n    \"Severe\": 2\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:27.382834Z","iopub.execute_input":"2025-03-06T16:26:27.383207Z","iopub.status.idle":"2025-03-06T16:26:27.388882Z","shell.execute_reply.started":"2025-03-06T16:26:27.383175Z","shell.execute_reply":"2025-03-06T16:26:27.387441Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_merged_df['level_code']    = final_merged_df['level'].map(LEVEL_LABELS)\nfinal_merged_df['severity_code'] = final_merged_df['severity'].map(SEVERITY_LABELS)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:30.123257Z","iopub.execute_input":"2025-03-06T16:26:30.123625Z","iopub.status.idle":"2025-03-06T16:26:30.140388Z","shell.execute_reply.started":"2025-03-06T16:26:30.123598Z","shell.execute_reply":"2025-03-06T16:26:30.139244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_merged_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:30.358351Z","iopub.execute_input":"2025-03-06T16:26:30.358678Z","iopub.status.idle":"2025-03-06T16:26:30.373857Z","shell.execute_reply.started":"2025-03-06T16:26:30.358653Z","shell.execute_reply":"2025-03-06T16:26:30.372582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"OUTPUT_DIR = '/kaggle/working'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:33.686384Z","iopub.execute_input":"2025-03-06T16:26:33.686712Z","iopub.status.idle":"2025-03-06T16:26:33.690878Z","shell.execute_reply.started":"2025-03-06T16:26:33.686687Z","shell.execute_reply":"2025-03-06T16:26:33.689758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"write_dir = os.path.join(OUTPUT_DIR, 'data', 'processed_metadata')\nos.makedirs(write_dir, exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:34.043921Z","iopub.execute_input":"2025-03-06T16:26:34.044264Z","iopub.status.idle":"2025-03-06T16:26:34.049084Z","shell.execute_reply.started":"2025-03-06T16:26:34.044236Z","shell.execute_reply":"2025-03-06T16:26:34.047925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_merged_df.to_csv(os.path.join(write_dir, 'processed_metadata.csv'), index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:35.394426Z","iopub.execute_input":"2025-03-06T16:26:35.394755Z","iopub.status.idle":"2025-03-06T16:26:35.826351Z","shell.execute_reply.started":"2025-03-06T16:26:35.39473Z","shell.execute_reply":"2025-03-06T16:26:35.825204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"condition_types = final_merged_df['condition'].unique().tolist()\nfor condition in condition_types:\n    df = final_merged_df[(final_merged_df['condition'] == condition)]\n    df.to_csv(os.path.join(write_dir, 'processed_metadata_' + condition.replace(\" \", \"\") + '.csv'),\n              index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:38.651528Z","iopub.execute_input":"2025-03-06T16:26:38.651937Z","iopub.status.idle":"2025-03-06T16:26:39.144609Z","shell.execute_reply.started":"2025-03-06T16:26:38.651904Z","shell.execute_reply":"2025-03-06T16:26:39.143588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!zip -r RSNA_Lumbar_metadata.zip /kaggle/working","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:14:02.427879Z","iopub.execute_input":"2025-03-06T16:14:02.428286Z","iopub.status.idle":"2025-03-06T16:14:03.090096Z","shell.execute_reply.started":"2025-03-06T16:14:02.428253Z","shell.execute_reply":"2025-03-06T16:14:03.088763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"condition_types","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T16:26:49.892369Z","iopub.execute_input":"2025-03-06T16:26:49.892701Z","iopub.status.idle":"2025-03-06T16:26:49.899057Z","shell.execute_reply.started":"2025-03-06T16:26:49.892676Z","shell.execute_reply":"2025-03-06T16:26:49.897639Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Example Patient: 4096820034","metadata":{}},{"cell_type":"code","source":"final_merged_df[final_merged_df.study_id == 4096820034]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T11:49:13.913613Z","iopub.execute_input":"2025-03-05T11:49:13.913974Z","iopub.status.idle":"2025-03-05T11:49:13.934297Z","shell.execute_reply.started":"2025-03-05T11:49:13.913945Z","shell.execute_reply":"2025-03-05T11:49:13.933259Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Summary of Stats:","metadata":{}},{"cell_type":"code","source":"print(f\"Number of patients/studies in training set: {final_merged_df['study_id'].nunique()}\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T11:49:13.935493Z","iopub.execute_input":"2025-03-05T11:49:13.935918Z","iopub.status.idle":"2025-03-05T11:49:13.941647Z","shell.execute_reply.started":"2025-03-05T11:49:13.935877Z","shell.execute_reply":"2025-03-05T11:49:13.940703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Number of series in training set: {final_merged_df['series_id'].nunique()}\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T11:49:13.942711Z","iopub.execute_input":"2025-03-05T11:49:13.943102Z","iopub.status.idle":"2025-03-05T11:49:13.961221Z","shell.execute_reply.started":"2025-03-05T11:49:13.943063Z","shell.execute_reply":"2025-03-05T11:49:13.96009Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"counts = final_merged_df.groupby('study_id')['series_id'].nunique()\nprint(Counter(counts))\n\nplt.figure(figsize=(6, 4))\nplt.hist(counts, bins=30)\nplt.title('Distribution of # series per Study ID')\nplt.xlabel('Number of # series')\nplt.ylabel('Frequency')\nplt.xticks(range(min(counts), max(counts) + 1, 2))  # Step by 2 for less crowding\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T11:49:13.96252Z","iopub.execute_input":"2025-03-05T11:49:13.962941Z","iopub.status.idle":"2025-03-05T11:49:14.275545Z","shell.execute_reply.started":"2025-03-05T11:49:13.962901Z","shell.execute_reply":"2025-03-05T11:49:14.274516Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Number of annotated labels/spine levels in training set: {final_merged_df.shape[0]}\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T11:49:14.276849Z","iopub.execute_input":"2025-03-05T11:49:14.277219Z","iopub.status.idle":"2025-03-05T11:49:14.282457Z","shell.execute_reply.started":"2025-03-05T11:49:14.277183Z","shell.execute_reply":"2025-03-05T11:49:14.281555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"name = final_merged_df['condition'].unique().tolist()\nplt.pie(final_merged_df['condition'].value_counts(normalize=True), \n        autopct='%1.f%%', \n        startangle=90, \n        wedgeprops=dict(width=0.25), \n        labeldistance=1.2, \n        counterclock=False, radius=1)\nplt.title(f'Distribution of diagnosed condition in entire dataset')\nplt.legend(name, bbox_to_anchor=(2, 0.5), prop={'size': 12}, markerscale = 3, framealpha=0.8, facecolor='white')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T11:49:14.283503Z","iopub.execute_input":"2025-03-05T11:49:14.283893Z","iopub.status.idle":"2025-03-05T11:49:14.62489Z","shell.execute_reply.started":"2025-03-05T11:49:14.283856Z","shell.execute_reply":"2025-03-05T11:49:14.623783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"name = ['Normal/Mild', 'Moderate', 'Severe']\np=['#d0d0d0', '#ffba07', '#ff0000']\nplt.pie(final_merged_df.severity.value_counts(normalize=True), \n        autopct = '%1.f%%', \n        colors = sns.color_palette(p), \n        startangle=90, \n        wedgeprops=dict(width=0.25), \n        labeldistance=1.2, \n        counterclock=False, radius=1)\nplt.title(f'Distribution of labels in entire dataset', color='blue')\nplt.legend(name, bbox_to_anchor=(1, 0.4), prop={'size': 12}, markerscale = 3, framealpha=0.8, facecolor='white')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T11:49:14.626126Z","iopub.execute_input":"2025-03-05T11:49:14.626595Z","iopub.status.idle":"2025-03-05T11:49:14.828399Z","shell.execute_reply.started":"2025-03-05T11:49:14.626456Z","shell.execute_reply":"2025-03-05T11:49:14.827381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize=(20, 4))\ncolours = {\"male\": \"#273c75\", \"female\": \"#44bd32\"}\nfor idx, d in enumerate(['foraminal', 'subarticular', 'canal']):\n    diagnosis = list(filter(lambda x: x.find(d) > -1, train.columns))\n    dff = train[diagnosis]\n    value_counts = dff.apply(pd.value_counts).fillna(0).T\n    value_counts.plot(kind='bar', stacked=True, ax=axes[idx], cmap='Set1_r')\n    axes[idx].tick_params(axis='x', labelsize=16) \n    axes[idx].set_title(f\"{d} distribution\",  fontsize=20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T11:49:14.829362Z","iopub.execute_input":"2025-03-05T11:49:14.829644Z","iopub.status.idle":"2025-03-05T11:49:16.030296Z","shell.execute_reply.started":"2025-03-05T11:49:14.829621Z","shell.execute_reply":"2025-03-05T11:49:16.029254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"counts = train_label.groupby('series_id')['condition'].nunique()\nprint(Counter(counts))\n\nplt.figure(figsize=(6, 4))\nplt.hist(counts, bins=30)\nplt.title('Distribution of #labels per series')\nplt.xlabel('Number of #labels')\nplt.ylabel('Frequency')\nplt.xticks(range(min(counts), max(counts) + 1, 2))  # Step by 2 for less crowding\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T11:49:16.031503Z","iopub.execute_input":"2025-03-05T11:49:16.031795Z","iopub.status.idle":"2025-03-05T11:49:16.254559Z","shell.execute_reply.started":"2025-03-05T11:49:16.031771Z","shell.execute_reply":"2025-03-05T11:49:16.253662Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Display Images for a patient","metadata":{}},{"cell_type":"code","source":"study_id = 4096820034\nseries_ids = final_merged_df['series_id'][final_merged_df['study_id'] == 4096820034].unique().tolist()\n\n## Descriptions corresponds to each series\nfinal_merged_df[['study_id', 'series_id', 'series_description']][final_merged_df.study_id == 4096820034].groupby('series_id').first().reindex(series_ids)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T11:49:16.255575Z","iopub.execute_input":"2025-03-05T11:49:16.255861Z","iopub.status.idle":"2025-03-05T11:49:16.275549Z","shell.execute_reply.started":"2025-03-05T11:49:16.255812Z","shell.execute_reply":"2025-03-05T11:49:16.274428Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Plot all Dicom Images for this patient","metadata":{}},{"cell_type":"code","source":"image_dir = os.path.join(DATA_DIR, 'train_images')\n\n# Function to generate image paths based on directory structure\ndef generate_image_paths(study_id, series_id):\n    \n    series_dir = os.path.join(image_dir, str(study_id), str(series_id))\n    images = sorted(os.listdir(series_dir))\n    image_paths = [os.path.join(series_dir, img) for img in images]\n    \n    return image_paths\n\n# Function to open and display DICOM images\ndef display_dicom_images(image_paths):\n    \n    n_images = len(image_paths)\n    n_cols = 3\n    n_rows = (n_images - 1) // n_cols + 1\n    \n    plt.figure(figsize=(5*n_cols, 5*n_rows))  \n    for i, path in enumerate(image_paths):\n        ds = pydicom.dcmread(path)\n        plt.subplot(n_rows, n_cols, i+1)\n        plt.imshow(ds.pixel_array, cmap=plt.cm.bone)\n        plt.title(f\"Image {path.split('/')[-1]}\")\n        plt.axis('off')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T11:54:15.25269Z","iopub.execute_input":"2025-03-05T11:54:15.253199Z","iopub.status.idle":"2025-03-05T11:54:15.264241Z","shell.execute_reply.started":"2025-03-05T11:54:15.253157Z","shell.execute_reply":"2025-03-05T11:54:15.262066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_paths = generate_image_paths(study_id, series_ids[0])\ndisplay_dicom_images(image_paths)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T11:54:28.592313Z","iopub.execute_input":"2025-03-05T11:54:28.592666Z","iopub.status.idle":"2025-03-05T11:54:30.783908Z","shell.execute_reply.started":"2025-03-05T11:54:28.592635Z","shell.execute_reply":"2025-03-05T11:54:30.782378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_paths = generate_image_paths(study_id, series_ids[1])\ndisplay_dicom_images(image_paths)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T11:54:43.594798Z","iopub.execute_input":"2025-03-05T11:54:43.595248Z","iopub.status.idle":"2025-03-05T11:54:45.785313Z","shell.execute_reply.started":"2025-03-05T11:54:43.595215Z","shell.execute_reply":"2025-03-05T11:54:45.783323Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Axial T2 - 1st Series","metadata":{}},{"cell_type":"code","source":"image_paths = generate_image_paths(study_id, series_ids[2])\ndisplay_dicom_images(image_paths)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T11:54:54.832999Z","iopub.execute_input":"2025-03-05T11:54:54.833333Z","iopub.status.idle":"2025-03-05T11:55:00.083979Z","shell.execute_reply.started":"2025-03-05T11:54:54.833307Z","shell.execute_reply":"2025-03-05T11:55:00.082819Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Axial T2- 2nd Series","metadata":{}},{"cell_type":"code","source":"image_paths = generate_image_paths(study_id, series_ids[3])\ndisplay_dicom_images(image_paths)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T11:55:41.566493Z","iopub.execute_input":"2025-03-05T11:55:41.566902Z","iopub.status.idle":"2025-03-05T11:55:44.993007Z","shell.execute_reply.started":"2025-03-05T11:55:41.566869Z","shell.execute_reply":"2025-03-05T11:55:44.990821Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Plot DICOM images with annotated coordinates for this patient","metadata":{}},{"cell_type":"code","source":"# Function to open and display DICOM images along with coordinates\ndef display_dicom_with_coordinates(image_paths, train_label_df):\n\n    fig, axs = plt.subplots(len(image_paths), 1, figsize=(8, 6*len(image_paths)))\n    \n    for idx, path in enumerate(image_paths):  # Display images\n        study_id = int(path.split('/')[-3])\n        series_id = int(path.split('/')[-2])\n        instance_number = int(path.split('/')[-1].split('.')[0])\n        series_desc = train_label_df['series_description'][(train_label_df['study_id'] == study_id) & \n                                                           (train_label_df['series_id'] == series_id) &\n                                                           (train_label_df['instance_number'] == instance_number)].unique()\n        \n        # Filter train_label coordinates for the current study and series\n        filtered_train_labels = train_label_df[(train_label_df['study_id'] == study_id) & \n                                               (train_label_df['series_id'] == series_id) &\n                                               (train_label_df['instance_number'] == instance_number)]\n        \n        # print(study_id, series_id, instance_number, series_desc, filtered_train_labels)\n        \n        # Read DICOM image\n        ds = pydicom.dcmread(path)\n        \n        # Plot DICOM image\n        axs[idx].imshow(ds.pixel_array, cmap='gray')\n        axs[idx].set_title(f\"Study ID: {study_id}, Series ID: {series_id} ({series_desc[0]})\")\n        axs[idx].axis('off')\n        \n        # Plot coordinates\n        for _, row in filtered_train_labels.iterrows():\n            axs[idx].plot(row['x'], row['y'], 'ro', markersize=5)\n            axs[idx].text(row['x']+8, row['y']+3, f\"{row['level']} ({row['severity']} {row['condition']})\", color='r',fontsize=10)\n        \n    plt.tight_layout()\n    plt.show()\n\n# Load DICOM files from a folder\ndef load_dicom_files(path_to_folder):\n    \n    files = [os.path.join(path_to_folder, f) for f in os.listdir(path_to_folder) if f.endswith('.dcm')]\n    files.sort(key=lambda x: int(os.path.splitext(os.path.basename(x))[0].split('-')[-1]))\n    \n    return files","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T11:56:04.768378Z","iopub.execute_input":"2025-03-05T11:56:04.768744Z","iopub.status.idle":"2025-03-05T11:56:04.778872Z","shell.execute_reply.started":"2025-03-05T11:56:04.768715Z","shell.execute_reply":"2025-03-05T11:56:04.777581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_paths = final_merged_df['image_path'][final_merged_df['study_id'] == study_id].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T11:56:16.731355Z","iopub.execute_input":"2025-03-05T11:56:16.73167Z","iopub.status.idle":"2025-03-05T11:56:16.737184Z","shell.execute_reply.started":"2025-03-05T11:56:16.731645Z","shell.execute_reply":"2025-03-05T11:56:16.736188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display DICOM images with coordinates\ndisplay_dicom_with_coordinates(image_paths, final_merged_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T11:56:22.911435Z","iopub.execute_input":"2025-03-05T11:56:22.911768Z","iopub.status.idle":"2025-03-05T11:56:27.669051Z","shell.execute_reply.started":"2025-03-05T11:56:22.911742Z","shell.execute_reply":"2025-03-05T11:56:27.66723Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Another Example with Severe Condition","metadata":{}},{"cell_type":"code","source":"study_id_2 = 4279958262","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T11:57:54.374518Z","iopub.execute_input":"2025-03-05T11:57:54.374927Z","iopub.status.idle":"2025-03-05T11:57:54.379768Z","shell.execute_reply.started":"2025-03-05T11:57:54.374897Z","shell.execute_reply":"2025-03-05T11:57:54.378408Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_paths = final_merged_df['image_path'][final_merged_df['study_id'] == study_id_2].unique()\ndisplay_dicom_with_coordinates(image_paths, final_merged_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T11:57:55.686568Z","iopub.execute_input":"2025-03-05T11:57:55.686958Z","iopub.status.idle":"2025-03-05T11:57:59.817677Z","shell.execute_reply.started":"2025-03-05T11:57:55.686927Z","shell.execute_reply":"2025-03-05T11:57:59.815913Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from pydrive.auth import GoogleAuth\n# from pydrive.drive import GoogleDrive\n# from google.colab import auth\n# from oauth2client.client import GoogleCredentials\n\n# # Authenticate and create the PyDrive client.\n# auth.authenticate_user()\n# gauth = GoogleAuth()\n# gauth.LocalWebserverAuth()\n# drive = GoogleDrive(gauth)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T19:31:18.936786Z","iopub.execute_input":"2025-03-05T19:31:18.937228Z","iopub.status.idle":"2025-03-05T19:32:54.309775Z","shell.execute_reply.started":"2025-03-05T19:31:18.937194Z","shell.execute_reply":"2025-03-05T19:32:54.308139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Upload a file\nfile_path = \"/kaggle/working/my_data.zip\"\nfile_drive = drive.CreateFile({'title': 'my_data.zip'})  \nfile_drive.SetContentFile(file_path)\nfile_drive.Upload()\n\nprint(\"File uploaded successfully to Google Drive.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}