{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport pydicom  as dicom\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-04T19:47:34.699307Z","iopub.execute_input":"2024-06-04T19:47:34.700463Z","iopub.status.idle":"2024-06-04T19:47:34.706491Z","shell.execute_reply.started":"2024-06-04T19:47:34.700426Z","shell.execute_reply":"2024-06-04T19:47:34.704938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting libraries\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:47:53.210925Z","iopub.execute_input":"2024-06-04T19:47:53.211345Z","iopub.status.idle":"2024-06-04T19:47:53.217475Z","shell.execute_reply.started":"2024-06-04T19:47:53.211317Z","shell.execute_reply":"2024-06-04T19:47:53.215840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Checking files in the directory\n!ls -l '../input/rsna-2024-lumbar-spine-degenerative-classification'\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:48:07.834962Z","iopub.execute_input":"2024-06-04T19:48:07.835694Z","iopub.status.idle":"2024-06-04T19:48:08.940872Z","shell.execute_reply.started":"2024-06-04T19:48:07.835654Z","shell.execute_reply":"2024-06-04T19:48:08.939444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Configuration\ndefault_color_1 = 'blue'","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:48:40.358439Z","iopub.execute_input":"2024-06-04T19:48:40.358920Z","iopub.status.idle":"2024-06-04T19:48:40.364864Z","shell.execute_reply.started":"2024-06-04T19:48:40.358860Z","shell.execute_reply":"2024-06-04T19:48:40.363364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Reading data\ndf_train_main = pd.read_csv('../input/rsna-2024-lumbar-spine-degenerative-classification/train.csv')\ndf_train_label = pd.read_csv('../input/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv')\ndf_train_desc = pd.read_csv('../input/rsna-2024-lumbar-spine-degenerative-classification/train_series_descriptions.csv')\ndf_test_desc = pd.read_csv('../input/rsna-2024-lumbar-spine-degenerative-classification/test_series_descriptions.csv')\ndf_sub = pd.read_csv('../input/rsna-2024-lumbar-spine-degenerative-classification/sample_submission.csv')\n","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:49:03.986869Z","iopub.execute_input":"2024-06-04T19:49:03.987378Z","iopub.status.idle":"2024-06-04T19:49:04.207332Z","shell.execute_reply.started":"2024-06-04T19:49:03.987344Z","shell.execute_reply":"2024-06-04T19:49:04.205878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Exploring structure and preview of data\ndf_train_main.info()\ndf_train_main.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:49:18.578917Z","iopub.execute_input":"2024-06-04T19:49:18.579858Z","iopub.status.idle":"2024-06-04T19:49:18.657574Z","shell.execute_reply.started":"2024-06-04T19:49:18.579821Z","shell.execute_reply":"2024-06-04T19:49:18.656068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Exploring structure and preview of data\ndf_train_label.info()\ndf_train_label.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:49:37.094914Z","iopub.execute_input":"2024-06-04T19:49:37.095411Z","iopub.status.idle":"2024-06-04T19:49:37.132544Z","shell.execute_reply.started":"2024-06-04T19:49:37.095374Z","shell.execute_reply":"2024-06-04T19:49:37.131433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Exploring structure and preview of data\ndf_train_label.info()\ndf_train_label.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:50:06.714021Z","iopub.execute_input":"2024-06-04T19:50:06.714950Z","iopub.status.idle":"2024-06-04T19:50:06.749088Z","shell.execute_reply.started":"2024-06-04T19:50:06.714908Z","shell.execute_reply":"2024-06-04T19:50:06.748030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data Preprocessing\n\nplt.figure(figsize=(5,1))\nplt.boxplot(df_train_label.instance_number, vert=False)\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:50:26.703350Z","iopub.execute_input":"2024-06-04T19:50:26.704380Z","iopub.status.idle":"2024-06-04T19:50:26.939249Z","shell.execute_reply.started":"2024-06-04T19:50:26.704338Z","shell.execute_reply":"2024-06-04T19:50:26.937906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Examining rows with very high instance numbers\nhigh_instance_numbers = df_train_label[df_train_label.instance_number > 150]","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:50:45.594352Z","iopub.execute_input":"2024-06-04T19:50:45.595557Z","iopub.status.idle":"2024-06-04T19:50:45.603224Z","shell.execute_reply.started":"2024-06-04T19:50:45.595518Z","shell.execute_reply":"2024-06-04T19:50:45.601918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cross-tabulating condition vs level\ncross_tab = pd.crosstab(df_train_label.condition, df_train_label.level)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:51:02.239064Z","iopub.execute_input":"2024-06-04T19:51:02.239489Z","iopub.status.idle":"2024-06-04T19:51:02.274676Z","shell.execute_reply.started":"2024-06-04T19:51:02.239455Z","shell.execute_reply":"2024-06-04T19:51:02.273531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_desc_info = df_train_desc.info()\ndf_train_desc_head = df_train_desc.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:51:26.416586Z","iopub.execute_input":"2024-06-04T19:51:26.417927Z","iopub.status.idle":"2024-06-04T19:51:26.431823Z","shell.execute_reply.started":"2024-06-04T19:51:26.417841Z","shell.execute_reply":"2024-06-04T19:51:26.430455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defalt_color='green'\n# Visualizing categories in the series descriptions\nseries_desc_counts = df_train_desc.series_description.value_counts()\nplt.figure(figsize=(5,1))\nplt.bar(series_desc_counts.index, series_desc_counts.values, color=defalt_color)\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:51:48.665994Z","iopub.execute_input":"2024-06-04T19:51:48.666427Z","iopub.status.idle":"2024-06-04T19:51:48.845951Z","shell.execute_reply.started":"2024-06-04T19:51:48.666394Z","shell.execute_reply":"2024-06-04T19:51:48.844834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data Integration\ndf_train_step_1 = pd.merge(left=df_train_label, right=df_train_main, how='left', on='study_id').reset_index(drop=True)\ndf_train = pd.merge(left=df_train_step_1, right=df_train_desc, how='left', on=['study_id', 'series_id']).reset_index(drop=True)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:52:11.466376Z","iopub.execute_input":"2024-06-04T19:52:11.467551Z","iopub.status.idle":"2024-06-04T19:52:11.685950Z","shell.execute_reply.started":"2024-06-04T19:52:11.467512Z","shell.execute_reply":"2024-06-04T19:52:11.684751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Converting IDs to categorical variables\ndf_train.study_id = df_train.study_id.astype('category')\ndf_train.series_id = df_train.series_id.astype('category')","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:52:29.128802Z","iopub.execute_input":"2024-06-04T19:52:29.129207Z","iopub.status.idle":"2024-06-04T19:52:29.140417Z","shell.execute_reply.started":"2024-06-04T19:52:29.129180Z","shell.execute_reply":"2024-06-04T19:52:29.139043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_desc_stats = df_train.describe(include='all')","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:53:48.887499Z","iopub.execute_input":"2024-06-04T19:53:48.887961Z","iopub.status.idle":"2024-06-04T19:53:49.287002Z","shell.execute_reply.started":"2024-06-04T19:53:48.887927Z","shell.execute_reply":"2024-06-04T19:53:49.285827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualizing \nsns.jointplot(data=df_train, x='x', y='y', color=defalt_color, alpha=0.25)\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:54:13.143822Z","iopub.execute_input":"2024-06-04T19:54:13.144614Z","iopub.status.idle":"2024-06-04T19:54:14.641039Z","shell.execute_reply.started":"2024-06-04T19:54:13.144581Z","shell.execute_reply":"2024-06-04T19:54:14.639584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Joining the first two tables based on the 'study_id' column\ndf_train_combined = pd.merge(df_train_main, df_train_label, on='study_id', how='inner')\ndf_train_combined","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:54:37.155846Z","iopub.execute_input":"2024-06-04T19:54:37.156243Z","iopub.status.idle":"2024-06-04T19:54:37.274183Z","shell.execute_reply.started":"2024-06-04T19:54:37.156214Z","shell.execute_reply":"2024-06-04T19:54:37.273026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Joining the combined dataframe with the third table based on 'study_id' and 'series_id' columns\ndf_train_final = pd.merge(df_train_combined, df_train_desc, on=['study_id', 'series_id'], how='left')\ndf_train_final","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:54:57.805031Z","iopub.execute_input":"2024-06-04T19:54:57.805490Z","iopub.status.idle":"2024-06-04T19:54:57.963192Z","shell.execute_reply.started":"2024-06-04T19:54:57.805459Z","shell.execute_reply":"2024-06-04T19:54:57.961797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_final['study_id'] = df_train_final['study_id'].astype('category')\ndf_train_final['series_id'] = df_train_final['series_id'].astype('category')","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:55:24.949177Z","iopub.execute_input":"2024-06-04T19:55:24.950133Z","iopub.status.idle":"2024-06-04T19:55:24.960186Z","shell.execute_reply.started":"2024-06-04T19:55:24.950097Z","shell.execute_reply":"2024-06-04T19:55:24.959049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#EDA\ndf_train_final_stats = df_train_final.describe(include='all')\ndf_train_final_stats","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:55:51.791934Z","iopub.execute_input":"2024-06-04T19:55:51.792438Z","iopub.status.idle":"2024-06-04T19:55:52.220093Z","shell.execute_reply.started":"2024-06-04T19:55:51.792402Z","shell.execute_reply":"2024-06-04T19:55:52.218953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(7, 5))\nplt.scatter(df_train_final['x'], df_train_final['y'], color='green', alpha=0.5)\nplt.title('Scatter Plot of Coordinates')\nplt.xlabel('X Coordinate')\nplt.ylabel('Y Coordinate')\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:56:10.689470Z","iopub.execute_input":"2024-06-04T19:56:10.689924Z","iopub.status.idle":"2024-06-04T19:56:11.164563Z","shell.execute_reply.started":"2024-06-04T19:56:10.689875Z","shell.execute_reply":"2024-06-04T19:56:11.163441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = df_train_main.columns.drop('study_id').tolist()\nprint(labels)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:56:33.991568Z","iopub.execute_input":"2024-06-04T19:56:33.992252Z","iopub.status.idle":"2024-06-04T19:56:33.998602Z","shell.execute_reply.started":"2024-06-04T19:56:33.992220Z","shell.execute_reply":"2024-06-04T19:56:33.997450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting bar plots for each categorical label separately\nfor label in labels:\n    plt.figure(figsize=(4, 3))\n    df_train_main[label].value_counts().plot(kind='bar', color='skyblue')\n    plt.title(f'Distribution of {label}')\n    plt.xlabel('Categories')\n    plt.ylabel('Frequency')\n    plt.xticks(rotation=45)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:56:52.381788Z","iopub.execute_input":"2024-06-04T19:56:52.382817Z","iopub.status.idle":"2024-06-04T19:56:58.930476Z","shell.execute_reply.started":"2024-06-04T19:56:52.382772Z","shell.execute_reply":"2024-06-04T19:56:58.929396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"freqs = pd.DataFrame(labels, columns=['label'])\nfreqs['p1'] = 1.0\nfreqs['p2'] = 0.0\nfreqs['p3'] = 0.0\n\nfor l in labels:\n    rel_counts = df_train_main[l].value_counts(normalize=True)\n    freqs.loc[freqs.label==l, 'p1'] = rel_counts['Normal/Mild']\n    freqs.loc[freqs.label==l, 'p2'] = rel_counts['Moderate']\n    freqs.loc[freqs.label==l, 'p3'] = rel_counts['Severe']\n\n\nfreqs","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:57:33.678650Z","iopub.execute_input":"2024-06-04T19:57:33.679101Z","iopub.status.idle":"2024-06-04T19:57:33.765139Z","shell.execute_reply.started":"2024-06-04T19:57:33.679068Z","shell.execute_reply":"2024-06-04T19:57:33.763800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the combined dataframe as a CSV file\ndf_train_final.to_csv('df_train.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:58:08.098398Z","iopub.execute_input":"2024-06-04T19:58:08.098860Z","iopub.status.idle":"2024-06-04T19:58:09.559690Z","shell.execute_reply.started":"2024-06-04T19:58:08.098826Z","shell.execute_reply":"2024-06-04T19:58:09.558558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_study = 4646740\ndf_ex = df_train[df_train.study_id==my_study]\ndf_ex","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:58:27.072542Z","iopub.execute_input":"2024-06-04T19:58:27.072969Z","iopub.status.idle":"2024-06-04T19:58:27.116488Z","shell.execute_reply.started":"2024-06-04T19:58:27.072937Z","shell.execute_reply":"2024-06-04T19:58:27.115380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_series = 3666319702\ndf_ex1 = df_ex[df_ex.series_id==my_series]\ndf_ex1","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:58:44.736228Z","iopub.execute_input":"2024-06-04T19:58:44.736974Z","iopub.status.idle":"2024-06-04T19:58:44.764024Z","shell.execute_reply.started":"2024-06-04T19:58:44.736939Z","shell.execute_reply":"2024-06-04T19:58:44.762865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_path = '../input/rsna-2024-lumbar-spine-degenerative-classification/train_images/' + str(my_study) + '/' + str(my_series) + '/'\nfor dirname, _, filenames in os.walk(my_path):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:59:02.745862Z","iopub.execute_input":"2024-06-04T19:59:02.746792Z","iopub.status.idle":"2024-06-04T19:59:02.757183Z","shell.execute_reply.started":"2024-06-04T19:59:02.746757Z","shell.execute_reply":"2024-06-04T19:59:02.756043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(1,15+1):\n    my_file = my_path + str(i) + '.dcm'\n    print(my_file)\n    # load image\n    ds = dicom.dcmread(my_file)\n    # and plot\n    plt.imshow(ds.pixel_array)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-04T19:59:30.197534Z","iopub.execute_input":"2024-06-04T19:59:30.197996Z","iopub.status.idle":"2024-06-04T19:59:34.610154Z","shell.execute_reply.started":"2024-06-04T19:59:30.197963Z","shell.execute_reply":"2024-06-04T19:59:34.609061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preview test data\ndf_test_desc.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-04T20:00:18.725416Z","iopub.execute_input":"2024-06-04T20:00:18.726129Z","iopub.status.idle":"2024-06-04T20:00:18.737332Z","shell.execute_reply.started":"2024-06-04T20:00:18.726097Z","shell.execute_reply":"2024-06-04T20:00:18.735714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preview sample submission\ndf_sub.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-04T20:00:31.795370Z","iopub.execute_input":"2024-06-04T20:00:31.796218Z","iopub.status.idle":"2024-06-04T20:00:31.808428Z","shell.execute_reply.started":"2024-06-04T20:00:31.796185Z","shell.execute_reply":"2024-06-04T20:00:31.807206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# setup baseline model just using observed frequencies\nn = df_sub.shape[0]\nfor i in range(0,n):\n    # extract label from id\n    current_label = df_sub.loc[i, 'row_id'].split('_',1)[1]\n    # look up frequencies in frequency table\n    p1 = freqs.loc[freqs.label==current_label, 'p1'].min()\n    p2 = freqs.loc[freqs.label==current_label, 'p2'].min()\n    p3 = freqs.loc[freqs.label==current_label, 'p3'].min()\n    # transfer results to submission table\n    df_sub.loc[i, 'normal_mild'] = p1\n    df_sub.loc[i, 'moderate'] = p2\n    df_sub.loc[i, 'severe'] = p3\ndf_sub","metadata":{"execution":{"iopub.status.busy":"2024-06-04T20:00:55.538061Z","iopub.execute_input":"2024-06-04T20:00:55.539441Z","iopub.status.idle":"2024-06-04T20:00:55.622345Z","shell.execute_reply.started":"2024-06-04T20:00:55.539406Z","shell.execute_reply":"2024-06-04T20:00:55.621283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save submission file\ndf_sub.to_csv('submission_freq.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T20:01:21.231260Z","iopub.execute_input":"2024-06-04T20:01:21.232327Z","iopub.status.idle":"2024-06-04T20:01:21.239512Z","shell.execute_reply.started":"2024-06-04T20:01:21.232286Z","shell.execute_reply":"2024-06-04T20:01:21.238330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nn = df_sub.shape[0]\nfor i in range(0,n):\n    df_sub.loc[i, 'normal_mild'] = 0.34\n    df_sub.loc[i, 'moderate'] = 0.33\n    df_sub.loc[i, 'severe'] = 0.33","metadata":{"execution":{"iopub.status.busy":"2024-06-04T20:01:45.000065Z","iopub.execute_input":"2024-06-04T20:01:45.000452Z","iopub.status.idle":"2024-06-04T20:01:45.032343Z","shell.execute_reply.started":"2024-06-04T20:01:45.000423Z","shell.execute_reply":"2024-06-04T20:01:45.031172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save submission file\ndf_sub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T20:02:04.418237Z","iopub.execute_input":"2024-06-04T20:02:04.418757Z","iopub.status.idle":"2024-06-04T20:02:04.427135Z","shell.execute_reply.started":"2024-06-04T20:02:04.418718Z","shell.execute_reply":"2024-06-04T20:02:04.425514Z"},"trusted":true},"execution_count":null,"outputs":[]}]}