{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Table of Contents\n* [Data Preparation](#data_prep)\n* [EDA](#eda)\n* [Example](#example)\n* [Submission](#sub)","metadata":{}},{"cell_type":"code","source":"# packages\n\n# standard\nimport numpy as np\nimport pandas as pd\nimport os\nimport time\n\n# plots\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nimport seaborn as sns\n\n# dicom\nimport pydicom as dicom","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-17T13:40:53.942791Z","iopub.execute_input":"2024-05-17T13:40:53.943255Z","iopub.status.idle":"2024-05-17T13:40:53.950305Z","shell.execute_reply.started":"2024-05-17T13:40:53.943219Z","shell.execute_reply":"2024-05-17T13:40:53.948949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# warning handling\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:40:53.955335Z","iopub.execute_input":"2024-05-17T13:40:53.955769Z","iopub.status.idle":"2024-05-17T13:40:53.961378Z","shell.execute_reply.started":"2024-05-17T13:40:53.955738Z","shell.execute_reply":"2024-05-17T13:40:53.960340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show files\n!ls -l '../input/rsna-2024-lumbar-spine-degenerative-classification'","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:40:53.972507Z","iopub.execute_input":"2024-05-17T13:40:53.973595Z","iopub.status.idle":"2024-05-17T13:40:55.106375Z","shell.execute_reply.started":"2024-05-17T13:40:53.973541Z","shell.execute_reply":"2024-05-17T13:40:55.104795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# configs\ndefault_color_1 = 'darkblue'","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:40:55.108956Z","iopub.execute_input":"2024-05-17T13:40:55.109400Z","iopub.status.idle":"2024-05-17T13:40:55.115692Z","shell.execute_reply.started":"2024-05-17T13:40:55.109360Z","shell.execute_reply":"2024-05-17T13:40:55.114196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read data\ndf_train_main = pd.read_csv('../input/rsna-2024-lumbar-spine-degenerative-classification/train.csv')\ndf_train_label = pd.read_csv('../input/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv')\ndf_train_desc = pd.read_csv('../input/rsna-2024-lumbar-spine-degenerative-classification/train_series_descriptions.csv')\ndf_test_desc = pd.read_csv('../input/rsna-2024-lumbar-spine-degenerative-classification/test_series_descriptions.csv')\ndf_sub = pd.read_csv('../input/rsna-2024-lumbar-spine-degenerative-classification/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:40:55.117434Z","iopub.execute_input":"2024-05-17T13:40:55.117832Z","iopub.status.idle":"2024-05-17T13:40:55.251483Z","shell.execute_reply.started":"2024-05-17T13:40:55.117800Z","shell.execute_reply":"2024-05-17T13:40:55.250184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='data_prep'></a>\n# Data Preparation","metadata":{}},{"cell_type":"markdown","source":"### Main Table","metadata":{}},{"cell_type":"code","source":"# structure\ndf_train_main.info()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:40:55.254724Z","iopub.execute_input":"2024-05-17T13:40:55.255095Z","iopub.status.idle":"2024-05-17T13:40:55.277800Z","shell.execute_reply.started":"2024-05-17T13:40:55.255064Z","shell.execute_reply":"2024-05-17T13:40:55.276586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preview\ndf_train_main.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:40:55.279152Z","iopub.execute_input":"2024-05-17T13:40:55.279492Z","iopub.status.idle":"2024-05-17T13:40:55.312475Z","shell.execute_reply.started":"2024-05-17T13:40:55.279463Z","shell.execute_reply":"2024-05-17T13:40:55.310988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Labels","metadata":{}},{"cell_type":"code","source":"# structure\ndf_train_label.info()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:40:55.314024Z","iopub.execute_input":"2024-05-17T13:40:55.314806Z","iopub.status.idle":"2024-05-17T13:40:55.341627Z","shell.execute_reply.started":"2024-05-17T13:40:55.314760Z","shell.execute_reply":"2024-05-17T13:40:55.340247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preview\ndf_train_label.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:40:55.343127Z","iopub.execute_input":"2024-05-17T13:40:55.343467Z","iopub.status.idle":"2024-05-17T13:40:55.358632Z","shell.execute_reply.started":"2024-05-17T13:40:55.343438Z","shell.execute_reply":"2024-05-17T13:40:55.357212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# look at categories\nfor f in ['instance_number','condition','level']:\n    print(df_train_label[f].value_counts())\n    print()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:40:55.360765Z","iopub.execute_input":"2024-05-17T13:40:55.361252Z","iopub.status.idle":"2024-05-17T13:40:55.388511Z","shell.execute_reply.started":"2024-05-17T13:40:55.361209Z","shell.execute_reply":"2024-05-17T13:40:55.387348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# numerical plot for instance number\nplt.figure(figsize=(7,1))\ndf_train_label.instance_number.plot(kind='box', vert=False)\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:40:55.390066Z","iopub.execute_input":"2024-05-17T13:40:55.390509Z","iopub.status.idle":"2024-05-17T13:40:55.611697Z","shell.execute_reply.started":"2024-05-17T13:40:55.390467Z","shell.execute_reply":"2024-05-17T13:40:55.610390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# look at few rows with the very high instance numbers\ndf_train_label[df_train_label.instance_number>150]","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:40:55.617162Z","iopub.execute_input":"2024-05-17T13:40:55.617587Z","iopub.status.idle":"2024-05-17T13:40:55.635751Z","shell.execute_reply.started":"2024-05-17T13:40:55.617529Z","shell.execute_reply":"2024-05-17T13:40:55.634596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# condition vs level\npd.crosstab(df_train_label.condition, df_train_label.level)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:40:55.637224Z","iopub.execute_input":"2024-05-17T13:40:55.638137Z","iopub.status.idle":"2024-05-17T13:40:55.680021Z","shell.execute_reply.started":"2024-05-17T13:40:55.638102Z","shell.execute_reply":"2024-05-17T13:40:55.678765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Descriptions","metadata":{}},{"cell_type":"code","source":"# structure\ndf_train_desc.info()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:40:55.681627Z","iopub.execute_input":"2024-05-17T13:40:55.681993Z","iopub.status.idle":"2024-05-17T13:40:55.695512Z","shell.execute_reply.started":"2024-05-17T13:40:55.681962Z","shell.execute_reply":"2024-05-17T13:40:55.694150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preview\ndf_train_desc.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:40:55.697269Z","iopub.execute_input":"2024-05-17T13:40:55.697706Z","iopub.status.idle":"2024-05-17T13:40:55.714180Z","shell.execute_reply.started":"2024-05-17T13:40:55.697662Z","shell.execute_reply":"2024-05-17T13:40:55.712461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# look at categories\ncounts = df_train_desc.series_description.value_counts()\nprint(counts)\nplt.figure(figsize=(7,3))\ncounts.plot(kind='bar', color=default_color_1)\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:40:55.715783Z","iopub.execute_input":"2024-05-17T13:40:55.716175Z","iopub.status.idle":"2024-05-17T13:40:55.941952Z","shell.execute_reply.started":"2024-05-17T13:40:55.716142Z","shell.execute_reply":"2024-05-17T13:40:55.940480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Combine Tables","metadata":{}},{"cell_type":"code","source":"# join first two tables\ndf_train_step_1 = pd.merge(left=df_train_label, right=df_train_main, how='left', on='study_id').reset_index(drop=True)\ndf_train_step_1.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:40:55.943515Z","iopub.execute_input":"2024-05-17T13:40:55.943973Z","iopub.status.idle":"2024-05-17T13:40:56.074482Z","shell.execute_reply.started":"2024-05-17T13:40:55.943931Z","shell.execute_reply":"2024-05-17T13:40:56.073204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# join with third table\ndf_train = pd.merge(left=df_train_step_1, right=df_train_desc, how='left', on=['study_id', 'series_id']).reset_index(drop=True)\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:40:56.076280Z","iopub.execute_input":"2024-05-17T13:40:56.076689Z","iopub.status.idle":"2024-05-17T13:40:56.188150Z","shell.execute_reply.started":"2024-05-17T13:40:56.076656Z","shell.execute_reply":"2024-05-17T13:40:56.186889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert ids to categorical\ndf_train.study_id = df_train.study_id.astype('category')\ndf_train.series_id = df_train.series_id.astype('category')","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:40:56.189677Z","iopub.execute_input":"2024-05-17T13:40:56.190079Z","iopub.status.idle":"2024-05-17T13:40:56.201157Z","shell.execute_reply.started":"2024-05-17T13:40:56.190046Z","shell.execute_reply":"2024-05-17T13:40:56.199796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='eda'></a>\n# EDA","metadata":{}},{"cell_type":"code","source":"# combined table - basic stats\ndf_train.describe(include='all').T","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:40:56.202729Z","iopub.execute_input":"2024-05-17T13:40:56.203172Z","iopub.status.idle":"2024-05-17T13:40:56.640767Z","shell.execute_reply.started":"2024-05-17T13:40:56.203131Z","shell.execute_reply":"2024-05-17T13:40:56.639515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot coordinates\nsns.jointplot(data=df_train, x='x', y='y', \n              color=default_color_1, alpha=0.25)\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:40:56.642317Z","iopub.execute_input":"2024-05-17T13:40:56.642767Z","iopub.status.idle":"2024-05-17T13:40:58.637142Z","shell.execute_reply.started":"2024-05-17T13:40:56.642733Z","shell.execute_reply":"2024-05-17T13:40:58.635883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot coordinates - colored by condition\nax = sns.jointplot(data=df_train, x='x', y='y', \n                   hue='condition', alpha=0.25)\nplt.legend(bbox_to_anchor=(1.2,1), loc=2)\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:40:58.638715Z","iopub.execute_input":"2024-05-17T13:40:58.639089Z","iopub.status.idle":"2024-05-17T13:41:01.940061Z","shell.execute_reply.started":"2024-05-17T13:40:58.639059Z","shell.execute_reply":"2024-05-17T13:41:01.938790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot coordinates - colored by level\nsns.jointplot(data=df_train, x='x', y='y', \n              hue='level', alpha=0.25)\nplt.legend(bbox_to_anchor=(1.2,1), loc=2)\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:41:01.941436Z","iopub.execute_input":"2024-05-17T13:41:01.941829Z","iopub.status.idle":"2024-05-17T13:41:05.385073Z","shell.execute_reply.started":"2024-05-17T13:41:01.941795Z","shell.execute_reply":"2024-05-17T13:41:05.383511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot coordinates - colored by description\nsns.jointplot(data=df_train, x='x', y='y', \n              hue='series_description', alpha=0.25)\nplt.legend(bbox_to_anchor=(1.2,1), loc=2)\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:41:05.386532Z","iopub.execute_input":"2024-05-17T13:41:05.386913Z","iopub.status.idle":"2024-05-17T13:41:08.679240Z","shell.execute_reply.started":"2024-05-17T13:41:05.386879Z","shell.execute_reply":"2024-05-17T13:41:08.678056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Label distributions","metadata":{}},{"cell_type":"code","source":"labels = df_train_main.columns.drop('study_id').tolist()\nprint(labels)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:41:08.681070Z","iopub.execute_input":"2024-05-17T13:41:08.681439Z","iopub.status.idle":"2024-05-17T13:41:08.688578Z","shell.execute_reply.started":"2024-05-17T13:41:08.681407Z","shell.execute_reply":"2024-05-17T13:41:08.687262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot label distributions\nfor l in labels:\n    plt.figure(figsize=(6,2))\n    counts = df_train_main[l].value_counts(normalize=True)\n    print(counts)\n    counts.plot(kind='bar', color=default_color_1)\n    plt.grid()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:41:08.689893Z","iopub.execute_input":"2024-05-17T13:41:08.690257Z","iopub.status.idle":"2024-05-17T13:41:13.355287Z","shell.execute_reply.started":"2024-05-17T13:41:08.690228Z","shell.execute_reply":"2024-05-17T13:41:13.353900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create frequency table for the labels\nfreqs = pd.DataFrame(labels, columns=['label'])\nfreqs['p1'] = 1.0\nfreqs['p2'] = 0.0\nfreqs['p3'] = 0.0\n\nfor l in labels:\n    rel_counts = df_train_main[l].value_counts(normalize=True)\n    freqs.loc[freqs.label==l, 'p1'] = rel_counts['Normal/Mild']\n    freqs.loc[freqs.label==l, 'p2'] = rel_counts['Moderate']\n    freqs.loc[freqs.label==l, 'p3'] = rel_counts['Severe']","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:41:13.357864Z","iopub.execute_input":"2024-05-17T13:41:13.358323Z","iopub.status.idle":"2024-05-17T13:41:13.431187Z","shell.execute_reply.started":"2024-05-17T13:41:13.358280Z","shell.execute_reply":"2024-05-17T13:41:13.429825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show frequency table\nfreqs","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:41:13.433083Z","iopub.execute_input":"2024-05-17T13:41:13.433503Z","iopub.status.idle":"2024-05-17T13:41:13.457318Z","shell.execute_reply.started":"2024-05-17T13:41:13.433470Z","shell.execute_reply":"2024-05-17T13:41:13.455888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# make combined table available for download\ndf_train.to_csv('df_train.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:41:13.458863Z","iopub.execute_input":"2024-05-17T13:41:13.459295Z","iopub.status.idle":"2024-05-17T13:41:14.948727Z","shell.execute_reply.started":"2024-05-17T13:41:13.459257Z","shell.execute_reply":"2024-05-17T13:41:14.947253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='example'></a>\n# Example","metadata":{}},{"cell_type":"code","source":"my_study = 4003253\ndf_ex = df_train[df_train.study_id==my_study]\ndf_ex","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:41:14.954586Z","iopub.execute_input":"2024-05-17T13:41:14.954955Z","iopub.status.idle":"2024-05-17T13:41:14.998742Z","shell.execute_reply.started":"2024-05-17T13:41:14.954927Z","shell.execute_reply":"2024-05-17T13:41:14.997441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_series = 702807833\ndf_ex1 = df_ex[df_ex.series_id==my_series]\ndf_ex1","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:41:15.000070Z","iopub.execute_input":"2024-05-17T13:41:15.000471Z","iopub.status.idle":"2024-05-17T13:41:15.029982Z","shell.execute_reply.started":"2024-05-17T13:41:15.000438Z","shell.execute_reply":"2024-05-17T13:41:15.028679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create file path and show files\nmy_path = '../input/rsna-2024-lumbar-spine-degenerative-classification/train_images/' + str(my_study) + '/' + str(my_series) + '/'\nfor dirname, _, filenames in os.walk(my_path):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:41:15.031766Z","iopub.execute_input":"2024-05-17T13:41:15.032232Z","iopub.status.idle":"2024-05-17T13:41:15.042547Z","shell.execute_reply.started":"2024-05-17T13:41:15.032190Z","shell.execute_reply":"2024-05-17T13:41:15.041340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(1,15+1):\n    my_file = my_path + str(i) + '.dcm'\n    print(my_file)\n    # load image\n    ds = dicom.dcmread(my_file)\n    # and plot\n    plt.imshow(ds.pixel_array)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:41:15.044000Z","iopub.execute_input":"2024-05-17T13:41:15.044387Z","iopub.status.idle":"2024-05-17T13:41:20.559871Z","shell.execute_reply.started":"2024-05-17T13:41:15.044336Z","shell.execute_reply":"2024-05-17T13:41:20.558457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='sub'></a>\n# Submission","metadata":{}},{"cell_type":"code","source":"# preview test data\ndf_test_desc.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:41:20.561621Z","iopub.execute_input":"2024-05-17T13:41:20.562069Z","iopub.status.idle":"2024-05-17T13:41:20.575538Z","shell.execute_reply.started":"2024-05-17T13:41:20.562026Z","shell.execute_reply":"2024-05-17T13:41:20.574242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preview sample submission\ndf_sub.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:41:20.577253Z","iopub.execute_input":"2024-05-17T13:41:20.577773Z","iopub.status.idle":"2024-05-17T13:41:20.595691Z","shell.execute_reply.started":"2024-05-17T13:41:20.577713Z","shell.execute_reply":"2024-05-17T13:41:20.594233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# setup baseline model just using observed frequencies\nn = df_sub.shape[0]\nfor i in range(0,n):\n    # extract label from id\n    current_label = df_sub.loc[i, 'row_id'].split('_',1)[1]\n    # look up frequencies in frequency table\n    p1 = freqs.loc[freqs.label==current_label, 'p1'].min()\n    p2 = freqs.loc[freqs.label==current_label, 'p2'].min()\n    p3 = freqs.loc[freqs.label==current_label, 'p3'].min()\n    # transfer results to submission table\n    df_sub.loc[i, 'normal_mild'] = p1\n    df_sub.loc[i, 'moderate'] = p2\n    df_sub.loc[i, 'severe'] = p3","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:42:37.830054Z","iopub.execute_input":"2024-05-17T13:42:37.830525Z","iopub.status.idle":"2024-05-17T13:42:37.900041Z","shell.execute_reply.started":"2024-05-17T13:42:37.830494Z","shell.execute_reply":"2024-05-17T13:42:37.898653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preview\ndf_sub","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:43:00.390219Z","iopub.execute_input":"2024-05-17T13:43:00.390639Z","iopub.status.idle":"2024-05-17T13:43:00.407492Z","shell.execute_reply.started":"2024-05-17T13:43:00.390609Z","shell.execute_reply":"2024-05-17T13:43:00.406408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save submission file\ndf_sub.to_csv('submission_freq.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T13:42:41.564205Z","iopub.execute_input":"2024-05-17T13:42:41.564663Z","iopub.status.idle":"2024-05-17T13:42:41.572696Z","shell.execute_reply.started":"2024-05-17T13:42:41.564627Z","shell.execute_reply":"2024-05-17T13:42:41.571578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# simple version, just fixed values\nn = df_sub.shape[0]\nfor i in range(0,n):\n    df_sub.loc[i, 'normal_mild'] = 0.34\n    df_sub.loc[i, 'moderate'] = 0.33\n    df_sub.loc[i, 'severe'] = 0.33","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save submission file\ndf_sub.to_csv('submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]}]}