{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfrom sklearn.utils import resample\n\nimport gc\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-19T13:06:40.788753Z","iopub.execute_input":"2023-04-19T13:06:40.789388Z","iopub.status.idle":"2023-04-19T13:06:41.918665Z","shell.execute_reply.started":"2023-04-19T13:06:40.789351Z","shell.execute_reply":"2023-04-19T13:06:41.917375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_DIR = os.path.join('/kaggle', 'input', 'tlvmc-parkinsons-freezing-gait-prediction')\nTRAIN_DIR = os.path.join(os.path.join(INPUT_DIR, 'train'))\nTEST_DIR = os.path.join(os.path.join(INPUT_DIR, 'test'))","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:06:41.921310Z","iopub.execute_input":"2023-04-19T13:06:41.921794Z","iopub.status.idle":"2023-04-19T13:06:41.930665Z","shell.execute_reply.started":"2023-04-19T13:06:41.921746Z","shell.execute_reply":"2023-04-19T13:06:41.928953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_file_name_to_series_name(file_name):\n    return file_name[:-4]","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:06:41.932405Z","iopub.execute_input":"2023-04-19T13:06:41.932913Z","iopub.status.idle":"2023-04-19T13:06:41.943593Z","shell.execute_reply.started":"2023-04-19T13:06:41.932863Z","shell.execute_reply":"2023-04-19T13:06:41.941989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = {}\nfor d_type in os.listdir(TRAIN_DIR):\n    if d_type != 'defog':\n        continue\n    for fname in os.listdir(os.path.join(TRAIN_DIR, d_type)):\n        train_data[convert_file_name_to_series_name(fname)] = pd.read_csv(os.path.join(TRAIN_DIR, d_type, fname))","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:06:41.948174Z","iopub.execute_input":"2023-04-19T13:06:41.948616Z","iopub.status.idle":"2023-04-19T13:07:09.343689Z","shell.execute_reply.started":"2023-04-19T13:06:41.948576Z","shell.execute_reply":"2023-04-19T13:07:09.342393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfs = [train_data[series_name] for series_name in train_data.keys()]\ndfs = [df.assign(Series=series_name) for series_name, df in train_data.items()]\ndefog_train_df = pd.concat(dfs, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:09.345498Z","iopub.execute_input":"2023-04-19T13:07:09.346186Z","iopub.status.idle":"2023-04-19T13:07:11.012484Z","shell.execute_reply.started":"2023-04-19T13:07:09.346150Z","shell.execute_reply":"2023-04-19T13:07:11.011264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:11.013675Z","iopub.execute_input":"2023-04-19T13:07:11.013992Z","iopub.status.idle":"2023-04-19T13:07:11.041326Z","shell.execute_reply.started":"2023-04-19T13:07:11.013962Z","shell.execute_reply":"2023-04-19T13:07:11.040067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_train_df = defog_train_df[(defog_train_df['Valid'] == True) & (defog_train_df['Task'] == True)]\ndefog_train_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:11.042746Z","iopub.execute_input":"2023-04-19T13:07:11.043073Z","iopub.status.idle":"2023-04-19T13:07:11.741117Z","shell.execute_reply.started":"2023-04-19T13:07:11.043041Z","shell.execute_reply":"2023-04-19T13:07:11.739989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del dfs, train_data\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:11.742389Z","iopub.execute_input":"2023-04-19T13:07:11.742703Z","iopub.status.idle":"2023-04-19T13:07:11.905531Z","shell.execute_reply.started":"2023-04-19T13:07:11.742672Z","shell.execute_reply":"2023-04-19T13:07:11.904117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = {}\nfor d_type in os.listdir(TRAIN_DIR):\n    if d_type != 'tdcsfog':\n        continue\n    for fname in os.listdir(os.path.join(TRAIN_DIR, d_type)):\n        train_data[convert_file_name_to_series_name(fname)] = pd.read_csv(os.path.join(TRAIN_DIR, d_type, fname))\n\ndfs = [train_data[series_name] for series_name in train_data.keys()]\ndfs = [df.assign(Series=series_name) for series_name, df in train_data.items()]\ntdcsfog_train_df = pd.concat(dfs, ignore_index=True)\n\ndel train_data, dfs\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:11.907221Z","iopub.execute_input":"2023-04-19T13:07:11.907707Z","iopub.status.idle":"2023-04-19T13:07:28.712057Z","shell.execute_reply.started":"2023-04-19T13:07:11.907659Z","shell.execute_reply":"2023-04-19T13:07:28.710803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_train_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:28.715977Z","iopub.execute_input":"2023-04-19T13:07:28.716324Z","iopub.status.idle":"2023-04-19T13:07:28.723665Z","shell.execute_reply.started":"2023-04-19T13:07:28.716291Z","shell.execute_reply":"2023-04-19T13:07:28.722486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_train_df['StartHesitation'] = pd.to_numeric(tdcsfog_train_df['StartHesitation'], downcast='integer')\ndefog_train_df['StartHesitation'] = pd.to_numeric(defog_train_df['StartHesitation'], downcast='integer')\ntdcsfog_train_df['Walking'] = pd.to_numeric(tdcsfog_train_df['Walking'], downcast='integer')\ndefog_train_df['Walking'] = pd.to_numeric(defog_train_df['Walking'], downcast='integer')\ntdcsfog_train_df['Turn'] = pd.to_numeric(tdcsfog_train_df['Turn'], downcast='integer')\ndefog_train_df['Turn'] = pd.to_numeric(defog_train_df['Turn'], downcast='integer')\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:28.724865Z","iopub.execute_input":"2023-04-19T13:07:28.725154Z","iopub.status.idle":"2023-04-19T13:07:29.312625Z","shell.execute_reply.started":"2023-04-19T13:07:28.725126Z","shell.execute_reply":"2023-04-19T13:07:29.311373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_meta_df = pd.read_csv(os.path.join(INPUT_DIR, 'defog_metadata.csv'))\ntdcsfog_meta_df = pd.read_csv(os.path.join(INPUT_DIR, 'tdcsfog_metadata.csv'))\nsubjects_df = pd.read_csv(os.path.join(INPUT_DIR, 'subjects.csv'))\ntasks_df = pd.read_csv(os.path.join(INPUT_DIR, 'tasks.csv'))","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:29.314473Z","iopub.execute_input":"2023-04-19T13:07:29.315236Z","iopub.status.idle":"2023-04-19T13:07:29.347852Z","shell.execute_reply.started":"2023-04-19T13:07:29.315188Z","shell.execute_reply":"2023-04-19T13:07:29.346742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_meta_df.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:29.349406Z","iopub.execute_input":"2023-04-19T13:07:29.350065Z","iopub.status.idle":"2023-04-19T13:07:29.360091Z","shell.execute_reply.started":"2023-04-19T13:07:29.350018Z","shell.execute_reply":"2023-04-19T13:07:29.358628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_meta_df['Visit'] = pd.to_numeric(defog_meta_df['Visit'], downcast='integer')\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:29.361991Z","iopub.execute_input":"2023-04-19T13:07:29.363217Z","iopub.status.idle":"2023-04-19T13:07:29.483761Z","shell.execute_reply.started":"2023-04-19T13:07:29.363169Z","shell.execute_reply":"2023-04-19T13:07:29.481377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_meta_df.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:29.487193Z","iopub.execute_input":"2023-04-19T13:07:29.488478Z","iopub.status.idle":"2023-04-19T13:07:29.497105Z","shell.execute_reply.started":"2023-04-19T13:07:29.488434Z","shell.execute_reply":"2023-04-19T13:07:29.496015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_meta_df['Visit'] = pd.to_numeric(tdcsfog_meta_df['Visit'], downcast='integer')\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:29.498597Z","iopub.execute_input":"2023-04-19T13:07:29.498968Z","iopub.status.idle":"2023-04-19T13:07:29.615488Z","shell.execute_reply.started":"2023-04-19T13:07:29.498928Z","shell.execute_reply":"2023-04-19T13:07:29.614334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects_df.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:29.617221Z","iopub.execute_input":"2023-04-19T13:07:29.617708Z","iopub.status.idle":"2023-04-19T13:07:29.629739Z","shell.execute_reply.started":"2023-04-19T13:07:29.617672Z","shell.execute_reply":"2023-04-19T13:07:29.628402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects_df['Visit'] = pd.to_numeric(subjects_df['Visit'], downcast='integer')\nsubjects_df['Age'] = pd.to_numeric(subjects_df['Age'], downcast='integer')\nsubjects_df['YearsSinceDx'] = pd.to_numeric(subjects_df['YearsSinceDx'], downcast='integer')\nsubjects_df['NFOGQ'] = pd.to_numeric(subjects_df['NFOGQ'], downcast='integer')\ngc.collect()\nsubjects_df.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:29.631100Z","iopub.execute_input":"2023-04-19T13:07:29.631456Z","iopub.status.idle":"2023-04-19T13:07:29.757750Z","shell.execute_reply.started":"2023-04-19T13:07:29.631422Z","shell.execute_reply":"2023-04-19T13:07:29.756555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def check_null_values(data_frame):\n    for column in data_frame.columns:\n        column_is_null = data_frame[column].isnull()\n        if column_is_null.any():\n            print('Column:', column, 'has', column_is_null.sum(), 'null values')","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:29.759078Z","iopub.execute_input":"2023-04-19T13:07:29.759489Z","iopub.status.idle":"2023-04-19T13:07:29.771237Z","shell.execute_reply.started":"2023-04-19T13:07:29.759439Z","shell.execute_reply":"2023-04-19T13:07:29.770199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_null_values(defog_meta_df)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:29.772645Z","iopub.execute_input":"2023-04-19T13:07:29.772958Z","iopub.status.idle":"2023-04-19T13:07:29.781591Z","shell.execute_reply.started":"2023-04-19T13:07:29.772928Z","shell.execute_reply":"2023-04-19T13:07:29.780595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_null_values(subjects_df)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:29.783139Z","iopub.execute_input":"2023-04-19T13:07:29.783815Z","iopub.status.idle":"2023-04-19T13:07:29.794802Z","shell.execute_reply.started":"2023-04-19T13:07:29.783779Z","shell.execute_reply":"2023-04-19T13:07:29.792981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_null_values(tasks_df)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:29.798186Z","iopub.execute_input":"2023-04-19T13:07:29.798968Z","iopub.status.idle":"2023-04-19T13:07:29.806660Z","shell.execute_reply.started":"2023-04-19T13:07:29.798927Z","shell.execute_reply":"2023-04-19T13:07:29.805715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tasks_df['Begin'].min()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:29.808062Z","iopub.execute_input":"2023-04-19T13:07:29.808400Z","iopub.status.idle":"2023-04-19T13:07:29.818004Z","shell.execute_reply.started":"2023-04-19T13:07:29.808351Z","shell.execute_reply":"2023-04-19T13:07:29.817052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Filling in null values for subjects_df","metadata":{}},{"cell_type":"code","source":"defog_meta_df['Subject'].isin(subjects_df[subjects_df['Visit'].isnull()]['Subject']).any()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:29.819414Z","iopub.execute_input":"2023-04-19T13:07:29.819786Z","iopub.status.idle":"2023-04-19T13:07:29.833684Z","shell.execute_reply.started":"2023-04-19T13:07:29.819753Z","shell.execute_reply":"2023-04-19T13:07:29.832547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since all the values NULL values for the visit column in subjects_df refer to tdcsfog_metada.csv, we don't do anything since tdcsfog_metada.csv already has a Visit column","metadata":{}},{"cell_type":"code","source":"subjects_df[subjects_df['UPDRSIII_Off'].isnull()]","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:29.835443Z","iopub.execute_input":"2023-04-19T13:07:29.835997Z","iopub.status.idle":"2023-04-19T13:07:29.865663Z","shell.execute_reply.started":"2023-04-19T13:07:29.835952Z","shell.execute_reply":"2023-04-19T13:07:29.864332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects_df['UPDRSIII_On'].corr(subjects_df['UPDRSIII_Off'])","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:29.867224Z","iopub.execute_input":"2023-04-19T13:07:29.867798Z","iopub.status.idle":"2023-04-19T13:07:29.876597Z","shell.execute_reply.started":"2023-04-19T13:07:29.867737Z","shell.execute_reply":"2023-04-19T13:07:29.875455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since we find a high correlation between UPDRSIII_On and UPDRSIII_Off, we can use values in UPDRSIII_On to predict missing values in UPDRSIII_Off.","metadata":{}},{"cell_type":"code","source":"corr = subjects_df['UPDRSIII_On'].corr(subjects_df['UPDRSIII_Off'])\nif corr > 0.5:\n    subjects_df['UPDRSIII_Off'].fillna(subjects_df['UPDRSIII_On']/corr, inplace=True)\n    subjects_df['UPDRSIII_On'].fillna(subjects_df['UPDRSIII_Off']*corr, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:29.878197Z","iopub.execute_input":"2023-04-19T13:07:29.878528Z","iopub.status.idle":"2023-04-19T13:07:29.891204Z","shell.execute_reply.started":"2023-04-19T13:07:29.878498Z","shell.execute_reply":"2023-04-19T13:07:29.890412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_null_values(subjects_df)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:29.900653Z","iopub.execute_input":"2023-04-19T13:07:29.900983Z","iopub.status.idle":"2023-04-19T13:07:29.908503Z","shell.execute_reply.started":"2023-04-19T13:07:29.900953Z","shell.execute_reply":"2023-04-19T13:07:29.907327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Converting categorical features to numerical","metadata":{}},{"cell_type":"code","source":"defog_meta_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:29.910160Z","iopub.execute_input":"2023-04-19T13:07:29.910489Z","iopub.status.idle":"2023-04-19T13:07:29.925359Z","shell.execute_reply.started":"2023-04-19T13:07:29.910459Z","shell.execute_reply":"2023-04-19T13:07:29.924474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_meta_df = defog_meta_df.replace({'Medication':{'on' : 1, 'off' : 0}})\ndefog_meta_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:29.926655Z","iopub.execute_input":"2023-04-19T13:07:29.926969Z","iopub.status.idle":"2023-04-19T13:07:29.940406Z","shell.execute_reply.started":"2023-04-19T13:07:29.926940Z","shell.execute_reply":"2023-04-19T13:07:29.939333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_train_df = defog_train_df.replace({'Task' : {True : 1, False : 0},'Valid' : {True : 1, False : 0}})","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:29.941589Z","iopub.execute_input":"2023-04-19T13:07:29.942089Z","iopub.status.idle":"2023-04-19T13:07:31.896238Z","shell.execute_reply.started":"2023-04-19T13:07:29.942058Z","shell.execute_reply":"2023-04-19T13:07:31.895012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:31.897721Z","iopub.execute_input":"2023-04-19T13:07:31.898082Z","iopub.status.idle":"2023-04-19T13:07:31.913866Z","shell.execute_reply.started":"2023-04-19T13:07:31.898049Z","shell.execute_reply":"2023-04-19T13:07:31.912629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects_df.shape, len(subjects_df['Subject'].unique())","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:31.915030Z","iopub.execute_input":"2023-04-19T13:07:31.915365Z","iopub.status.idle":"2023-04-19T13:07:31.927321Z","shell.execute_reply.started":"2023-04-19T13:07:31.915322Z","shell.execute_reply":"2023-04-19T13:07:31.926147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Above shows that there are also duplicates for subject IDs. We'll remove duplicates such that the row with greater values for \"Visits\" column is kept. We also convert the 'Sex' column to 0 and 1, for male and female respectively.","metadata":{}},{"cell_type":"code","source":"subjects_df = subjects_df.replace({'Sex' : {'M' : 0, 'F' : 1}})","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:31.929012Z","iopub.execute_input":"2023-04-19T13:07:31.929410Z","iopub.status.idle":"2023-04-19T13:07:31.939259Z","shell.execute_reply.started":"2023-04-19T13:07:31.929377Z","shell.execute_reply":"2023-04-19T13:07:31.938204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects_df.sort_values(by='Visit', ascending=False, inplace=True)\nsubjects_df.drop_duplicates(subset=['Subject'], keep='first', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:31.941215Z","iopub.execute_input":"2023-04-19T13:07:31.941753Z","iopub.status.idle":"2023-04-19T13:07:31.952032Z","shell.execute_reply.started":"2023-04-19T13:07:31.941715Z","shell.execute_reply":"2023-04-19T13:07:31.950681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Merging dataframes","metadata":{}},{"cell_type":"markdown","source":"### Merging dfog","metadata":{}},{"cell_type":"code","source":"defog_subject_meta_merged_df = pd.merge(subjects_df, defog_meta_df, on='Subject')\ndefog_subject_meta_merged_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:31.953385Z","iopub.execute_input":"2023-04-19T13:07:31.953794Z","iopub.status.idle":"2023-04-19T13:07:31.984152Z","shell.execute_reply.started":"2023-04-19T13:07:31.953761Z","shell.execute_reply":"2023-04-19T13:07:31.983017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_subject_meta_merged_df[defog_subject_meta_merged_df[['Subject', 'Id']].duplicated()]","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:31.985869Z","iopub.execute_input":"2023-04-19T13:07:31.986353Z","iopub.status.idle":"2023-04-19T13:07:32.002215Z","shell.execute_reply.started":"2023-04-19T13:07:31.986295Z","shell.execute_reply":"2023-04-19T13:07:32.001334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(defog_subject_meta_merged_df['Visit_x'] >= defog_subject_meta_merged_df['Visit_y']).all()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:32.003547Z","iopub.execute_input":"2023-04-19T13:07:32.003848Z","iopub.status.idle":"2023-04-19T13:07:32.011753Z","shell.execute_reply.started":"2023-04-19T13:07:32.003820Z","shell.execute_reply":"2023-04-19T13:07:32.010525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We drop Visit_y, and rename Visit_x to Visit. This is because Visit_y is coming from defog_metadata and doesn't record all visits.","metadata":{}},{"cell_type":"code","source":"defog_subject_meta_merged_df.drop(columns=['Visit_y'], inplace=True)\ndefog_subject_meta_merged_df.rename(columns={'Visit_x':'Visit'}, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:32.013095Z","iopub.execute_input":"2023-04-19T13:07:32.014134Z","iopub.status.idle":"2023-04-19T13:07:32.026009Z","shell.execute_reply.started":"2023-04-19T13:07:32.014098Z","shell.execute_reply":"2023-04-19T13:07:32.024856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_subject_meta_merged_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:32.027512Z","iopub.execute_input":"2023-04-19T13:07:32.027834Z","iopub.status.idle":"2023-04-19T13:07:32.049458Z","shell.execute_reply.started":"2023-04-19T13:07:32.027803Z","shell.execute_reply":"2023-04-19T13:07:32.048219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_subject_meta_merged_df['Sex'] = pd.to_numeric(defog_subject_meta_merged_df['Sex'], downcast='integer')\ndefog_subject_meta_merged_df['Visit'] = pd.to_numeric(defog_subject_meta_merged_df['Visit'], downcast='integer')\ndefog_subject_meta_merged_df.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:32.051386Z","iopub.execute_input":"2023-04-19T13:07:32.052198Z","iopub.status.idle":"2023-04-19T13:07:32.064143Z","shell.execute_reply.started":"2023-04-19T13:07:32.052147Z","shell.execute_reply":"2023-04-19T13:07:32.062908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we merge defog_subject_meta_merged_df with defog_train_df","metadata":{}},{"cell_type":"code","source":"defog_train_merged_df = pd.merge(defog_subject_meta_merged_df, defog_train_df, right_on='Series', left_on='Id')\ndefog_train_merged_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:32.065736Z","iopub.execute_input":"2023-04-19T13:07:32.066521Z","iopub.status.idle":"2023-04-19T13:07:33.242433Z","shell.execute_reply.started":"2023-04-19T13:07:32.066474Z","shell.execute_reply":"2023-04-19T13:07:33.241569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_train_merged_df['Medication'] = pd.to_numeric(defog_train_merged_df['Medication'], downcast='integer')\ndefog_train_merged_df['Task'] = pd.to_numeric(defog_train_merged_df['Task'], downcast='integer')\ndefog_train_merged_df['Valid'] = pd.to_numeric(defog_train_merged_df['Valid'], downcast='integer')\ngc.collect()\ndefog_train_merged_df.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:33.243510Z","iopub.execute_input":"2023-04-19T13:07:33.244329Z","iopub.status.idle":"2023-04-19T13:07:33.495421Z","shell.execute_reply.started":"2023-04-19T13:07:33.244268Z","shell.execute_reply":"2023-04-19T13:07:33.494329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_train_merged_df.drop(columns=['Series', 'Valid', 'Task', 'Id', 'Subject'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:33.496673Z","iopub.execute_input":"2023-04-19T13:07:33.496995Z","iopub.status.idle":"2023-04-19T13:07:34.208227Z","shell.execute_reply.started":"2023-04-19T13:07:33.496964Z","shell.execute_reply":"2023-04-19T13:07:34.207053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_train_merged_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:34.209663Z","iopub.execute_input":"2023-04-19T13:07:34.210004Z","iopub.status.idle":"2023-04-19T13:07:34.228807Z","shell.execute_reply.started":"2023-04-19T13:07:34.209971Z","shell.execute_reply":"2023-04-19T13:07:34.227696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del defog_train_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:34.230100Z","iopub.execute_input":"2023-04-19T13:07:34.230427Z","iopub.status.idle":"2023-04-19T13:07:34.359890Z","shell.execute_reply.started":"2023-04-19T13:07:34.230397Z","shell.execute_reply":"2023-04-19T13:07:34.358981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_train_merged_df['Time'] = defog_train_merged_df['Time'].astype(np.float64)\ndefog_train_merged_df['Time'] = defog_train_merged_df['Time']/100","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:34.362206Z","iopub.execute_input":"2023-04-19T13:07:34.362768Z","iopub.status.idle":"2023-04-19T13:07:34.402797Z","shell.execute_reply.started":"2023-04-19T13:07:34.362719Z","shell.execute_reply":"2023-04-19T13:07:34.401849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Merging tdcsfog","metadata":{}},{"cell_type":"code","source":"tdcsfog_subject_meta_merged_df = pd.merge(subjects_df, tdcsfog_meta_df, on='Subject')\ntdcsfog_subject_meta_merged_df['Visit_x'] = tdcsfog_subject_meta_merged_df['Visit_y']\ntdcsfog_subject_meta_merged_df.drop(columns=['Visit_y', 'Test'], inplace=True)\ntdcsfog_subject_meta_merged_df.rename(columns = {'Visit_x' : 'Visit'}, inplace=True)\ntdcsfog_subject_meta_merged_df = tdcsfog_subject_meta_merged_df.replace({'Medication':{'on' : 1, 'off' : 0}})\ntdcsfog_subject_meta_merged_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:34.404172Z","iopub.execute_input":"2023-04-19T13:07:34.404487Z","iopub.status.idle":"2023-04-19T13:07:34.431983Z","shell.execute_reply.started":"2023-04-19T13:07:34.404458Z","shell.execute_reply":"2023-04-19T13:07:34.430851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_subject_meta_merged_df['Sex'] = pd.to_numeric(tdcsfog_subject_meta_merged_df['Sex'], downcast='integer')\ntdcsfog_subject_meta_merged_df['Visit'] = pd.to_numeric(tdcsfog_subject_meta_merged_df['Visit'], downcast='integer')\ntdcsfog_subject_meta_merged_df['Medication'] = pd.to_numeric(tdcsfog_subject_meta_merged_df['Medication'], downcast='integer')\ngc.collect()\ntdcsfog_subject_meta_merged_df.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:34.433330Z","iopub.execute_input":"2023-04-19T13:07:34.433646Z","iopub.status.idle":"2023-04-19T13:07:34.554443Z","shell.execute_reply.started":"2023-04-19T13:07:34.433617Z","shell.execute_reply":"2023-04-19T13:07:34.553625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_train_merged_df = pd.merge(tdcsfog_subject_meta_merged_df, tdcsfog_train_df, right_on='Series', left_on='Id')\ntdcsfog_train_merged_df.drop(columns=['Series', 'Id', 'Subject'], inplace=True)\ntdcsfog_train_merged_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:34.555341Z","iopub.execute_input":"2023-04-19T13:07:34.555662Z","iopub.status.idle":"2023-04-19T13:07:37.795438Z","shell.execute_reply.started":"2023-04-19T13:07:34.555623Z","shell.execute_reply":"2023-04-19T13:07:37.794377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_train_merged_df.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:37.796932Z","iopub.execute_input":"2023-04-19T13:07:37.797400Z","iopub.status.idle":"2023-04-19T13:07:37.807252Z","shell.execute_reply.started":"2023-04-19T13:07:37.797353Z","shell.execute_reply":"2023-04-19T13:07:37.806085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del tdcsfog_train_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:37.808752Z","iopub.execute_input":"2023-04-19T13:07:37.809088Z","iopub.status.idle":"2023-04-19T13:07:37.956280Z","shell.execute_reply.started":"2023-04-19T13:07:37.809059Z","shell.execute_reply":"2023-04-19T13:07:37.955040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_train_merged_df['Time'] = tdcsfog_train_merged_df['Time'].astype(np.float64)\ntdcsfog_train_merged_df['Time'] = tdcsfog_train_merged_df['Time']/128","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:37.957433Z","iopub.execute_input":"2023-04-19T13:07:37.957869Z","iopub.status.idle":"2023-04-19T13:07:38.024868Z","shell.execute_reply.started":"2023-04-19T13:07:37.957822Z","shell.execute_reply":"2023-04-19T13:07:38.023681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Merging tdcsfog & defog","metadata":{}},{"cell_type":"code","source":"train_df = pd.concat([defog_train_merged_df, tdcsfog_train_merged_df])\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:38.026567Z","iopub.execute_input":"2023-04-19T13:07:38.027029Z","iopub.status.idle":"2023-04-19T13:07:38.526238Z","shell.execute_reply.started":"2023-04-19T13:07:38.026982Z","shell.execute_reply":"2023-04-19T13:07:38.525249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_dist = [len(train_df[train_df['Walking'] == 1]), len(train_df[train_df['Turn'] == 1]), len(train_df[train_df['StartHesitation'] == 1])]\nclass_dist, np.sum(class_dist)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:38.527792Z","iopub.execute_input":"2023-04-19T13:07:38.528142Z","iopub.status.idle":"2023-04-19T13:07:39.289712Z","shell.execute_reply.started":"2023-04-19T13:07:38.528107Z","shell.execute_reply":"2023-04-19T13:07:39.288887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_df[(train_df['Walking'] == 0) & (train_df['Turn'] == 0) & (train_df['StartHesitation'] == 0)])","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:39.291022Z","iopub.execute_input":"2023-04-19T13:07:39.291571Z","iopub.status.idle":"2023-04-19T13:07:39.681681Z","shell.execute_reply.started":"2023-04-19T13:07:39.291538Z","shell.execute_reply":"2023-04-19T13:07:39.680860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:39.682957Z","iopub.execute_input":"2023-04-19T13:07:39.683525Z","iopub.status.idle":"2023-04-19T13:07:39.692502Z","shell.execute_reply.started":"2023-04-19T13:07:39.683491Z","shell.execute_reply":"2023-04-19T13:07:39.691229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del defog_train_merged_df, tdcsfog_train_merged_df, subjects_df, defog_meta_df, tdcsfog_meta_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:39.694239Z","iopub.execute_input":"2023-04-19T13:07:39.694646Z","iopub.status.idle":"2023-04-19T13:07:39.813118Z","shell.execute_reply.started":"2023-04-19T13:07:39.694614Z","shell.execute_reply":"2023-04-19T13:07:39.812322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df = train_df.sample(frac=1).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:39.814651Z","iopub.execute_input":"2023-04-19T13:07:39.815643Z","iopub.status.idle":"2023-04-19T13:07:39.823086Z","shell.execute_reply.started":"2023-04-19T13:07:39.815592Z","shell.execute_reply":"2023-04-19T13:07:39.821971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in train_df.columns:\n    corrs = [train_df[col].corr(train_df['StartHesitation']),  train_df[col].corr(train_df['Turn']), train_df[col].corr(train_df['Walking'])]\n    print('corrs between target and col:', col, 'are:', corrs, np.array(corrs).mean())","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:39.824400Z","iopub.execute_input":"2023-04-19T13:07:39.825345Z","iopub.status.idle":"2023-04-19T13:07:50.210403Z","shell.execute_reply.started":"2023-04-19T13:07:39.825311Z","shell.execute_reply":"2023-04-19T13:07:50.209089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df.drop(columns=['Sex', 'UPDRSIII_On', 'UPDRSIII_Off', 'Medication', 'Age'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:50.211736Z","iopub.execute_input":"2023-04-19T13:07:50.212146Z","iopub.status.idle":"2023-04-19T13:07:50.216794Z","shell.execute_reply.started":"2023-04-19T13:07:50.212104Z","shell.execute_reply":"2023-04-19T13:07:50.215560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:50.218099Z","iopub.execute_input":"2023-04-19T13:07:50.219170Z","iopub.status.idle":"2023-04-19T13:07:50.341462Z","shell.execute_reply.started":"2023-04-19T13:07:50.219133Z","shell.execute_reply":"2023-04-19T13:07:50.340250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:50.342498Z","iopub.execute_input":"2023-04-19T13:07:50.342815Z","iopub.status.idle":"2023-04-19T13:07:50.367560Z","shell.execute_reply.started":"2023-04-19T13:07:50.342784Z","shell.execute_reply":"2023-04-19T13:07:50.366410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"y = train_df[['StartHesitation', 'Turn' , 'Walking']].to_numpy()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:50.368873Z","iopub.execute_input":"2023-04-19T13:07:50.369192Z","iopub.status.idle":"2023-04-19T13:07:50.395730Z","shell.execute_reply.started":"2023-04-19T13:07:50.369161Z","shell.execute_reply":"2023-04-19T13:07:50.394374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_df.drop(columns=['StartHesitation', 'Walking', 'Turn']).to_numpy()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:50.397400Z","iopub.execute_input":"2023-04-19T13:07:50.398119Z","iopub.status.idle":"2023-04-19T13:07:51.084424Z","shell.execute_reply.started":"2023-04-19T13:07:50.398070Z","shell.execute_reply":"2023-04-19T13:07:51.083336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from imblearn.under_sampling import RandomUnderSampler\n\n# rus = RandomUnderSampler(random_state=42, replacement=True)# fit predictor and target variable\n# X, y = rus.fit_resample(X, y)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:51.086231Z","iopub.execute_input":"2023-04-19T13:07:51.086589Z","iopub.status.idle":"2023-04-19T13:07:51.090714Z","shell.execute_reply.started":"2023-04-19T13:07:51.086556Z","shell.execute_reply":"2023-04-19T13:07:51.089784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:51.092155Z","iopub.execute_input":"2023-04-19T13:07:51.092610Z","iopub.status.idle":"2023-04-19T13:07:51.215246Z","shell.execute_reply.started":"2023-04-19T13:07:51.092577Z","shell.execute_reply":"2023-04-19T13:07:51.213984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nfrom xgboost import XGBClassifier, XGBRegressor\nfrom sklearn.model_selection import train_test_split\n\n# Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=42)\n\n# Fit a XGB model\nrf = RandomForestRegressor(n_estimators=60, random_state=42, max_depth=7)\nrf.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:07:51.216696Z","iopub.execute_input":"2023-04-19T13:07:51.217019Z","iopub.status.idle":"2023-04-19T13:55:37.475380Z","shell.execute_reply.started":"2023-04-19T13:07:51.216988Z","shell.execute_reply":"2023-04-19T13:55:37.473872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model performance on the testing data\ny_preds = rf.predict(X_test).clip(0.0, 1.0)\nlabels = []\nfor y_pred in y_preds:\n    if y_pred.max() < 0.5:\n        labels.append(np.array([0, 0, 0]))\n    else:\n        label = np.array([0, 0, 0])\n        label[np.argmax(y_pred)] = 1\n        labels.append(label)\nlabels = np.array(labels)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:55:37.477045Z","iopub.execute_input":"2023-04-19T13:55:37.477848Z","iopub.status.idle":"2023-04-19T13:56:06.503459Z","shell.execute_reply.started":"2023-04-19T13:55:37.477804Z","shell.execute_reply":"2023-04-19T13:56:06.502233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import metrics\nprint('test accuracy:', metrics.accuracy_score(y_test, labels))","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:56:06.505491Z","iopub.execute_input":"2023-04-19T13:56:06.505849Z","iopub.status.idle":"2023-04-19T13:56:07.168250Z","shell.execute_reply.started":"2023-04-19T13:56:06.505815Z","shell.execute_reply":"2023-04-19T13:56:07.166947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(metrics.average_precision_score(y_test, y_preds))","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-04-19T13:56:07.169648Z","iopub.execute_input":"2023-04-19T13:56:07.169969Z","iopub.status.idle":"2023-04-19T13:56:09.293236Z","shell.execute_reply.started":"2023-04-19T13:56:07.169936Z","shell.execute_reply":"2023-04-19T13:56:09.291999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del X, y, X_train, X_test, y_train, y_test\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:56:09.294677Z","iopub.execute_input":"2023-04-19T13:56:09.295107Z","iopub.status.idle":"2023-04-19T13:56:09.303562Z","shell.execute_reply.started":"2023-04-19T13:56:09.295059Z","shell.execute_reply":"2023-04-19T13:56:09.302336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test","metadata":{}},{"cell_type":"code","source":"submission_df = pd.DataFrame(columns=['Id', 'StartHesitation', 'Turn', 'Walking'])\ntest_result_dfs = []\nfor d_type in os.listdir(TEST_DIR):\n    for fname in os.listdir(os.path.join(TEST_DIR, d_type)):\n        test_data_df = pd.read_csv(os.path.join(TEST_DIR, d_type, fname))\n        test_data_df['Series'] = convert_file_name_to_series_name(fname)\n        \n        if d_type == 'defog':\n            test_data_df = pd.merge(defog_subject_meta_merged_df, test_data_df, right_on='Series', left_on='Id')\n            time = list(test_data_df['Time'])\n            test_data_df['Time'] = test_data_df['Time'].astype(np.float64)\n            test_data_df['Time'] = test_data_df['Time']/100\n        else:\n            test_data_df = pd.merge(tdcsfog_subject_meta_merged_df, test_data_df, right_on='Series', left_on='Id')\n            time = list(test_data_df['Time'])\n            test_data_df['Time'] = test_data_df['Time'].astype(np.float64)\n            test_data_df['Time'] = test_data_df['Time']/128\n            \n        test_data_df.drop(columns=['Series', 'Id', 'Subject'], inplace=True)\n        \n        result_df = pd.DataFrame(rf.predict(test_data_df.to_numpy()).clip(0.0, 1.0), columns=['StartHesitation', 'Turn', 'Walking'])\n        result_df.insert(0, 'Id', [convert_file_name_to_series_name(fname) + '_' + str(int(time[i])) for i in range(len(time))])\n        test_result_dfs.append(result_df)\npd.concat(test_result_dfs).to_csv(os.path.join('/kaggle', 'working', 'submission.csv'), index=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T13:56:09.305084Z","iopub.execute_input":"2023-04-19T13:56:09.305513Z","iopub.status.idle":"2023-04-19T13:56:12.051452Z","shell.execute_reply.started":"2023-04-19T13:56:09.305476Z","shell.execute_reply":"2023-04-19T13:56:12.050306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}