{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\n\nThe goal of this competition is to detect freezing of gait (FOG), a debilitating symptom that afflicts many people with Parkinson’s disease. It is requred to develop a machine learning model trained on data collected from a wearable 3D lower back sensor to better understand when and why FOG episodes occur.","metadata":{}},{"cell_type":"code","source":"# import libraries\nimport os\nimport tqdm\nimport glob\nimport numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:30.887382Z","iopub.execute_input":"2023-05-29T10:16:30.888872Z","iopub.status.idle":"2023-05-29T10:16:30.904092Z","shell.execute_reply.started":"2023-05-29T10:16:30.888819Z","shell.execute_reply":"2023-05-29T10:16:30.902500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# parent directory\npdir = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction'","metadata":{"_uuid":"2196131a-5746-46bb-be50-cda74634e26a","_cell_guid":"68bb70a7-5168-480e-ac47-58c0b4c1d4c0","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-29T10:16:30.911973Z","iopub.execute_input":"2023-05-29T10:16:30.912366Z","iopub.status.idle":"2023-05-29T10:16:30.921925Z","shell.execute_reply.started":"2023-05-29T10:16:30.912326Z","shell.execute_reply":"2023-05-29T10:16:30.920408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nts = time.time()\nimport datetime \ndatetime.datetime.fromtimestamp(ts).strftime('%Y-%m-%d %H:%M:%S')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Prepare training data","metadata":{}},{"cell_type":"code","source":"# read tdcs meta data\ndf_tdcs_meta = pd.read_csv(os.path.join(pdir, 'tdcsfog_metadata.csv'))\ndf_tdcs_meta.head()","metadata":{"_uuid":"5de3cc90-900d-46e0-a429-e6d7d0d56563","_cell_guid":"c7e79e61-783b-4519-b49a-30a234bddff6","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-29T10:16:30.923870Z","iopub.execute_input":"2023-05-29T10:16:30.924910Z","iopub.status.idle":"2023-05-29T10:16:30.961177Z","shell.execute_reply.started":"2023-05-29T10:16:30.924827Z","shell.execute_reply":"2023-05-29T10:16:30.959988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read dfog meta data\ndf_defog_meta = pd.read_csv(os.path.join(pdir, 'defog_metadata.csv'))\ndf_defog_meta.head()","metadata":{"_uuid":"49dffe7d-2b15-46a8-aad9-194a7d7295d7","_cell_guid":"89d4fb2d-cebe-4770-81c5-0c23963de29e","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-29T10:16:30.965815Z","iopub.execute_input":"2023-05-29T10:16:30.966259Z","iopub.status.idle":"2023-05-29T10:16:30.984706Z","shell.execute_reply.started":"2023-05-29T10:16:30.966218Z","shell.execute_reply":"2023-05-29T10:16:30.983741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read patient data\ndf_subjects = pd.read_csv(os.path.join(pdir, 'subjects.csv'))\ndf_subjects.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:30.986429Z","iopub.execute_input":"2023-05-29T10:16:30.987099Z","iopub.status.idle":"2023-05-29T10:16:31.013191Z","shell.execute_reply.started":"2023-05-29T10:16:30.987058Z","shell.execute_reply":"2023-05-29T10:16:31.011685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load tdcsfog data\n# list of all tdcsfog csv file path\ntdcs_file_path = glob.glob(os.path.join(pdir, 'train', 'tdcsfog', '*.csv'), recursive=True)","metadata":{"_uuid":"122ff7ef-77f3-48de-9c41-4d70a556358a","_cell_guid":"36d3eeed-7074-46e5-8312-16581a3d0750","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-29T10:16:31.015414Z","iopub.execute_input":"2023-05-29T10:16:31.015972Z","iopub.status.idle":"2023-05-29T10:16:31.028661Z","shell.execute_reply.started":"2023-05-29T10:16:31.015888Z","shell.execute_reply":"2023-05-29T10:16:31.027321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#############\n#Limit the number of files to be read in order to reduce the time required for model training.\n# remove this when training with full data\n#tdcs_file_path = tdcs_file_path[::25] \nprint(f'the number of files to be read: {len(tdcs_file_path)}')","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:31.030178Z","iopub.execute_input":"2023-05-29T10:16:31.031339Z","iopub.status.idle":"2023-05-29T10:16:31.039414Z","shell.execute_reply.started":"2023-05-29T10:16:31.031283Z","shell.execute_reply":"2023-05-29T10:16:31.037949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize a DataFrame to combine data from multiple CSV files.\ndf_tdcs = pd.DataFrame()\n\n# load tdcsfog time series in combination with metadata.\nfor fp in tqdm.tqdm(tdcs_file_path):    \n    \n    # load data into a variable 'tmp'.\n    tmp = pd.read_csv(fp)\n    \n    # get file Id from csv file name.\n    file_id = os.path.basename(fp).replace(\".csv\", \"\")\n    \n    # get subject Id.\n    subject = df_tdcs_meta.loc[df_tdcs_meta['Id'] == file_id, 'Subject'].iloc[0]\n    \n    # add metadata.\n    tmp['Medication'] = df_tdcs_meta.loc[df_tdcs_meta['Id'] == file_id, 'Medication'].iloc[0]\n    tmp['Subject'] = df_subjects.loc[df_subjects['Subject'] == subject, 'Subject'].iloc[0]\n    tmp['Age'] = df_subjects.loc[df_subjects['Subject'] == subject, 'Age'].iloc[0]\n    tmp['Sex'] = df_subjects.loc[df_subjects['Subject'] == subject, 'Sex'].iloc[0]\n    tmp['YearsSinceDx'] = df_subjects.loc[df_subjects['Subject'] == subject, 'YearsSinceDx'].iloc[0]\n    tmp['NFOGQ'] =df_subjects.loc[df_subjects['Subject'] == subject, 'NFOGQ'].iloc[0]\n    \n    # insert lags\n #   tmp['AccV_lag3'] = tmp.AccV.shift(3)\n    tmp['AccV_lag2'] = tmp.AccV.shift(2)\n    tmp['AccV_lag1'] = tmp.AccV.shift(1)\n#    tmp['AccML_lag3'] = tmp.AccML.shift(3)\n    tmp['AccML_lag2'] = tmp.AccML.shift(2)\n    tmp['AccML_lag1'] = tmp.AccML.shift(1)\n#    tmp['AccAP_lag3'] = tmp.AccAP.shift(3)\n    tmp['AccAP_lag2'] = tmp.AccAP.shift(2)\n    tmp['AccAP_lag1'] = tmp.AccAP.shift(1)\n    \n    # concat the data\n    df_tdcs = pd.concat([df_tdcs, tmp]).reset_index(drop=True)\n\n#print(df_tdcs)","metadata":{"_uuid":"959929af-0416-49a8-ba19-3c782cba6b75","_cell_guid":"587ccc9c-9aa4-4e0d-8e0d-fde988fae76c","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-29T10:16:31.041422Z","iopub.execute_input":"2023-05-29T10:16:31.042380Z","iopub.status.idle":"2023-05-29T10:16:33.342435Z","shell.execute_reply.started":"2023-05-29T10:16:31.042322Z","shell.execute_reply":"2023-05-29T10:16:33.341083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_tdcs.head()","metadata":{"_uuid":"7a76c6cd-1fd3-4514-acf8-bd04d409b71f","_cell_guid":"f1a70a00-f11c-48f1-9693-2be6be49545b","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-29T10:16:33.344004Z","iopub.execute_input":"2023-05-29T10:16:33.345036Z","iopub.status.idle":"2023-05-29T10:16:33.373547Z","shell.execute_reply.started":"2023-05-29T10:16:33.344991Z","shell.execute_reply":"2023-05-29T10:16:33.372076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load defog data\n# list of all tdcsfog csv file path\ndefog_file_path = glob.glob(os.path.join(pdir, 'train', 'defog', '*.csv'), recursive=True)","metadata":{"_uuid":"3c67b582-2c33-4b4a-9b2c-6389186cafb6","_cell_guid":"87ef3063-051f-4ecd-b0e2-6d2aa12769b5","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-29T10:16:33.375277Z","iopub.execute_input":"2023-05-29T10:16:33.375751Z","iopub.status.idle":"2023-05-29T10:16:33.383083Z","shell.execute_reply.started":"2023-05-29T10:16:33.375709Z","shell.execute_reply":"2023-05-29T10:16:33.381715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#############\n# Limit the number of files to be read in order to reduce the time required for model training.\n# remove this when training with full data\n#defog_file_path = defog_file_path[::20]\nprint(f'the number of files to be read: {len(defog_file_path)}')","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:33.384976Z","iopub.execute_input":"2023-05-29T10:16:33.385426Z","iopub.status.idle":"2023-05-29T10:16:33.397210Z","shell.execute_reply.started":"2023-05-29T10:16:33.385384Z","shell.execute_reply":"2023-05-29T10:16:33.395669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize a DataFrame to combine data from multiple CSV files.\ndf_defog = pd.DataFrame()\nfor fp in tqdm.tqdm(defog_file_path):\n    # load data into a variable 'tmp'.\n    tmp = pd.read_csv(fp)\n    \n    # get file Id from csv file name.\n    file_id = os.path.basename(fp).replace(\".csv\", \"\")\n    \n    # get subject Id.\n    subject = df_defog_meta.loc[df_defog_meta['Id'] == file_id, 'Subject'].iloc[0]\n    \n    # add metadata.\n    tmp['Medication'] = df_defog_meta.loc[df_defog_meta['Id'] == file_id, 'Medication'].iloc[0]\n    tmp['Subject'] = df_subjects.loc[df_subjects['Subject'] == subject, 'Subject'].iloc[0]\n    tmp['Age'] = df_subjects.loc[df_subjects['Subject'] == subject, 'Age'].iloc[0]\n    tmp['Sex'] = df_subjects.loc[df_subjects['Subject'] == subject, 'Sex'].iloc[0]\n    tmp['YearsSinceDx'] = df_subjects.loc[df_subjects['Subject'] == subject, 'YearsSinceDx'].iloc[0]\n    tmp['NFOGQ'] =df_subjects.loc[df_subjects['Subject'] == subject, 'NFOGQ'].iloc[0]\n    \n    # extract data from the time period where Valid and Task are both True.\n    tmp = tmp[(tmp['Valid'] == True) & (tmp['Task']==True)]\n    tmp = tmp.drop(['Valid', 'Task'], axis=1)\n    \n    # insert lags\n #   tmp['AccV_lag3'] = tmp.AccV.shift(3)\n    tmp['AccV_lag2'] = tmp.AccV.shift(2)\n    tmp['AccV_lag1'] = tmp.AccV.shift(1)\n#    tmp['AccML_lag3'] = tmp.AccML.shift(3)\n    tmp['AccML_lag2'] = tmp.AccML.shift(2)\n    tmp['AccML_lag1'] = tmp.AccML.shift(1)\n#    tmp['AccAP_lag3'] = tmp.AccAP.shift(3)\n    tmp['AccAP_lag2'] = tmp.AccAP.shift(2)\n    tmp['AccAP_lag1'] = tmp.AccAP.shift(1)\n    \n    # concat the data\n    df_defog = pd.concat([df_defog, tmp]).reset_index(drop=True)","metadata":{"_uuid":"8725c2b6-a496-4e42-aaa9-b468b045bf89","_cell_guid":"33ca7a88-9330-468b-a2a7-8aea8c179047","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-29T10:16:33.398821Z","iopub.execute_input":"2023-05-29T10:16:33.399193Z","iopub.status.idle":"2023-05-29T10:16:34.469037Z","shell.execute_reply.started":"2023-05-29T10:16:33.399160Z","shell.execute_reply":"2023-05-29T10:16:34.467671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check the contents of the df_defog\ndf_defog.head()","metadata":{"_uuid":"b1dc546d-e781-4aa4-a647-df65ad15aa30","_cell_guid":"9f2155ce-2316-42c5-b948-478d340aa5d9","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-29T10:16:34.476685Z","iopub.execute_input":"2023-05-29T10:16:34.477155Z","iopub.status.idle":"2023-05-29T10:16:34.503355Z","shell.execute_reply.started":"2023-05-29T10:16:34.477113Z","shell.execute_reply":"2023-05-29T10:16:34.501938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prepare the training data\n# concat tdcs and defog data.\ndf_train = pd.concat([df_tdcs, df_defog]).reset_index(drop=True)\ndf_train.head()","metadata":{"_uuid":"3e8c59b1-26c8-4363-8c77-1b80e2be3795","_cell_guid":"4a3ad625-d594-4acd-b5f2-a5755c2590a1","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-29T10:16:34.505311Z","iopub.execute_input":"2023-05-29T10:16:34.505837Z","iopub.status.idle":"2023-05-29T10:16:34.687216Z","shell.execute_reply.started":"2023-05-29T10:16:34.505784Z","shell.execute_reply":"2023-05-29T10:16:34.685920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# encode string columns into 0/1 format\ndf_train['Medication'] = np.where(df_train['Medication']=='on', 1, 0)\ndf_train['Sex'] = np.where(df_train['Sex']=='M', 1, 0)\ndf_train.head()","metadata":{"_uuid":"ecb40426-38e4-4b94-b519-62a9504e006f","_cell_guid":"19a02577-4bea-42cc-9580-9b7350cf3373","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-29T10:16:34.688999Z","iopub.execute_input":"2023-05-29T10:16:34.689403Z","iopub.status.idle":"2023-05-29T10:16:34.860572Z","shell.execute_reply.started":"2023-05-29T10:16:34.689362Z","shell.execute_reply":"2023-05-29T10:16:34.859296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train=df_train.fillna(0)\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:34.861963Z","iopub.execute_input":"2023-05-29T10:16:34.862315Z","iopub.status.idle":"2023-05-29T10:16:34.976140Z","shell.execute_reply.started":"2023-05-29T10:16:34.862281Z","shell.execute_reply":"2023-05-29T10:16:34.974868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split data into features and target.\ny = df_train[['StartHesitation', 'Turn', 'Walking']]                       # target\nX = df_train.drop(['StartHesitation', 'Turn', 'Walking', 'Time', 'Subject'], axis=1)  # feature","metadata":{"_uuid":"f7959f11-7e58-472b-a10f-8b3490adeb03","_cell_guid":"ec0fe77a-0fc0-4fb5-bc05-7c34344f0d92","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-29T10:16:34.977915Z","iopub.execute_input":"2023-05-29T10:16:34.978309Z","iopub.status.idle":"2023-05-29T10:16:35.005584Z","shell.execute_reply.started":"2023-05-29T10:16:34.978270Z","shell.execute_reply":"2023-05-29T10:16:35.004334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show the counts where each y value is not 0\nprint(len(X), len(y), len(y[y['StartHesitation']>0]), len(y[y['Turn']>0]), len(y[y['Walking']>0]))","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:35.007303Z","iopub.execute_input":"2023-05-29T10:16:35.007881Z","iopub.status.idle":"2023-05-29T10:16:35.025363Z","shell.execute_reply.started":"2023-05-29T10:16:35.007825Z","shell.execute_reply":"2023-05-29T10:16:35.023860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train / test split data \nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.2, random_state = 42, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:35.027154Z","iopub.execute_input":"2023-05-29T10:16:35.027531Z","iopub.status.idle":"2023-05-29T10:16:35.608768Z","shell.execute_reply.started":"2023-05-29T10:16:35.027494Z","shell.execute_reply":"2023-05-29T10:16:35.607410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Perpare test data","metadata":{}},{"cell_type":"code","source":"# tdcs test file\n# list of all tdcsfog csv file path\ntdcs_test_file_path = glob.glob(os.path.join(pdir, 'test', 'tdcsfog', '*.csv'), recursive=True)\nprint(f'the number of files to be read: {len(tdcs_test_file_path)}')","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:35.610472Z","iopub.execute_input":"2023-05-29T10:16:35.611092Z","iopub.status.idle":"2023-05-29T10:16:35.618938Z","shell.execute_reply.started":"2023-05-29T10:16:35.611040Z","shell.execute_reply":"2023-05-29T10:16:35.617365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize a DataFrame to combine data from multiple CSV files.\ndf_tdcs_test = pd.DataFrame()\n\nfor fp in tqdm.tqdm(tdcs_test_file_path):\n    \n    # load data into a variable 'tmp'.\n    tmp = pd.read_csv(fp)\n    \n    # get file Id from csv file name.\n    file_id = os.path.basename(fp).replace(\".csv\", \"\")\n    \n    # get subject Id.\n    subject = df_tdcs_meta.loc[df_tdcs_meta['Id'] == file_id, 'Subject'].iloc[0]\n    \n    # add metadata.\n    tmp['Medication'] = df_tdcs_meta.loc[df_tdcs_meta['Id'] == file_id, 'Medication'].iloc[0]\n    tmp['Subject'] = df_subjects.loc[df_subjects['Subject'] == subject, 'Subject'].iloc[0]\n    tmp['Age'] = df_subjects.loc[df_subjects['Subject'] == subject, 'Age'].iloc[0]\n    tmp['Sex'] = df_subjects.loc[df_subjects['Subject'] == subject, 'Sex'].iloc[0]\n    tmp['YearsSinceDx'] = df_subjects.loc[df_subjects['Subject'] == subject, 'YearsSinceDx'].iloc[0]\n    tmp['NFOGQ'] =df_subjects.loc[df_subjects['Subject'] == subject, 'NFOGQ'].iloc[0]\n    \n    # insert lags\n #   tmp['AccV_lag3'] = tmp.AccV.shift(3)\n    tmp['AccV_lag2'] = tmp.AccV.shift(2)\n    tmp['AccV_lag1'] = tmp.AccV.shift(1)\n#    tmp['AccML_lag3'] = tmp.AccML.shift(3)\n    tmp['AccML_lag2'] = tmp.AccML.shift(2)\n    tmp['AccML_lag1'] = tmp.AccML.shift(1)\n#    tmp['AccAP_lag3'] = tmp.AccAP.shift(3)\n    tmp['AccAP_lag2'] = tmp.AccAP.shift(2)\n    tmp['AccAP_lag1'] = tmp.AccAP.shift(1)\n    \n    # add Id data to submit.\n    tmp['Id'] = file_id + '_' + tmp['Time'].astype(str)\n    \n    # concat the data\n    df_tdcs_test = pd.concat([df_tdcs_test, tmp]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:35.620875Z","iopub.execute_input":"2023-05-29T10:16:35.621395Z","iopub.status.idle":"2023-05-29T10:16:35.674927Z","shell.execute_reply.started":"2023-05-29T10:16:35.621344Z","shell.execute_reply":"2023-05-29T10:16:35.673890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check the contents of the df_tdcs_test\ndf_tdcs_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:35.676220Z","iopub.execute_input":"2023-05-29T10:16:35.677295Z","iopub.status.idle":"2023-05-29T10:16:35.701807Z","shell.execute_reply.started":"2023-05-29T10:16:35.677252Z","shell.execute_reply":"2023-05-29T10:16:35.700393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# defog test file\n# list of all tdcsfog csv file path\ndefog_test_file_path = glob.glob(os.path.join(pdir, 'test', 'defog', '*.csv'), recursive=True)\nprint(f'the number of files to be read: {len(defog_test_file_path)}')","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:35.703333Z","iopub.execute_input":"2023-05-29T10:16:35.703698Z","iopub.status.idle":"2023-05-29T10:16:35.716107Z","shell.execute_reply.started":"2023-05-29T10:16:35.703661Z","shell.execute_reply":"2023-05-29T10:16:35.714334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize a DataFrame to combine data from multiple CSV files.\ndf_defog_test = pd.DataFrame()\nfor fp in tqdm.tqdm(defog_test_file_path):\n    # load data into a variable 'tmp'.\n    tmp = pd.read_csv(fp)\n    \n    # get file Id from csv file name.\n    file_id = os.path.basename(fp).replace(\".csv\", \"\")\n    \n    # get subject Id.\n    subject = df_defog_meta.loc[df_defog_meta['Id'] == file_id, 'Subject'].iloc[0]\n    \n    # add metadata.\n    tmp['Medication'] = df_defog_meta.loc[df_defog_meta['Id'] == file_id, 'Medication'].iloc[0]\n    tmp['Subject'] = df_subjects.loc[df_subjects['Subject'] == subject, 'Subject'].iloc[0]\n    tmp['Age'] = df_subjects.loc[df_subjects['Subject'] == subject, 'Age'].iloc[0]\n    tmp['Sex'] = df_subjects.loc[df_subjects['Subject'] == subject, 'Sex'].iloc[0]\n    tmp['YearsSinceDx'] = df_subjects.loc[df_subjects['Subject'] == subject, 'YearsSinceDx'].iloc[0]\n    tmp['NFOGQ'] =df_subjects.loc[df_subjects['Subject'] == subject, 'NFOGQ'].iloc[0]\n    \n    # insert lags\n #   tmp['AccV_lag3'] = tmp.AccV.shift(3)\n    tmp['AccV_lag2'] = tmp.AccV.shift(2)\n    tmp['AccV_lag1'] = tmp.AccV.shift(1)\n#    tmp['AccML_lag3'] = tmp.AccML.shift(3)\n    tmp['AccML_lag2'] = tmp.AccML.shift(2)\n    tmp['AccML_lag1'] = tmp.AccML.shift(1)\n#    tmp['AccAP_lag3'] = tmp.AccAP.shift(3)\n    tmp['AccAP_lag2'] = tmp.AccAP.shift(2)\n    tmp['AccAP_lag1'] = tmp.AccAP.shift(1)\n    \n    # add Id data to submit.\n    tmp['Id'] = file_id + '_' + tmp['Time'].astype(str)\n    \n    # concat the data\n    df_defog_test = pd.concat([df_defog_test, tmp]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:35.717806Z","iopub.execute_input":"2023-05-29T10:16:35.718264Z","iopub.status.idle":"2023-05-29T10:16:36.505220Z","shell.execute_reply.started":"2023-05-29T10:16:35.718215Z","shell.execute_reply":"2023-05-29T10:16:36.503970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check the contents of the df_defog_test\ndf_defog_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:36.507154Z","iopub.execute_input":"2023-05-29T10:16:36.507539Z","iopub.status.idle":"2023-05-29T10:16:36.535322Z","shell.execute_reply.started":"2023-05-29T10:16:36.507503Z","shell.execute_reply":"2023-05-29T10:16:36.533869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# concat tdcs and defog test data. -> combined test data set\ndf_test = pd.concat([df_tdcs_test, df_defog_test]).reset_index(drop=True)\n\n# encode string columns into 0/1 format\ndf_test['Medication'] = np.where(df_test['Medication']=='on', 1, 0)\ndf_test['Sex'] = np.where(df_test['Sex']=='M', 1, 0)\ndisplay(df_test)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:36.536798Z","iopub.execute_input":"2023-05-29T10:16:36.537196Z","iopub.status.idle":"2023-05-29T10:16:36.812804Z","shell.execute_reply.started":"2023-05-29T10:16:36.537157Z","shell.execute_reply":"2023-05-29T10:16:36.811468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test=df_test.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:36.814516Z","iopub.execute_input":"2023-05-29T10:16:36.815299Z","iopub.status.idle":"2023-05-29T10:16:36.913967Z","shell.execute_reply.started":"2023-05-29T10:16:36.815253Z","shell.execute_reply":"2023-05-29T10:16:36.912553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split data into submission Id and feature.\nId = df_test['Id']                             # Id for submission data\nX_test_set = df_test.drop(['Time', 'Id', 'Subject'], axis=1)  # feature of test data\nX_test_set.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:36.953051Z","iopub.execute_input":"2023-05-29T10:16:36.954029Z","iopub.status.idle":"2023-05-29T10:16:36.999378Z","shell.execute_reply.started":"2023-05-29T10:16:36.953986Z","shell.execute_reply":"2023-05-29T10:16:36.998030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(X_test_set)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:37.001998Z","iopub.execute_input":"2023-05-29T10:16:37.003292Z","iopub.status.idle":"2023-05-29T10:16:37.011032Z","shell.execute_reply.started":"2023-05-29T10:16:37.003245Z","shell.execute_reply":"2023-05-29T10:16:37.009739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Train with different models","metadata":{"_uuid":"3024d062-4df8-4d33-863f-1ad1b8cca545","_cell_guid":"f76cc607-ee02-423c-acf5-ad112ae99c2e","execution":{"iopub.status.busy":"2023-04-20T15:23:33.233731Z","iopub.execute_input":"2023-04-20T15:23:33.23416Z","iopub.status.idle":"2023-04-20T15:23:57.236348Z","shell.execute_reply.started":"2023-04-20T15:23:33.234125Z","shell.execute_reply":"2023-04-20T15:23:57.235273Z"},"trusted":true}},{"cell_type":"code","source":"#features_SH = [\"AccV\", \"AccML\", \"AccAP\", \"Medication\", \"Age\", \"Sex\", \"YearsSinceDx\", \"NFOGQ\",\"AccV_lag2\", \"AccV_lag1\", \"AccML_lag2\", \"AccML_lag1\", \"AccAP_lag2\", \"AccAP_lag1\"]\n#features_Turn= [\"AccV\", \"AccML\", \"AccAP\", \"Medication\", \"Age\", \"Sex\", \"YearsSinceDx\", \"NFOGQ\",\"AccV_lag2\", \"AccV_lag1\", \"AccML_lag2\", \"AccML_lag1\", \"AccAP_lag2\", \"AccAP_lag1\"]\n#features_Walk = [\"AccV\", \"AccML\", \"AccAP\", \"Medication\", \"Age\", \"Sex\", \"YearsSinceDx\", \"NFOGQ\",\"AccV_lag2\", \"AccV_lag1\", \"AccML_lag2\", \"AccML_lag1\", \"AccAP_lag2\", \"AccAP_lag1\"]\n","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:28:27.621054Z","iopub.execute_input":"2023-05-29T10:28:27.623149Z","iopub.status.idle":"2023-05-29T10:28:27.632830Z","shell.execute_reply.started":"2023-05-29T10:28:27.623087Z","shell.execute_reply":"2023-05-29T10:28:27.630819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#pip install xgboost","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:37.042622Z","iopub.execute_input":"2023-05-29T10:16:37.043055Z","iopub.status.idle":"2023-05-29T10:16:37.054683Z","shell.execute_reply.started":"2023-05-29T10:16:37.043012Z","shell.execute_reply":"2023-05-29T10:16:37.053371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train with XGBoost\nfrom xgboost import XGBClassifier\n\nxgbSH = XGBClassifier(base_score=0.5, booster='gblinear',\n                       n_estimators=300,\n                       early_stopping_rounds=200,\n                       eval_metric='auc',\n                       objective='binary:logistic',\n                       max_depth=8,\n                       learning_rate=0.01,\n                       colsample_bytree=1,\n                       subsample=0.5)\n\nxgbSH.fit(X_train, np.array(y_train.StartHesitation), eval_set=[(X_train, np.array(y_train.StartHesitation)), (X_test, np.array(y_test.StartHesitation))], verbose=100)\n","metadata":{"_uuid":"4ea773d9-05f7-4fe2-b76e-84faf68bac9a","_cell_guid":"1dcfc42c-9cb1-4e32-9130-0ace6151ddf7","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-29T10:28:32.371047Z","iopub.execute_input":"2023-05-29T10:28:32.371620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgbTurn = XGBClassifier(base_score=0.5, booster='gblinear',\n                       n_estimators=300,\n                       early_stopping_rounds=200,\n                       eval_metric='auc',\n                       objective='binary:logistic',\n                       max_depth=8,\n                       learning_rate=0.01,\n                       colsample_bytree=1,\n                       subsample=0.5)\n\nxgbTurn.fit(X_train, np.array(y_train.Turn), eval_set=[(X_train, np.array(y_train.Turn)), (X_test, np.array(y_test.Turn))], verbose=100)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:37.277419Z","iopub.status.idle":"2023-05-29T10:16:37.277921Z","shell.execute_reply.started":"2023-05-29T10:16:37.277670Z","shell.execute_reply":"2023-05-29T10:16:37.277695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgbWalk = XGBClassifier(base_score=0.5, booster='gblinear',\n                       n_estimators=300,\n                       early_stopping_rounds=200,\n                       eval_metric='auc',\n                       objective='binary:logistic',\n                       max_depth=8,\n                       learning_rate=0.01,\n                       colsample_bytree=1,\n                       subsample=0.5)\n \nxgbWalk.fit(X_train, y_train.Walking, eval_set=[(X_train, y_train.Walking), (X_test, y_test.Walking)], verbose=100)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:37.280382Z","iopub.status.idle":"2023-05-29T10:16:37.281057Z","shell.execute_reply.started":"2023-05-29T10:16:37.280700Z","shell.execute_reply":"2023-05-29T10:16:37.280735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Predict for test data","metadata":{}},{"cell_type":"code","source":"# calculate prediction using trained model.\n#prediction_hesitation = xgbSH.predict(X_test_set) \nprediction_hesitation = xgbSH.predict_proba(X_test_set) \nxgbSH.feature_importances_","metadata":{"_uuid":"e59f8bff-f981-436b-87f1-5defadedc402","_cell_guid":"76189a00-af78-413f-af4c-2f7fba2e112d","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-29T10:16:37.282843Z","iopub.status.idle":"2023-05-29T10:16:37.283560Z","shell.execute_reply.started":"2023-05-29T10:16:37.283237Z","shell.execute_reply":"2023-05-29T10:16:37.283273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#prediction_turn = xgbTurn.predict(X_test_set)\nprediction_turn = xgbTurn.predict_proba(X_test_set)\nxgbTurn.feature_importances_","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:37.287377Z","iopub.status.idle":"2023-05-29T10:16:37.288003Z","shell.execute_reply.started":"2023-05-29T10:16:37.287666Z","shell.execute_reply":"2023-05-29T10:16:37.287699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#prediction_walking = xgbWalk.predict(X_test_set)\nprediction_walking = xgbWalk.predict_proba(X_test_set)\nxgbWalk.feature_importances_","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.min(prediction_hesitation[:,1]),  np.max(prediction_hesitation[:,1]), np.min(prediction_turn[:,1]),  np.max(prediction_turn[:,1]), np.min(prediction_walking[:,1]),  np.max(prediction_walking[:,1])","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:37.290470Z","iopub.status.idle":"2023-05-29T10:16:37.291150Z","shell.execute_reply.started":"2023-05-29T10:16:37.290787Z","shell.execute_reply":"2023-05-29T10:16:37.290820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Prepare submit data","metadata":{}},{"cell_type":"code","source":"# Prepare submit data\n#submit = pd.DataFrame(Id, columns=['Id'])\n#submit['StartHesitation'] = prediction_hesitation\n#submit['Turn'] = prediction_turn\n#submit['Walking'] = prediction_walking","metadata":{"_uuid":"d238bfd9-9c0b-47cd-b5f2-a45e945eecbd","_cell_guid":"f6740d35-e645-4c15-ac2e-c87e658782b3","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-29T10:16:37.292793Z","iopub.status.idle":"2023-05-29T10:16:37.293440Z","shell.execute_reply.started":"2023-05-29T10:16:37.293187Z","shell.execute_reply":"2023-05-29T10:16:37.293218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare submit data - prob prediction\nsubmit = pd.DataFrame(Id, columns=['Id'])\nsubmit['StartHesitation'] = prediction_hesitation[:,1]\nsubmit['Turn'] = prediction_turn[:,1]\nsubmit['Walking'] = prediction_walking[:,1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(submit)","metadata":{"_uuid":"7e6ea758-7414-4ed8-b433-526e44810134","_cell_guid":"b96670ad-502a-4e1b-b0ca-a23fa578e9cf","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-29T10:16:37.297044Z","iopub.status.idle":"2023-05-29T10:16:37.297555Z","shell.execute_reply.started":"2023-05-29T10:16:37.297314Z","shell.execute_reply":"2023-05-29T10:16:37.297342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the created submission data.\nsubmit.to_csv('submission.csv', index=False)","metadata":{"_uuid":"f259e4b4-9a82-4ab4-8b08-4c4bdc8e0e34","_cell_guid":"7a869a75-d947-4a60-aea0-4f8b8f4cb05c","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-29T10:16:37.298975Z","iopub.status.idle":"2023-05-29T10:16:37.299444Z","shell.execute_reply.started":"2023-05-29T10:16:37.299217Z","shell.execute_reply":"2023-05-29T10:16:37.299241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save and retrieve model for future prediction\n#import joblib\n\n# Save the model to disk.\n#joblib.dump(xgbSH1, 'xgbSH.joblib')\n#joblib.dump(xgbTurn1, 'xgbTurn.joblib')\n#joblib.dump(xgbWalk1, 'xgbWalk.joblib')\n\n# Load the saved models from disk.\n#xgbSH1 = joblib.load('xgbSH1.joblib')\n#xgbTurn1 = joblib.load('xgbTurn1.joblib')\n#xgbWalk1 = joblib.load('xgbWalk1.joblib')\n\n# Use the loaded models to make predictions on test data.\n#prediction_hesitation = xgbSH1.predict(X_test_set)\n#prediction_turn = xgbTurn1.predict(X_test_set)\n#prediction_walking = xgbWalk1.predict(X_test_set)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T10:16:37.301058Z","iopub.status.idle":"2023-05-29T10:16:37.301718Z","shell.execute_reply.started":"2023-05-29T10:16:37.301381Z","shell.execute_reply":"2023-05-29T10:16:37.301417Z"},"trusted":true},"execution_count":null,"outputs":[]}]}