{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install tsflex --no-index --find-links=/kaggle/input/time-series-tools\n!pip install seglearn --no-index --find-links=/kaggle/input/time-series-tools","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:06.777387Z","iopub.execute_input":"2023-04-21T01:55:06.778015Z","iopub.status.idle":"2023-04-21T01:55:30.885010Z","shell.execute_reply.started":"2023-04-21T01:55:06.777967Z","shell.execute_reply":"2023-04-21T01:55:30.883325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom tqdm.auto import tqdm\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn import *\nimport glob\n\np = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/'\n\ntrain = glob.glob(p+'train/**/**')\ntest = glob.glob(p+'test/**/**')\nsubjects = pd.read_csv(p+'subjects.csv')\ntasks = pd.read_csv(p+'tasks.csv')\nsub = pd.read_csv(p+'sample_submission.csv')\nevents = pd.read_csv(p+'events.csv')\n\ntdcsfog_metadata=pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv')\ndefog_metadata=pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv')\n# daily_metadata=pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/daily_metadata.csv')\ntdcsfog_metadata['Module']='tdcsfog'\ndefog_metadata['Module']='defog'\n# daily_metadata['Module']='daily'\nmetadata=pd.concat([tdcsfog_metadata,defog_metadata])","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:30.887706Z","iopub.execute_input":"2023-04-21T01:55:30.888144Z","iopub.status.idle":"2023-04-21T01:55:31.155229Z","shell.execute_reply.started":"2023-04-21T01:55:30.888100Z","shell.execute_reply":"2023-04-21T01:55:31.153640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Understanding the Data (Train, Unlabeled, and Metadata)¶","metadata":{}},{"cell_type":"markdown","source":"* The tDCS FOG (tdcsfog) dataset, comprising data series collected in the lab, as subjects completed a FOG-provoking protocol.\n* The DeFOG (defog) dataset, comprising data series collected in the subject's home, as subjects completed a FOG-provoking protocol\n* The Daily Living (daily) dataset, comprising one week of continuous 24/7 recordings from sixty-five subjects. Forty-five subjects exhibit FOG symptoms and also have series in the defog dataset, while the other twenty subjects do not exhibit FOG symptoms and do not have series elsewhere in the data.","metadata":{}},{"cell_type":"markdown","source":"# tDCS FOG (tdcsfog) dataset","metadata":{}},{"cell_type":"code","source":"tdcsfog_example_df = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/003f117e14.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:31.156923Z","iopub.execute_input":"2023-04-21T01:55:31.157326Z","iopub.status.idle":"2023-04-21T01:55:31.173902Z","shell.execute_reply.started":"2023-04-21T01:55:31.157286Z","shell.execute_reply":"2023-04-21T01:55:31.172225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_example_df","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:31.177494Z","iopub.execute_input":"2023-04-21T01:55:31.177880Z","iopub.status.idle":"2023-04-21T01:55:31.198761Z","shell.execute_reply.started":"2023-04-21T01:55:31.177844Z","shell.execute_reply":"2023-04-21T01:55:31.197667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_example_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:31.200032Z","iopub.execute_input":"2023-04-21T01:55:31.200535Z","iopub.status.idle":"2023-04-21T01:55:31.219949Z","shell.execute_reply.started":"2023-04-21T01:55:31.200495Z","shell.execute_reply":"2023-04-21T01:55:31.218071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_example_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:31.221772Z","iopub.execute_input":"2023-04-21T01:55:31.222122Z","iopub.status.idle":"2023-04-21T01:55:31.262064Z","shell.execute_reply.started":"2023-04-21T01:55:31.222086Z","shell.execute_reply":"2023-04-21T01:55:31.260766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* AccV, AccML, and AccAP Acceleration from a lower-back sensor on three axes: V - vertical (Up-Down), ML - mediolateral (Side-Side), AP - anteroposterior (Back-Forward). Data is in units of m/s^2 for","metadata":{}},{"cell_type":"code","source":"for column in ['AccV','AccML', 'AccAP']:\n    plt.plot(tdcsfog_example_df['Time'], tdcsfog_example_df[column], color='red',)\n    plt.title(f\"{column} Time Siries\", fontsize=14)\n    plt.xlabel('Time', fontsize=14)\n    plt.ylabel(f'{column}', fontsize=14)\n    plt.grid(True)\n\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:31.263987Z","iopub.execute_input":"2023-04-21T01:55:31.265028Z","iopub.status.idle":"2023-04-21T01:55:31.963070Z","shell.execute_reply.started":"2023-04-21T01:55:31.264976Z","shell.execute_reply":"2023-04-21T01:55:31.961589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# tdcsfog Meta Data","metadata":{}},{"cell_type":"markdown","source":"Identifies each series in the tdcsfog dataset by a unique Subject, Visit, Test, Medication condition.\n* \n* Visit Lab visits consist of a baseline assessment, two post-treatment assessments for different treatment stages, and one follow-up assessment.\n* Test Which of three test types was performed, with 3 the most challenging.\n* Medication Subjects may have been either off or on anti-parkinsonian medication during the recording","metadata":{}},{"cell_type":"code","source":"tdcsfog_metadata","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:31.964772Z","iopub.execute_input":"2023-04-21T01:55:31.965461Z","iopub.status.idle":"2023-04-21T01:55:31.986296Z","shell.execute_reply.started":"2023-04-21T01:55:31.965424Z","shell.execute_reply":"2023-04-21T01:55:31.984748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_metadata.describe()","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:31.988113Z","iopub.execute_input":"2023-04-21T01:55:31.988505Z","iopub.status.idle":"2023-04-21T01:55:32.013984Z","shell.execute_reply.started":"2023-04-21T01:55:31.988466Z","shell.execute_reply":"2023-04-21T01:55:32.012616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_metadata.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:32.018314Z","iopub.execute_input":"2023-04-21T01:55:32.018726Z","iopub.status.idle":"2023-04-21T01:55:32.033705Z","shell.execute_reply.started":"2023-04-21T01:55:32.018688Z","shell.execute_reply":"2023-04-21T01:55:32.032009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_metadata.Subject.unique()","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:32.035369Z","iopub.execute_input":"2023-04-21T01:55:32.035710Z","iopub.status.idle":"2023-04-21T01:55:32.043893Z","shell.execute_reply.started":"2023-04-21T01:55:32.035676Z","shell.execute_reply":"2023-04-21T01:55:32.042658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data = tdcsfog_metadata, x = \"Test\")","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:32.045286Z","iopub.execute_input":"2023-04-21T01:55:32.045746Z","iopub.status.idle":"2023-04-21T01:55:32.262256Z","shell.execute_reply.started":"2023-04-21T01:55:32.045712Z","shell.execute_reply":"2023-04-21T01:55:32.261021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Events","metadata":{}},{"cell_type":"markdown","source":"Metadata for each FoG event in all data series. The event times agree with the labels in the data series.\n\n* Id The data series the event occured in.\n* Init Time (s) the event began.\n* Completion Time (s) the event ended.\n* Type Whether StartHesitation, Turn, or Walking.\n* Kinetic Whether the event was kinetic (1) and involved movement, or akinetic (0) and static.","metadata":{}},{"cell_type":"code","source":"events","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:32.263600Z","iopub.execute_input":"2023-04-21T01:55:32.263927Z","iopub.status.idle":"2023-04-21T01:55:32.287953Z","shell.execute_reply.started":"2023-04-21T01:55:32.263886Z","shell.execute_reply":"2023-04-21T01:55:32.286289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Subject\n","metadata":{}},{"cell_type":"markdown","source":"Metadata for each Subject in the study, including their Age and Sex as well as:\n\n* Visit Only available for subjects in the daily and defog datasets.\n* YearsSinceDx Years since Parkinson's diagnosis.\n* UPDRSIIIOn/UPDRSIIIOff Unified Parkinson's Disease Rating Scale score during on/off medication respectively.\n* NFOGQ Self-report FoG questionnaire score. See: https://pubmed.ncbi.nlm.nih.gov/19660949/","metadata":{}},{"cell_type":"code","source":"subjects.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:32.289494Z","iopub.execute_input":"2023-04-21T01:55:32.290227Z","iopub.status.idle":"2023-04-21T01:55:32.305401Z","shell.execute_reply.started":"2023-04-21T01:55:32.290180Z","shell.execute_reply":"2023-04-21T01:55:32.304067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(subjects['Age'], bins = 10)\nplt.xlabel('Ages')\nplt.ylabel('Count')","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:32.307284Z","iopub.execute_input":"2023-04-21T01:55:32.307754Z","iopub.status.idle":"2023-04-21T01:55:32.569057Z","shell.execute_reply.started":"2023-04-21T01:55:32.307714Z","shell.execute_reply":"2023-04-21T01:55:32.567976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data = subjects, x = \"Sex\")","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:32.570943Z","iopub.execute_input":"2023-04-21T01:55:32.571485Z","iopub.status.idle":"2023-04-21T01:55:32.710006Z","shell.execute_reply.started":"2023-04-21T01:55:32.571431Z","shell.execute_reply":"2023-04-21T01:55:32.708567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(subjects['YearsSinceDx'], bins = 20)\nplt.xlabel('YearsSinceDx')\nplt.ylabel('Count')","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:32.713179Z","iopub.execute_input":"2023-04-21T01:55:32.714204Z","iopub.status.idle":"2023-04-21T01:55:32.933818Z","shell.execute_reply.started":"2023-04-21T01:55:32.714147Z","shell.execute_reply":"2023-04-21T01:55:32.932425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tasks","metadata":{}},{"cell_type":"markdown","source":"Task metadata for series in the defog dataset. (Not relevant for the series in the fog or daily datasets.)\n\n* Id The data series where the task was measured.\n* Begin Time (s) the task began.\n* End Time (s) the task ended.\n* Task One of seven tasks types in the DeFOG protocol, described on this page.","metadata":{}},{"cell_type":"code","source":"tasks","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:32.935646Z","iopub.execute_input":"2023-04-21T01:55:32.936514Z","iopub.status.idle":"2023-04-21T01:55:32.956219Z","shell.execute_reply.started":"2023-04-21T01:55:32.936451Z","shell.execute_reply":"2023-04-21T01:55:32.954671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tasks.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:32.958311Z","iopub.execute_input":"2023-04-21T01:55:32.958839Z","iopub.status.idle":"2023-04-21T01:55:32.977012Z","shell.execute_reply.started":"2023-04-21T01:55:32.958786Z","shell.execute_reply":"2023-04-21T01:55:32.975720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preprossesing ","metadata":{}},{"cell_type":"code","source":"cols_to_mean = ['Visit', 'UPDRSIII_Off', 'UPDRSIII_On']\nmean_vals = subjects[cols_to_mean].mean()\n\nfor col in cols_to_mean:\n    subjects[col].fillna(mean_vals[col], inplace=True)\n    \nsubjects.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:32.978583Z","iopub.execute_input":"2023-04-21T01:55:32.979050Z","iopub.status.idle":"2023-04-21T01:55:33.008565Z","shell.execute_reply.started":"2023-04-21T01:55:32.978994Z","shell.execute_reply":"2023-04-21T01:55:33.006886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects['Sex'] = np.where(subjects['Sex'] != 'M', 0, 1)","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:33.010620Z","iopub.execute_input":"2023-04-21T01:55:33.011854Z","iopub.status.idle":"2023-04-21T01:55:33.024536Z","shell.execute_reply.started":"2023-04-21T01:55:33.011811Z","shell.execute_reply":"2023-04-21T01:55:33.022869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bins = [0,10,20,30,40,50,60,70,80,90,100]\nlabels = [0,1,2,3,4,5,6,7,8,9]\nsubjects['Age_Category'] = pd.cut(subjects['Age'],bins,labels=labels)","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:33.026010Z","iopub.execute_input":"2023-04-21T01:55:33.026483Z","iopub.status.idle":"2023-04-21T01:55:33.039092Z","shell.execute_reply.started":"2023-04-21T01:55:33.026432Z","shell.execute_reply":"2023-04-21T01:55:33.037764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bins = [0,5,10,15,20,25,30,35]\nlabels = [0,1,2,3,4,5,6]\nsubjects['YearsSinceDx_Category'] = pd.cut(subjects['YearsSinceDx'],bins,labels=labels)","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:33.040870Z","iopub.execute_input":"2023-04-21T01:55:33.041351Z","iopub.status.idle":"2023-04-21T01:55:33.054177Z","shell.execute_reply.started":"2023-04-21T01:55:33.041279Z","shell.execute_reply":"2023-04-21T01:55:33.052994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects.drop(['Age', 'YearsSinceDx'], axis=1, inplace=True)\nsubjects = subjects.reset_index()","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:33.055707Z","iopub.execute_input":"2023-04-21T01:55:33.056231Z","iopub.status.idle":"2023-04-21T01:55:33.070814Z","shell.execute_reply.started":"2023-04-21T01:55:33.056192Z","shell.execute_reply":"2023-04-21T01:55:33.069772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:33.072225Z","iopub.execute_input":"2023-04-21T01:55:33.073546Z","iopub.status.idle":"2023-04-21T01:55:33.104000Z","shell.execute_reply.started":"2023-04-21T01:55:33.073505Z","shell.execute_reply":"2023-04-21T01:55:33.102800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tasks['Duration'] = tasks['End'] - tasks['Begin']\ntasks = pd.pivot_table(tasks, values=['Duration'], index=['Id'], columns=['Task'], aggfunc='sum', fill_value=0)\ntasks.columns = [c[-1] for c in tasks.columns]\ntasks = tasks.reset_index()\ntasks['t_kmeans'] = cluster.KMeans(n_clusters=10, random_state=3).fit_predict(tasks[tasks.columns[1:]])","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:33.105421Z","iopub.execute_input":"2023-04-21T01:55:33.106138Z","iopub.status.idle":"2023-04-21T01:55:33.192777Z","shell.execute_reply.started":"2023-04-21T01:55:33.106100Z","shell.execute_reply":"2023-04-21T01:55:33.191685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects = subjects.fillna(0).groupby('Subject').median()\nsubjects['s_kmeans'] = cluster.KMeans(n_clusters=10, random_state=3).fit_predict(subjects[subjects.columns[1:]])\nsubjects=subjects.rename(columns={'Sex': 's_Sex','Visit':'s_Visit','Age_Category':'s_Age_Category','YearsSinceDx_Category':'s_YearsSinceDx_Category','UPDRSIII_On':'s_UPDRSIII_On','UPDRSIII_Off':'s_UPDRSIII_Off','NFOGQ':'s_NFOGQ'})","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:33.194267Z","iopub.execute_input":"2023-04-21T01:55:33.195505Z","iopub.status.idle":"2023-04-21T01:55:33.259204Z","shell.execute_reply.started":"2023-04-21T01:55:33.195458Z","shell.execute_reply":"2023-04-21T01:55:33.258015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_complex=metadata.merge(subjects,how='left',on='Subject').copy()\nmetadata_complex['Medication']=metadata_complex['Medication'].factorize()[0]","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:33.266903Z","iopub.execute_input":"2023-04-21T01:55:33.267325Z","iopub.status.idle":"2023-04-21T01:55:33.280081Z","shell.execute_reply.started":"2023-04-21T01:55:33.267287Z","shell.execute_reply":"2023-04-21T01:55:33.279023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(metadata_complex.dtypes)","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:33.281572Z","iopub.execute_input":"2023-04-21T01:55:33.282972Z","iopub.status.idle":"2023-04-21T01:55:33.290207Z","shell.execute_reply.started":"2023-04-21T01:55:33.282920Z","shell.execute_reply":"2023-04-21T01:55:33.288511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Previous Notebook score of .198\n\n[Notebook Link](https://www.kaggle.com/code/ashtonfrias/gait-prediction-part-2)\n\nImplementing tsflex feature collection **","metadata":{}},{"cell_type":"markdown","source":"# Create a tsflex feature collection","metadata":{}},{"cell_type":"code","source":"from seglearn.feature_functions import base_features, emg_features\n\nfrom tsflex.features import FeatureCollection, MultipleFeatureDescriptors\nfrom tsflex.features.integrations import seglearn_feature_dict_wrapper\n\n\nbasic_feats = MultipleFeatureDescriptors(\n    functions=seglearn_feature_dict_wrapper(base_features()),\n    series_names=['AccV', 'AccML', 'AccAP'],\n    windows=[5_000],\n    strides=[5_000],\n)\n\nemg_feats = emg_features()\ndel emg_feats['simple square integral'] # is same as abs_energy (which is in base_features)\n\nemg_feats = MultipleFeatureDescriptors(\n    functions=seglearn_feature_dict_wrapper(emg_feats),\n    series_names=['AccV', 'AccML', 'AccAP'],\n    windows=[5_000],\n    strides=[5_000],\n)\n\nfc = FeatureCollection([basic_feats, emg_feats])","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:33.291581Z","iopub.execute_input":"2023-04-21T01:55:33.292683Z","iopub.status.idle":"2023-04-21T01:55:33.304432Z","shell.execute_reply.started":"2023-04-21T01:55:33.292644Z","shell.execute_reply":"2023-04-21T01:55:33.303356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_time(df, freq):\n    if freq == 100:\n        df['Time'] /= 100\n        df['Time'] = round(df['Time'], 3)\n    elif freq == 128:\n        df['Time'] = round(df['Time'] / 128, )\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:33.305957Z","iopub.execute_input":"2023-04-21T01:55:33.306611Z","iopub.status.idle":"2023-04-21T01:55:33.320608Z","shell.execute_reply.started":"2023-04-21T01:55:33.306572Z","shell.execute_reply":"2023-04-21T01:55:33.319635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_acc(df):\n    df['AccV'] *= 9.80665\n    df['AccML'] *= 9.80665\n    df['AccAP'] *= 9.80665\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:33.322114Z","iopub.execute_input":"2023-04-21T01:55:33.322743Z","iopub.status.idle":"2023-04-21T01:55:33.333687Z","shell.execute_reply.started":"2023-04-21T01:55:33.322706Z","shell.execute_reply":"2023-04-21T01:55:33.332174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Extract the features (with Time_frac feature)","metadata":{}},{"cell_type":"code","source":"import pathlib\ndef reader(f):\n    try:\n        df = pd.read_csv(f, index_col=\"Time\", usecols=['Time', 'AccV', 'AccML', 'AccAP', 'StartHesitation', 'Turn' , 'Walking'])\n        \n        df['Id'] = f.split('/')[-1].split('.')[0]\n        df['Module'] = pathlib.Path(f).parts[-2]\n        \n        df['Time_frac']=(df.index/df.index.max()).values#currently the index of data is actually \"Time\"\n        \n        df = pd.merge(df, tasks[['Id','t_kmeans']], how='left', on='Id').fillna(-1)\n        df = pd.merge(df, metadata_complex[['Id','Subject']+['Visit','Test','Medication','s_Sex','s_kmeans']], how='left', on='Id').fillna(-1)\n        df_feats = fc.calculate(df, return_df=True, include_final_window=True, approve_sparsity=True, window_idx=\"begin\").astype(np.float32)\n        df = df.merge(df_feats, how=\"left\", left_index=True, right_index=True)\n        df.fillna(method=\"ffill\", inplace=True)\n        return df\n    except: pass\ntrain = pd.concat([reader(f) for f in tqdm(train)]).fillna(0); print(train.shape)\ncols = [c for c in train.columns if c not in ['Id','Subject','Module', 'Time', 'StartHesitation', 'Turn' , 'Walking', 'Valid', 'Task','Event']]\npcols = ['StartHesitation', 'Turn' , 'Walking']\nscols = ['Id', 'StartHesitation', 'Turn' , 'Walking']\n","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:55:33.335085Z","iopub.execute_input":"2023-04-21T01:55:33.336253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=train.reset_index(drop=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tuning Hyperparams","metadata":{}},{"cell_type":"markdown","source":"**Tuning done on personal computer**","metadata":{}},{"cell_type":"code","source":"# from sklearn.model_selection import RandomizedSearchCV\n# from pprint import pprint\n\n# n_estimators = [int(x) for x in np.linspace(start = 50, stop = 150, num = 5)]\n# max_depth = [int(x) for x in np.linspace(5, 15, num = 3)]\n# max_depth.append(None)\n# min_samples_split = [2, 5, 10]\n# min_samples_leaf = [1, 2, 4]\n# max_features = ['sqrt', 'log2']\n\n# random_grid = {'n_estimators': n_estimators,\n#                 'max_depth': max_depth,\n#                 'min_samples_split': min_samples_split,\n#                 'min_samples_leaf': min_samples_leaf,\n#                 'max_features': max_features\n#               }\n# pprint(random_grid)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.model_selection import GridSearchCV\n\n# rf = ensemble.RandomForestRegressor(random_state=42)\n\n# grid_search = GridSearchCV(rf, param_grid=random_grid, cv=5, n_jobs=-1)\n\n# grid_search.fit(X, Y)\n\n# print(\"Best hyperparameters: \", grid_search.best_params_)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Best hyperparameters: {'max_depth': 10, 'max_features': 'sqrt', 'min_samples_leaf': 1, 'min_samples_split': 2, 'n_estimators': 125}","metadata":{}},{"cell_type":"markdown","source":"# Train Random Forest Regressor","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn import metrics\nimport numpy as np\n\nN_FOLDS = 5\nkfold = GroupKFold(n_splits=N_FOLDS)\ngroup_var = train.Subject\ngroups = kfold.split(train, groups=group_var)\nregs = []\ncvs = []\n\nfor fold, (tr_idx,te_idx ) in enumerate(tqdm(groups, total=N_FOLDS, desc=\"Folds\")):\n    tr_idx = pd.Series(tr_idx).sample(n=2000000, random_state=42).values  # 2000000\n    x_tr,y_tr=train.loc[tr_idx,cols].to_numpy(),train.loc[tr_idx,pcols].to_numpy()\n    x_te,y_te=train.loc[te_idx,cols].to_numpy(),train.loc[te_idx,pcols].to_numpy()\n    \n    base_regressor = RandomForestRegressor(max_depth= 10, min_samples_leaf=1, min_samples_split=2, n_estimators=125, max_features='sqrt', random_state=42, n_jobs=-1)\n    \n    base_regressor.fit(x_tr, y_tr)\n    regs.append(base_regressor)\n    \n    #Make predictions and compute performance metrics\n    y_pred = base_regressor.predict(x_te)\n    cv = metrics.mean_squared_error(y_te, y_pred, squared=False)\n    cvs.append(cv)\n    \nprint(cvs)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test and Submit","metadata":{}},{"cell_type":"code","source":"sub['t'] = 0\nsubmission = []\nfor f in test:\n    df = pd.read_csv(f)\n    df.set_index('Time', drop=True, inplace=True)\n\n    df['Id'] = f.split('/')[-1].split('.')[0]\n    df['Time_frac']=(df.index/df.index.max()).values\n    df = pd.merge(df, tasks[['Id','t_kmeans']], how='left', on='Id').fillna(-1)\n    df = pd.merge(df, metadata_complex[['Id','Subject']+['Visit','Test','Medication','s_Sex','s_kmeans']], how='left', on='Id').fillna(-1)\n    df_feats = fc.calculate(df, return_df=True, include_final_window=True, approve_sparsity=True, window_idx=\"begin\")\n    \n    df = df.merge(df_feats, how=\"left\", left_index=True, right_index=True)\n    df.fillna(method=\"ffill\", inplace=True)\n    \n    res_vals=[]\n    for i_fold in range(N_FOLDS):\n        res_val=np.round(regs[i_fold].predict(df[cols].to_numpy()).clip(0.0,1.0),3)\n        res_vals.append(np.expand_dims(res_val,axis=2))\n    res_vals=np.mean(np.concatenate(res_vals,axis=2),axis=2)\n    res = pd.DataFrame(res_vals, columns=pcols)\n    \n    df = pd.concat([df,res], axis=1)\n    df['Id'] = df['Id'].astype(str) + '_' + df.index.astype(str)\n    submission.append(df[scols])\n    \nsubmission = pd.concat(submission)\nsubmission = pd.merge(sub[['Id']], submission, how='left', on='Id').fillna(0.0)\nsubmission[scols].to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}