{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# This notebook uses light-gbm to tackle the Parkinson's Freezing of Gait (FOG) Prediction problem\n\nThanks to JEROENVDD* for pointing out the utility of seglearn and tsflex.\n\ntsflex - \"tsflex is a toolkit for flexible time series processing & feature extraction, that is efficient and makes few assumptions about sequence data.\" (https://github.com/predict-idlab/tsflex)\n\nseglearn - \"provides an integrated pipeline for segmentation, feature extraction, feature processing, and final estimator. Seglearn provides a flexible approach to multivariate time series and related contextual (meta) data for classification, regression, and forecasting problems\" (https://github.com/dmbee/seglearn)\n\n*https://www.kaggle.com/code/jeroenvdd/time-series-tsflex","metadata":{}},{"cell_type":"code","source":"# Install tsflex and seglearn - pull in all the whl files that are required for this notebook\n!pip install tsflex -q --no-index --find-links=file:///kaggle/input/time-series-tools\n!pip install seglearn -q --no-index --find-links=file:///kaggle/input/time-series-tools","metadata":{"execution":{"iopub.status.busy":"2023-04-17T00:53:55.427002Z","iopub.execute_input":"2023-04-17T00:53:55.427719Z","iopub.status.idle":"2023-04-17T00:54:17.330656Z","shell.execute_reply.started":"2023-04-17T00:53:55.427680Z","shell.execute_reply":"2023-04-17T00:54:17.329292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Datasets\n\nNote the two labelled datasets that we have access to, from the competition webpage:\n\"The tDCS FOG (tdcsfog) dataset, comprising data series collected in the lab, as subjects completed a FOG-provoking protocol.\"\n\"The DeFOG (defog) dataset, comprising data series collected in the subject's home, as subjects completed a FOG-provoking protocol\"","metadata":{}},{"cell_type":"code","source":"# import commonly used libraries\nimport os\nimport numpy as np\nimport pandas as pd\nimport pathlib\nfrom tqdm.auto import tqdm\nfrom sklearn import *\nimport glob\nimport matplotlib.pyplot as plt\nimport pickle\n\n# using light-gbm as our model\nimport lightgbm as lgb\nfrom sklearn.multioutput import MultiOutputRegressor\nfrom sklearn.metrics import average_precision_score\nfrom sklearn.base import clone\n\n# import time-series libraries\nimport seglearn\nimport tsflex\nfrom sklearn.model_selection import GroupKFold # add k fold","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2023-04-17T00:54:17.332905Z","iopub.execute_input":"2023-04-17T00:54:17.333281Z","iopub.status.idle":"2023-04-17T00:54:17.341215Z","shell.execute_reply.started":"2023-04-17T00:54:17.333237Z","shell.execute_reply":"2023-04-17T00:54:17.339990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# link to dataset\ndataset_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/'\n\n# create train and test directories\ntrain = glob.glob(dataset_path + 'train/**/**')\ntest = glob.glob(dataset_path + 'test/**/**')\n\n# create additional csvs\nsubjects = pd.read_csv(dataset_path + 'subjects.csv')\ntasks = pd.read_csv(dataset_path + 'tasks.csv')\nSampleSubmission = pd.read_csv(dataset_path + 'sample_submission.csv')\n\n# combine the two datasets mentioned - tdcsfog and defog\ntdcsfog_metadata=pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv')\ndefog_metadata=pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv')\ntdcsfog_metadata['Module']='tdcsfog'\ndefog_metadata['Module']='defog'\nmetadata=pd.concat([tdcsfog_metadata,defog_metadata])","metadata":{"execution":{"iopub.status.busy":"2023-04-17T01:26:13.941143Z","iopub.execute_input":"2023-04-17T01:26:13.941850Z","iopub.status.idle":"2023-04-17T01:26:14.171038Z","shell.execute_reply.started":"2023-04-17T01:26:13.941791Z","shell.execute_reply":"2023-04-17T01:26:14.169735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# feature engineering\n\n# For Tasks: create feature to capture duration of event\ntasks['Duration'] = tasks['End'] - tasks['Begin']\ntasks = pd.pivot_table(tasks, values=['Duration'], index=['Id'], columns=['Task'], aggfunc='sum', fill_value=0)\ntasks.columns = [c[-1] for c in tasks.columns]\ntasks = tasks.reset_index()\n\n# fill data with KMeans\ntasks['t_kmeans'] = cluster.KMeans(n_clusters=10, random_state=3).fit_predict(tasks[tasks.columns[1:]])\n\n# For Subjects: fill data\nsubjects = subjects.fillna(0).groupby('Subject').median()\nsubjects = subjects.reset_index()\n\n# fill data with KMeans\nsubjects['s_kmeans'] = cluster.KMeans(n_clusters=10, random_state=3).fit_predict(subjects[subjects.columns[1:]])\nsubjects=subjects.rename(columns={'Visit':'s_Visit','Age':'s_Age','YearsSinceDx':'s_YearsSinceDx','UPDRSIII_On':'s_UPDRSIII_On','UPDRSIII_Off':'s_UPDRSIII_Off','NFOGQ':'s_NFOGQ'})\n\ncomplex_featlist=['Visit','Test','Medication','s_Visit','s_Age','s_YearsSinceDx','s_UPDRSIII_On','s_UPDRSIII_Off','s_NFOGQ','s_kmeans']\nmetadata_complex=metadata.merge(subjects,how='left',on='Subject').copy()\nmetadata_complex['Medication']=metadata_complex['Medication'].factorize()[0]","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-17T01:26:19.323592Z","iopub.execute_input":"2023-04-17T01:26:19.324821Z","iopub.status.idle":"2023-04-17T01:26:19.475617Z","shell.execute_reply.started":"2023-04-17T01:26:19.324760Z","shell.execute_reply":"2023-04-17T01:26:19.472873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# using tsflex library for time-series feature extraction\n# MultipleFeatureDescriptors class from seglearn for defining multiple feature descriptors\n\n# define a window of 5k\nwindowSize = [5_000]\n\n# create feature descriptors for base features\nbaseFeatures = tsflex.features.MultipleFeatureDescriptors(\n    functions=tsflex.features.integrations.seglearn_feature_dict_wrapper(seglearn.feature_functions.base_features()),\n    series_names=['AccV', 'AccML', 'AccAP'],\n    windows=windowSize,\n    strides=windowSize, # even stride\n)\n\n# create feature descriptors for EMG features\nemgFeatures = emg_features()\ndel emgFeatures['simple square integral'] # delete duplicate category\n\nemgFeatures = tsflex.features.MultipleFeatureDescriptors(\n    functions=tsflex.features.integrations.seglearn_feature_dict_wrapper(emgFeatures),\n    series_names=['AccV', 'AccML', 'AccAP'],\n    windows=windowSize,\n    strides=windowSize, # even stride\n)\n\nfc = tsflex.features.FeatureCollection([baseFeatures, emgFeatures])","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-17T01:26:24.571491Z","iopub.execute_input":"2023-04-17T01:26:24.572331Z","iopub.status.idle":"2023-04-17T01:26:24.583885Z","shell.execute_reply.started":"2023-04-17T01:26:24.572284Z","shell.execute_reply":"2023-04-17T01:26:24.582381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def readFunction(f):\n    try:\n        # import the csv file as a pandas dataframe\n        df = pd.read_csv(f, index_col=\"Time\", usecols=['Time', 'AccV', 'AccML', 'AccAP', 'StartHesitation', 'Turn' , 'Walking'])\n        \n        # edit features\n        df['Id'] = f.split('/')[-1].split('.')[0]\n        df['Module'] = pathlib.Path(f).parts[-2]\n        df['Time_frac'] = (df.index/df.index.max()).values #currently the index of data is actually \"Time\"\n        \n        # merge vecs\n        df = pd.merge(df, tasks[['Id','t_kmeans']], how='left', on='Id').fillna(-1)\n        df = pd.merge(df, metadata_complex[['Id','Subject']+['Visit','Test','Medication','s_kmeans']], how='left', on='Id').fillna(-1)\n        df_feats = fc.calculate(df, return_df=True, include_final_window=True, approve_sparsity=True, window_idx=\"begin\").astype(np.float32)\n        df = df.merge(df_feats, how=\"left\", left_index=True, right_index=True)\n        df.fillna(method=\"ffill\", inplace=True)\n        return df\n    \n    except: pass\n\n# if not os.path.exists(dataset_path + 'data.pkl'):\n#     # run read function\ntrain = pd.concat([readFunction(f) for f in tqdm(train)]).fillna(0)\ntrain = train.reset_index(drop=True)\n# the file is too big to be saved in a pickle file!\n#     # Save dataframe to a pickle file\n#     try:\n#         with open('/kaggle/working/data.pkl', 'wb') as file:\n#             pickle.dump(train, file)\n#         print('saved file to: ' + '/kaggle/working/data.pkl')\n#     else:\n#         print(\"couldn't save, likely read-only\")\n# else:\n#     # Load dataframe from the pickle file\n#     with open('/kaggle/working/data.pkl', 'rb') as file:\n#         train = pickle.load(file)\n#     print('loaded file from: ' + '/kaggle/working/data.pkl')\n\ncols = [c for c in train.columns if c not in ['Id','Subject','Module', 'Time', 'StartHesitation', 'Turn' , 'Walking', 'Valid', 'Task','Event']]\npcols = ['StartHesitation', 'Turn' , 'Walking']\nscols = ['Id', 'StartHesitation', 'Turn' , 'Walking']","metadata":{"execution":{"iopub.status.busy":"2023-04-17T01:26:27.753291Z","iopub.execute_input":"2023-04-17T01:26:27.753916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Training","metadata":{}},{"cell_type":"code","source":"# custom fit function\nclass LGBMMultiOutputRegressor(MultiOutputRegressor):\n    def fit(self, X, y, eval_set, **fit_params):\n        self.estimators_ = [clone(self.estimator) for _ in range(y.shape[1])]\n        \n        for i, estimator in enumerate(self.estimators_):\n            if eval_set:\n                fit_params['eval_set'] = [(eval_set[0], eval_set[1][:, i])]\n            estimator.fit(X, y[:, i], **fit_params)\n        \n        return self\n\n# patching ap scoring\ndef lgbm_ap(y_true, y_pred):\n    score = average_precision_score(y_true, y_pred)\n    return 'average_precision', score, True","metadata":{"execution":{"iopub.status.busy":"2023-04-17T01:02:32.476823Z","iopub.execute_input":"2023-04-17T01:02:32.477273Z","iopub.status.idle":"2023-04-17T01:02:32.488633Z","shell.execute_reply.started":"2023-04-17T01:02:32.477227Z","shell.execute_reply":"2023-04-17T01:02:32.487242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# setting up for 5-fold cross validation\nnumFolds = 5\nkFolds = GroupKFold(numFolds)\ngroupVariable = train.Subject\ngroups = kFolds.split(train, groups = groupVariable)\nregs=[]\ncvs=[]\n\nfor fold, (trainIndex,testIndex) in enumerate(tqdm(groups, total = numFolds)):\n    trainIndex=pd.Series(trainIndex).sample(n = 200,random_state = 1).values\n    \n    # Create a light-gbm regressor\n    base_regressor = lgb.LGBMRegressor(\n    max_depth = 8,\n    learning_rate = 0.2,\n    n_estimators = 300,\n    subsample = 0.99,\n    min_child_weight = 3.12,\n    colsample_bytree = 0.5\n    )\n\n    # Wrap the base regressor with the MultiOutputRegressor\n    multioutput_regressor = LGBMMultiOutputRegressor(base_regressor)\n\n    # set up data\n    # train\n    xTrain = train.loc[trainIndex,cols].to_numpy()\n    yTrain = train.loc[trainIndex,pcols].to_numpy()\n    \n    # test\n    xTest = train.loc[testIndex,cols].to_numpy()\n    yTest = train.loc[testIndex,pcols].to_numpy()\n    \n    # define early stopping callback\n    early_stopping = lgb.callback.early_stopping(\n        stopping_rounds=25, # number of rounds to wait before stopping if no improvement\n        verbose=False, # print message when stopping\n        first_metric_only=True,\n    )\n    \n    # train light-gbm\n    multioutput_regressor.fit(\n    xTrain,\n    yTrain,\n    eval_set = (xTest,yTest),\n    eval_metric=lgbm_ap,\n    early_stopping_rounds=25\n    )","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-04-17T01:02:32.491016Z","iopub.execute_input":"2023-04-17T01:02:32.491398Z","iopub.status.idle":"2023-04-17T01:12:43.444685Z","shell.execute_reply.started":"2023-04-17T01:02:32.491360Z","shell.execute_reply":"2023-04-17T01:12:43.443617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predict for test","metadata":{}},{"cell_type":"code","source":"SampleSubmission['t'] = 0\nSampleSubmission = []\n\nfor f in test:\n    df = pd.read_csv(f)\n    df.set_index('Time', drop=True, inplace=True)\n\n    df['Id'] = f.split('/')[-1].split('.')[0]\n    df['Time_frac']=(df.index/df.index.max()).values#currently the index of data is actually \"Time\"\n    df = pd.merge(df, tasks[['Id','t_kmeans']], how='left', on='Id').fillna(-1)\n    df = pd.merge(df, metadata_complex[['Id','Subject']+['Visit','Test','Medication','s_kmeans']], how='left', on='Id').fillna(-1)\n    df_feats = fc.calculate(df, return_df=True, include_final_window=True, approve_sparsity=True, window_idx=\"begin\")\n    df = df.merge(df_feats, how=\"left\", left_index=True, right_index=True)\n    df.fillna(method=\"ffill\", inplace=True)\n    \n    res_vals=[]\n\n    for i_fold in range(numFolds):\n        res_val=np.round(regs[i_fold].predict(df[cols]).clip(0.0,1.0),3)\n        res_vals.append(np.expand_dims(res_val,axis=2))\n    res_vals=np.mean(np.concatenate(res_vals,axis=2),axis=2)\n    res = pd.DataFrame(res_vals, columns=pcols)\n    \n    df = pd.concat([df,res], axis=1)\n    df['Id'] = df['Id'].astype(str) + '_' + df.index.astype(str)\n    SampleSubmission.append(df[scols])\n    \nSampleSubmission = pd.concat(SampleSubmission)\nSampleSubmission = pd.merge(SampleSubmission[['Id']], SampleSubmission, how='left', on='Id').fillna(0.0)\nSampleSubmission[scols].to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-17T01:21:27.946036Z","iopub.execute_input":"2023-04-17T01:21:27.947285Z","iopub.status.idle":"2023-04-17T01:21:27.974171Z","shell.execute_reply.started":"2023-04-17T01:21:27.947235Z","shell.execute_reply":"2023-04-17T01:21:27.972555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2023-04-17T01:12:45.683733Z","iopub.status.idle":"2023-04-17T01:12:45.684224Z","shell.execute_reply.started":"2023-04-17T01:12:45.683990Z","shell.execute_reply":"2023-04-17T01:12:45.684013Z"},"trusted":true},"execution_count":null,"outputs":[]}]}