{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Credit to 此般浅薄 for the initial inspiration","metadata":{}},{"cell_type":"code","source":"# Install tsflex and seglearn\n!pip install tsflex --no-index --find-links=file:///kaggle/input/time-series-tools\n!pip install seglearn --no-index --find-links=file:///kaggle/input/time-series-tools","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":26.425736,"end_time":"2023-04-16T22:41:22.382325","exception":false,"start_time":"2023-04-16T22:40:55.956589","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:30:41.242724Z","iopub.execute_input":"2023-04-21T01:30:41.243137Z","iopub.status.idle":"2023-04-21T01:31:05.875275Z","shell.execute_reply.started":"2023-04-21T01:30:41.243099Z","shell.execute_reply":"2023-04-21T01:31:05.873632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn import *\nimport glob\nfrom tqdm.auto import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom os import path\nfrom pathlib import Path\nfrom seglearn.feature_functions import base_features, emg_features\nfrom tsflex.features import FeatureCollection, MultipleFeatureDescriptors\nfrom tsflex.features.integrations import seglearn_feature_dict_wrapper\nfrom sklearn.model_selection import GroupKFold\nimport lightgbm as lgb\nfrom sklearn.multioutput import MultiOutputRegressor\nfrom sklearn.base import clone\nfrom sklearn.metrics import average_precision_score","metadata":{"papermill":{"duration":2.755431,"end_time":"2023-04-16T22:41:25.148066","exception":false,"start_time":"2023-04-16T22:41:22.392635","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:31:05.877797Z","iopub.execute_input":"2023-04-21T01:31:05.878280Z","iopub.status.idle":"2023-04-21T01:31:08.601308Z","shell.execute_reply.started":"2023-04-21T01:31:05.878239Z","shell.execute_reply":"2023-04-21T01:31:08.600294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Grab important files","metadata":{"papermill":{"duration":0.009576,"end_time":"2023-04-16T22:41:25.167520","exception":false,"start_time":"2023-04-16T22:41:25.157944","status":"completed"},"tags":[]}},{"cell_type":"code","source":"root = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/'\n\ntrain = glob.glob(path.join(root, 'train/**/**'))\ntest = glob.glob(path.join(root, 'test/**/**'))\n\nsubjects = pd.read_csv(path.join(root, 'subjects.csv'))\ntasks = pd.read_csv(path.join(root, 'tasks.csv'))\nevents = pd.read_csv(path.join(root, 'events.csv'))\n\ntdcsfog_metadata = pd.read_csv(path.join(root, 'tdcsfog_metadata.csv'))\ndefog_metadata = pd.read_csv(path.join(root, 'defog_metadata.csv')) \n\ntdcsfog_metadata['Module'] = 'tdcsfog'\ndefog_metadata['Module'] = 'defog'\n\nfull_metadata = pd.concat([tdcsfog_metadata, defog_metadata])","metadata":{"papermill":{"duration":0.170669,"end_time":"2023-04-16T22:41:25.347896","exception":false,"start_time":"2023-04-16T22:41:25.177227","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:31:08.602597Z","iopub.execute_input":"2023-04-21T01:31:08.602929Z","iopub.status.idle":"2023-04-21T01:31:08.749108Z","shell.execute_reply.started":"2023-04-21T01:31:08.602897Z","shell.execute_reply":"2023-04-21T01:31:08.748115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects.loc[subjects['Subject'] == 'fe5d84', 'Sex'] = 'F'","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:31:08.751967Z","iopub.execute_input":"2023-04-21T01:31:08.752315Z","iopub.status.idle":"2023-04-21T01:31:08.758469Z","shell.execute_reply.started":"2023-04-21T01:31:08.752281Z","shell.execute_reply":"2023-04-21T01:31:08.757339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seed = 100\ncluster_size = 8","metadata":{"papermill":{"duration":0.02392,"end_time":"2023-04-16T22:41:25.577077","exception":false,"start_time":"2023-04-16T22:41:25.553157","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:31:08.760314Z","iopub.execute_input":"2023-04-21T01:31:08.761109Z","iopub.status.idle":"2023-04-21T01:31:08.776238Z","shell.execute_reply.started":"2023-04-21T01:31:08.761062Z","shell.execute_reply":"2023-04-21T01:31:08.775046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects['Sex'] = subjects['Sex'].factorize()[0]\nsubjects = subjects.fillna(0).groupby('Subject').median()\nsubjects['s_group'] = cluster.KMeans(n_clusters = cluster_size, random_state = seed).fit_predict(subjects[subjects.columns[1:]])\nnew_names = {'Visit':'s_visit','Age':'s_age','YearsSinceDx':'s_years','UPDRSIII_On':'s_on','UPDRSIII_Off':'s_off','NFOGQ':'s_NFOGQ', 'Sex': 's_sex'}\nsubjects = subjects.rename(columns = new_names)\nsubjects","metadata":{"papermill":{"duration":0.110973,"end_time":"2023-04-16T22:41:25.698532","exception":false,"start_time":"2023-04-16T22:41:25.587559","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:31:08.777921Z","iopub.execute_input":"2023-04-21T01:31:08.779552Z","iopub.status.idle":"2023-04-21T01:31:08.870787Z","shell.execute_reply.started":"2023-04-21T01:31:08.779514Z","shell.execute_reply":"2023-04-21T01:31:08.869562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tasks['Duration'] = tasks['End'] - tasks['Begin']\ntasks = pd.pivot_table(tasks, values=['Duration'], index=['Id'], columns=['Task'], aggfunc='sum', fill_value=0)\ntasks.columns = [c[1] for c in tasks.columns]\ntasks = tasks.reset_index()\ntasks['t_group'] = cluster.KMeans(n_clusters = cluster_size, random_state = seed).fit_predict(tasks[tasks.columns[1:]])","metadata":{"papermill":{"duration":0.108699,"end_time":"2023-04-16T22:41:25.903139","exception":false,"start_time":"2023-04-16T22:41:25.794440","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:31:08.872068Z","iopub.execute_input":"2023-04-21T01:31:08.872422Z","iopub.status.idle":"2023-04-21T01:31:08.948153Z","shell.execute_reply.started":"2023-04-21T01:31:08.872362Z","shell.execute_reply":"2023-04-21T01:31:08.946914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# merge the subjects with the metadata\nmetadata_w_subjects = full_metadata.merge(subjects, how='left', on='Subject').copy()\nfeatures = metadata_w_subjects.columns","metadata":{"papermill":{"duration":0.060364,"end_time":"2023-04-16T22:41:26.141472","exception":false,"start_time":"2023-04-16T22:41:26.081108","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:31:08.949542Z","iopub.execute_input":"2023-04-21T01:31:08.950104Z","iopub.status.idle":"2023-04-21T01:31:08.962007Z","shell.execute_reply.started":"2023-04-21T01:31:08.950064Z","shell.execute_reply":"2023-04-21T01:31:08.960776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_w_subjects['Medication'] = metadata_w_subjects['Medication'].factorize()[0]","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:31:08.963763Z","iopub.execute_input":"2023-04-21T01:31:08.964093Z","iopub.status.idle":"2023-04-21T01:31:08.970239Z","shell.execute_reply.started":"2023-04-21T01:31:08.964060Z","shell.execute_reply":"2023-04-21T01:31:08.968995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Extract from seglearn.feature_functions import base_features, emg_features\n\nfrom tsflex.features import FeatureCollection, MultipleFeatureDescriptors\nfrom tsflex.features.integrations import seglearn_feature_dict_wrapper from the time series data itself","metadata":{"papermill":{"duration":0.021931,"end_time":"2023-04-16T22:41:26.370019","exception":false,"start_time":"2023-04-16T22:41:26.348088","status":"completed"},"tags":[]}},{"cell_type":"code","source":"basic_feats = MultipleFeatureDescriptors(\n    functions=seglearn_feature_dict_wrapper(base_features()),\n    series_names=['AccV', 'AccML', 'AccAP'],\n    windows=[5000],\n    strides=[5000],\n)\n\nemg_feats = emg_features()\ndel emg_feats['simple square integral'] # is same as abs_energy (which is in base_features)\n\nemg_feats = MultipleFeatureDescriptors(\n    functions=seglearn_feature_dict_wrapper(emg_feats),\n    series_names=['AccV', 'AccML', 'AccAP'],\n    windows=[5000],\n    strides=[5000],\n)\n\nfc = FeatureCollection([basic_feats, emg_feats])","metadata":{"papermill":{"duration":0.032203,"end_time":"2023-04-16T22:41:26.517639","exception":false,"start_time":"2023-04-16T22:41:26.485436","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:31:08.974165Z","iopub.execute_input":"2023-04-21T01:31:08.974543Z","iopub.status.idle":"2023-04-21T01:31:08.984300Z","shell.execute_reply.started":"2023-04-21T01:31:08.974497Z","shell.execute_reply":"2023-04-21T01:31:08.983112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reader(file):\n    try:\n        df = pd.read_csv(file, index_col='Time', usecols=['Time', 'AccV', 'AccML', 'AccAP', 'StartHesitation', 'Turn' , 'Walking'])\n\n        path_split = file.split('/')\n        df['Id'] = path_split[-1].split('.')[0]\n        dataset = Path(file).parts[-2]\n        df['Module'] = dataset\n        \n        # this is done because the speeds are at different rates for the datasets\n#         if dataset == 'tdcsfog':\n#             df.AccV = df.AccV / 9.80665\n#             df.AccML = df.AccML / 9.80665\n#             df.AccAP = df.AccAP / 9.80665\n\n        df['Time_frac']=(df.index/df.index.max()).values\n        \n        df = pd.merge(df, tasks[['Id','t_group']], how='left', on='Id').fillna(-1)\n        \n        df = pd.merge(df, metadata_w_subjects[['Id','Subject', 'Visit','Test','Medication','s_group']], how='left', on='Id').fillna(-1)\n        \n        df_feats = fc.calculate(df, return_df=True, include_final_window=True, approve_sparsity=True, window_idx=\"begin\").astype(np.float32)\n        df = df.merge(df_feats, how=\"left\", left_index=True, right_index=True)\n        \n#         # stride\n#         df[\"Stride\"] = df[\"AccV\"] + df[\"AccML\"] + df[\"AccAP\"]\n\n#         # step\n#         df[\"Step\"] = np.sqrt(abs(df[\"Stride\"]))\n    \n        df.fillna(method=\"ffill\", inplace=True)\n        \n        return df\n    except: pass\n\ntrain = pd.concat([reader(f) for f in tqdm(train)]).fillna(0); print(train.shape)\ncols = [c for c in train.columns if c not in ['Id','Subject','Module', 'Time', 'StartHesitation', 'Turn' , 'Walking', 'Valid', 'Task','Event']]\npcols = ['StartHesitation', 'Turn' , 'Walking']\nscols = ['Id', 'StartHesitation', 'Turn' , 'Walking']\ntrain=train.reset_index(drop=True)","metadata":{"papermill":{"duration":533.977331,"end_time":"2023-04-16T22:50:20.513889","exception":false,"start_time":"2023-04-16T22:41:26.536558","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:31:08.985904Z","iopub.execute_input":"2023-04-21T01:31:08.986270Z","iopub.status.idle":"2023-04-21T01:39:16.081832Z","shell.execute_reply.started":"2023-04-21T01:31:08.986236Z","shell.execute_reply":"2023-04-21T01:39:16.079813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"papermill":{"duration":0.048438,"end_time":"2023-04-16T22:50:20.575881","exception":false,"start_time":"2023-04-16T22:50:20.527443","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:39:16.085905Z","iopub.execute_input":"2023-04-21T01:39:16.086577Z","iopub.status.idle":"2023-04-21T01:39:16.122094Z","shell.execute_reply.started":"2023-04-21T01:39:16.086495Z","shell.execute_reply":"2023-04-21T01:39:16.120863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_params_ = {'colsample_bytree': 0.5282057895135501,\n 'learning_rate': 0.22659963168004743,\n 'max_depth': 8,\n 'min_child_weight': 3.1233911067827616,\n 'n_estimators': 291,\n 'subsample': 0.9961057796456088,\n }\n\ndef custom_average_precision(y_true, y_pred):\n    score = average_precision_score(y_true, y_pred)\n    return 'average_precision', score, True\n\nclass LGBMMultiOutputRegressor(MultiOutputRegressor):\n    def fit(self, X, y, eval_set=None, **fit_params):\n        self.estimators_ = [clone(self.estimator) for _ in range(y.shape[1])]\n        \n        for i, estimator in enumerate(self.estimators_):\n            if eval_set:\n                fit_params['eval_set'] = [(eval_set[0], eval_set[1][:, i])]\n            estimator.fit(X, y[:, i], **fit_params)\n        \n        return self","metadata":{"papermill":{"duration":0.031497,"end_time":"2023-04-16T22:50:29.071479","exception":false,"start_time":"2023-04-16T22:50:29.039982","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:39:16.134463Z","iopub.execute_input":"2023-04-21T01:39:16.134812Z","iopub.status.idle":"2023-04-21T01:39:16.145204Z","shell.execute_reply.started":"2023-04-21T01:39:16.134779Z","shell.execute_reply":"2023-04-21T01:39:16.143994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kfold = GroupKFold(5)\ngroups=kfold.split(train, groups=train.Subject)\n\nregs = []\ncvs = []\n\nfor _, (tr_idx, te_idx) in enumerate(tqdm(groups, total=5, desc=\"Folds\")):\n    \n    tr_idx = pd.Series(tr_idx).sample(n=2000000,random_state=42).values\n\n    multioutput_regressor = LGBMMultiOutputRegressor(lgb.LGBMRegressor(**best_params_))\n\n    x_train = train.loc[tr_idx, cols].to_numpy()\n    y_train = train.loc[tr_idx, pcols].to_numpy()\n    \n    x_test = train.loc[te_idx, cols].to_numpy()\n    y_test = train.loc[te_idx, pcols].to_numpy()\n\n    multioutput_regressor.fit(\n        x_train, y_train,\n        eval_set=(x_test, y_test),\n        eval_metric=custom_average_precision,\n        early_stopping_rounds=15,\n        verbose = 0,\n    )\n    \n    regs.append(multioutput_regressor)\n    \n    cv = metrics.average_precision_score(y_test, multioutput_regressor.predict(x_test).clip(0.0,1.0))\n    \n    cvs.append(cv)\n    \nprint(cvs)\nprint(np.mean(cvs))","metadata":{"papermill":{"duration":943.279901,"end_time":"2023-04-16T23:06:12.366865","exception":false,"start_time":"2023-04-16T22:50:29.086964","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:39:16.146814Z","iopub.execute_input":"2023-04-21T01:39:16.147155Z","iopub.status.idle":"2023-04-21T01:51:08.031261Z","shell.execute_reply.started":"2023-04-21T01:39:16.147123Z","shell.execute_reply":"2023-04-21T01:51:08.029721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(path.join(root, 'sample_submission.csv'))\nsubmission = []\n\nfor f in test:\n    df = pd.read_csv(f)\n    df.set_index('Time', drop=True, inplace=True)\n\n    df['Id'] = f.split('/')[-1].split('.')[0]\n\n    dataset = Path(f).parts[-2]\n        \n#     if dataset == 'tdcsfog':\n#         df.AccV = df.AccV / 9.80665\n#         df.AccML = df.AccML / 9.80665\n#         df.AccAP = df.AccAP / 9.80665\n            \n    df['Time_frac']=(df.index/df.index.max()).values\n    df = pd.merge(df, tasks[['Id','t_group']], how='left', on='Id').fillna(-1)\n\n    df = pd.merge(df, metadata_w_subjects[['Id','Subject', 'Visit','Test','Medication','s_group']], how='left', on='Id').fillna(-1)\n    df_feats = fc.calculate(df, return_df=True, include_final_window=True, approve_sparsity=True, window_idx=\"begin\")\n    df = df.merge(df_feats, how=\"left\", left_index=True, right_index=True)\n    df.fillna(method=\"ffill\", inplace=True)\n\n#     # stride\n#     df[\"Stride\"] = df[\"AccV\"] + df[\"AccML\"] + df[\"AccAP\"]\n\n#     # step\n#     df[\"Step\"] = np.sqrt(abs(df[\"Stride\"]))\n        \n    res_vals = []\n    \n    for i_fold in range(5):\n        \n        pred = regs[i_fold].predict(df[cols]).clip(0.0,1.0)\n        res_vals.append(np.expand_dims(np.round(pred, 3), axis = 2))\n        \n    res_vals = np.mean(np.concatenate(res_vals, axis = 2), axis = 2)\n    res = pd.DataFrame(res_vals, columns=pcols)\n    \n    df = pd.concat([df,res], axis=1)\n    df['Id'] = df['Id'].astype(str) + '_' + df.index.astype(str)\n    submission.append(df[scols])\n    \nsubmission = pd.concat(submission)\nsubmission = pd.merge(sub[['Id']], submission, how='left', on='Id').fillna(0.0)\nsubmission[scols].to_csv('submission.csv', index=False)","metadata":{"papermill":{"duration":10.790056,"end_time":"2023-04-16T23:06:23.205940","exception":false,"start_time":"2023-04-16T23:06:12.415884","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:51:08.033808Z","iopub.execute_input":"2023-04-21T01:51:08.034312Z","iopub.status.idle":"2023-04-21T01:51:16.946572Z","shell.execute_reply.started":"2023-04-21T01:51:08.034249Z","shell.execute_reply":"2023-04-21T01:51:16.945259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"papermill":{"duration":0.076052,"end_time":"2023-04-16T23:06:23.331235","exception":false,"start_time":"2023-04-16T23:06:23.255183","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:51:16.948492Z","iopub.execute_input":"2023-04-21T01:51:16.949566Z","iopub.status.idle":"2023-04-21T01:51:16.967779Z","shell.execute_reply.started":"2023-04-21T01:51:16.949502Z","shell.execute_reply":"2023-04-21T01:51:16.966552Z"},"trusted":true},"execution_count":null,"outputs":[]}]}