{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\nInsallling Some Library from from Kaggle","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":26.425736,"end_time":"2023-04-16T22:41:22.382325","exception":false,"start_time":"2023-04-16T22:40:55.956589","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-19T03:21:23.137384Z","iopub.execute_input":"2023-04-19T03:21:23.137885Z","iopub.status.idle":"2023-04-19T03:21:47.985617Z","shell.execute_reply.started":"2023-04-19T03:21:23.137841Z","shell.execute_reply":"2023-04-19T03:21:47.984117Z"}}},{"cell_type":"code","source":"# Install tsflex and seglearn\n!pip install tsflex --no-index --find-links=file:///kaggle/input/time-series-tools\n!pip install seglearn --no-index --find-links=file:///kaggle/input/time-series-tools","metadata":{"execution":{"iopub.status.busy":"2023-06-01T09:53:26.350506Z","iopub.execute_input":"2023-06-01T09:53:26.351124Z","iopub.status.idle":"2023-06-01T09:53:58.263965Z","shell.execute_reply.started":"2023-06-01T09:53:26.351073Z","shell.execute_reply":"2023-06-01T09:53:58.262285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Download and import necessary library to run the project","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn import *\nimport glob\nfrom tqdm.auto import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom os import path\nfrom pathlib import Path\nfrom seglearn.feature_functions import base_features, emg_features\nfrom tsflex.features import FeatureCollection, MultipleFeatureDescriptors\nfrom tsflex.features.integrations import seglearn_feature_dict_wrapper\nfrom sklearn.model_selection import GroupKFold\nimport lightgbm as lgb\nfrom sklearn.multioutput import MultiOutputRegressor\nfrom sklearn.base import clone\nfrom sklearn.metrics import average_precision_score","metadata":{"papermill":{"duration":2.755431,"end_time":"2023-04-16T22:41:25.148066","exception":false,"start_time":"2023-04-16T22:41:22.392635","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-01T09:53:58.267001Z","iopub.execute_input":"2023-06-01T09:53:58.267541Z","iopub.status.idle":"2023-06-01T09:54:00.396281Z","shell.execute_reply.started":"2023-06-01T09:53:58.267486Z","shell.execute_reply":"2023-06-01T09:54:00.394505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Grab important files\nData Pre-procesing","metadata":{"papermill":{"duration":0.009576,"end_time":"2023-04-16T22:41:25.16752","exception":false,"start_time":"2023-04-16T22:41:25.157944","status":"completed"},"tags":[]}},{"cell_type":"code","source":"root = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/'\n\ntrain = glob.glob(path.join(root, 'train/**/**'))\ntest = glob.glob(path.join(root, 'test/**/**'))\n\nsubjects = pd.read_csv(path.join(root, 'subjects.csv'))\ntasks = pd.read_csv(path.join(root, 'tasks.csv'))\nevents = pd.read_csv(path.join(root, 'events.csv'))\n\ntdcsfog_metadata = pd.read_csv(path.join(root, 'tdcsfog_metadata.csv'))\ndefog_metadata = pd.read_csv(path.join(root, 'defog_metadata.csv')) \n\ntdcsfog_metadata['Module'] = 'tdcsfog'\ndefog_metadata['Module'] = 'defog'\n\nfull_metadata = pd.concat([tdcsfog_metadata, defog_metadata])","metadata":{"papermill":{"duration":0.170669,"end_time":"2023-04-16T22:41:25.347896","exception":false,"start_time":"2023-04-16T22:41:25.177227","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-01T09:54:00.398346Z","iopub.execute_input":"2023-06-01T09:54:00.399776Z","iopub.status.idle":"2023-06-01T09:54:00.571552Z","shell.execute_reply.started":"2023-06-01T09:54:00.399724Z","shell.execute_reply":"2023-06-01T09:54:00.570028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects.loc[subjects['Subject'] == 'fe5d84', 'Sex'] = 'F'","metadata":{"papermill":{"duration":0.045398,"end_time":"2023-04-16T22:41:25.403361","exception":false,"start_time":"2023-04-16T22:41:25.357963","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-01T09:54:00.575671Z","iopub.execute_input":"2023-06-01T09:54:00.576063Z","iopub.status.idle":"2023-06-01T09:54:00.608901Z","shell.execute_reply.started":"2023-06-01T09:54:00.576028Z","shell.execute_reply":"2023-06-01T09:54:00.606974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seed = 3\ncluster_size = 8","metadata":{"papermill":{"duration":0.033146,"end_time":"2023-04-16T22:41:25.44677","exception":false,"start_time":"2023-04-16T22:41:25.413624","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-01T09:54:00.610840Z","iopub.execute_input":"2023-06-01T09:54:00.611201Z","iopub.status.idle":"2023-06-01T09:54:00.625931Z","shell.execute_reply.started":"2023-06-01T09:54:00.611168Z","shell.execute_reply":"2023-06-01T09:54:00.624096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects['Sex'] = subjects['Sex'].factorize()[0]\nsubjects = subjects.fillna(0).groupby('Subject').median()\nsubjects['s_group'] = cluster.KMeans(n_clusters = cluster_size, random_state = seed).fit_predict(subjects[subjects.columns[1:]])\nnew_names = {'Visit':'s_visit','Age':'s_age','YearsSinceDx':'s_years','UPDRSIII_On':'s_on','UPDRSIII_Off':'s_off','NFOGQ':'s_NFOGQ', 'Sex': 's_sex'}\nsubjects = subjects.rename(columns = new_names)\nsubjects","metadata":{"papermill":{"duration":0.035344,"end_time":"2023-04-16T22:41:25.492325","exception":false,"start_time":"2023-04-16T22:41:25.456981","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-01T09:54:00.628061Z","iopub.execute_input":"2023-06-01T09:54:00.628573Z","iopub.status.idle":"2023-06-01T09:54:00.648289Z","shell.execute_reply.started":"2023-06-01T09:54:00.628528Z","shell.execute_reply":"2023-06-01T09:54:00.646718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tasks['Duration'] = tasks['End'] - tasks['Begin']\ntasks = pd.pivot_table(tasks, values=['Duration'], index=['Id'], columns=['Task'], aggfunc='sum', fill_value=0)\ntasks.columns = [c[1] for c in tasks.columns]\ntasks = tasks.reset_index()\ntasks['t_group'] = cluster.KMeans(n_clusters = cluster_size, random_state = seed).fit_predict(tasks[tasks.columns[1:]])","metadata":{"papermill":{"duration":0.039573,"end_time":"2023-04-16T22:41:25.542547","exception":false,"start_time":"2023-04-16T22:41:25.502974","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-01T09:54:00.650048Z","iopub.execute_input":"2023-06-01T09:54:00.650495Z","iopub.status.idle":"2023-06-01T09:54:00.678803Z","shell.execute_reply.started":"2023-06-01T09:54:00.650452Z","shell.execute_reply":"2023-06-01T09:54:00.675931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# merge the subjects with the metadata\nmetadata_w_subjects = full_metadata.merge(subjects, how='left', on='Subject').copy()\nfeatures = metadata_w_subjects.columns","metadata":{"papermill":{"duration":0.02392,"end_time":"2023-04-16T22:41:25.577077","exception":false,"start_time":"2023-04-16T22:41:25.553157","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-01T09:54:00.680892Z","iopub.execute_input":"2023-06-01T09:54:00.681383Z","iopub.status.idle":"2023-06-01T09:54:00.690118Z","shell.execute_reply.started":"2023-06-01T09:54:00.681340Z","shell.execute_reply":"2023-06-01T09:54:00.688890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_w_subjects['Medication'] = metadata_w_subjects['Medication'].factorize()[0]","metadata":{"papermill":{"duration":0.110973,"end_time":"2023-04-16T22:41:25.698532","exception":false,"start_time":"2023-04-16T22:41:25.587559","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-01T09:54:00.691790Z","iopub.execute_input":"2023-06-01T09:54:00.692328Z","iopub.status.idle":"2023-06-01T09:54:00.796356Z","shell.execute_reply.started":"2023-06-01T09:54:00.692288Z","shell.execute_reply":"2023-06-01T09:54:00.794965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Extract from seglearn.feature_functions import base_features, emg_features\nfrom tsflex.features import FeatureCollection, MultipleFeatureDescriptors from tsflex.features.integrations import seglearn_feature_dict_wrapper from the time series data itself","metadata":{}},{"cell_type":"code","source":"basic_feats = MultipleFeatureDescriptors(\n    functions=seglearn_feature_dict_wrapper(base_features()),\n    series_names=['AccV', 'AccML', 'AccAP'],\n    windows=[5000],\n    strides=[5000],\n)\n\nemg_feats = emg_features()\ndel emg_feats['simple square integral'] # is same as abs_energy (which is in base_features)\n\nemg_feats = MultipleFeatureDescriptors(\n    functions=seglearn_feature_dict_wrapper(emg_feats),\n    series_names=['AccV', 'AccML', 'AccAP'],\n    windows=[5000],\n    strides=[5000],\n)\n\nfc = FeatureCollection([basic_feats, emg_feats])","metadata":{"papermill":{"duration":0.030331,"end_time":"2023-04-16T22:41:25.739763","exception":false,"start_time":"2023-04-16T22:41:25.709432","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-01T09:54:00.802206Z","iopub.execute_input":"2023-06-01T09:54:00.802698Z","iopub.status.idle":"2023-06-01T09:54:00.819498Z","shell.execute_reply.started":"2023-06-01T09:54:00.802652Z","shell.execute_reply":"2023-06-01T09:54:00.817796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reader(file):\n    try:\n        df = pd.read_csv(file, index_col='Time', usecols=['Time', 'AccV', 'AccML', 'AccAP', 'StartHesitation', 'Turn' , 'Walking'])\n\n        path_split = file.split('/')\n        df['Id'] = path_split[-1].split('.')[0]\n        dataset = Path(file).parts[-2]\n        df['Module'] = dataset\n        \n        # this is done because the speeds are at different rates for the datasets\n#         if dataset == 'tdcsfog':\n#             df.AccV = df.AccV / 9.80665\n#             df.AccML = df.AccML / 9.80665\n#             df.AccAP = df.AccAP / 9.80665\n\n        df['Time_frac']=(df.index/df.index.max()).values\n        \n        df = pd.merge(df, tasks[['Id','t_group']], how='left', on='Id').fillna(-1)\n        \n        df = pd.merge(df, metadata_w_subjects[['Id','Subject', 'Visit','Test','Medication','s_group']], how='left', on='Id').fillna(-1)\n        \n        df_feats = fc.calculate(df, return_df=True, include_final_window=True, approve_sparsity=True, window_idx=\"begin\").astype(np.float32)\n        df = df.merge(df_feats, how=\"left\", left_index=True, right_index=True)\n        \n#         # stride\n#         df[\"Stride\"] = df[\"AccV\"] + df[\"AccML\"] + df[\"AccAP\"]\n\n#         # step\n#         df[\"Step\"] = np.sqrt(abs(df[\"Stride\"]))\n    \n        df.fillna(method=\"ffill\", inplace=True)\n        \n        return df\n    except: pass\n\ntrain = pd.concat([reader(f) for f in tqdm(train)]).fillna(0); print(train.shape)\ncols = [c for c in train.columns if c not in ['Id','Subject','Module', 'Time', 'StartHesitation', 'Turn' , 'Walking', 'Valid', 'Task','Event']]\npcols = ['StartHesitation', 'Turn' , 'Walking']\nscols = ['Id', 'StartHesitation', 'Turn' , 'Walking']\ntrain=train.reset_index(drop=True)","metadata":{"papermill":{"duration":0.032383,"end_time":"2023-04-16T22:41:25.783325","exception":false,"start_time":"2023-04-16T22:41:25.750942","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-01T09:54:00.821298Z","iopub.execute_input":"2023-06-01T09:54:00.821856Z","iopub.status.idle":"2023-06-01T09:54:00.843743Z","shell.execute_reply.started":"2023-06-01T09:54:00.821807Z","shell.execute_reply":"2023-06-01T09:54:00.842086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"papermill":{"duration":0.108699,"end_time":"2023-04-16T22:41:25.903139","exception":false,"start_time":"2023-04-16T22:41:25.79444","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-01T09:54:00.846001Z","iopub.execute_input":"2023-06-01T09:54:00.846559Z","iopub.status.idle":"2023-06-01T09:54:00.941195Z","shell.execute_reply.started":"2023-06-01T09:54:00.846515Z","shell.execute_reply":"2023-06-01T09:54:00.939513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_params_ = {'colsample_bytree': 0.5282057895135501,\n 'learning_rate': 0.22659963168004743,\n 'max_depth': 8,\n 'min_child_weight': 3.1233911067827616,\n 'n_estimators': 291,\n 'subsample': 0.9961057796456088,\n }\n\ndef custom_average_precision(y_true, y_pred):\n    score = average_precision_score(y_true, y_pred)\n    return 'average_precision', score, True\n\nclass LGBMMultiOutputRegressor(MultiOutputRegressor):\n    def fit(self, X, y, eval_set=None, **fit_params):\n        self.estimators_ = [clone(self.estimator) for _ in range(y.shape[1])]\n        \n        for i, estimator in enumerate(self.estimators_):\n            if eval_set:\n                fit_params['eval_set'] = [(eval_set[0], eval_set[1][:, i])]\n            estimator.fit(X, y[:, i], **fit_params)\n        \n        return self","metadata":{"papermill":{"duration":0.060299,"end_time":"2023-04-16T22:41:25.976025","exception":false,"start_time":"2023-04-16T22:41:25.915726","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-01T09:54:00.943355Z","iopub.execute_input":"2023-06-01T09:54:00.944277Z","iopub.status.idle":"2023-06-01T09:54:00.996264Z","shell.execute_reply.started":"2023-06-01T09:54:00.944221Z","shell.execute_reply":"2023-06-01T09:54:00.994634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kfold = GroupKFold(5)\ngroups=kfold.split(train, groups=train.Subject)\n\nregs = []\ncvs = []\n\nfor _, (tr_idx, te_idx) in enumerate(tqdm(groups, total=5, desc=\"Folds\")):\n    \n    tr_idx = pd.Series(tr_idx).sample(n=2000000,random_state=42).values\n\n    multioutput_regressor = LGBMMultiOutputRegressor(lgb.LGBMRegressor(**best_params_))\n\n    x_train = train.loc[tr_idx, cols].to_numpy()\n    y_train = train.loc[tr_idx, pcols].to_numpy()\n    \n    x_test = train.loc[te_idx, cols].to_numpy()\n    y_test = train.loc[te_idx, pcols].to_numpy()\n\n    multioutput_regressor.fit(\n        x_train, y_train,\n        eval_set=(x_test, y_test),\n        eval_metric=custom_average_precision,\n        early_stopping_rounds=15,\n        verbose = 0,\n    )\n    \n    regs.append(multioutput_regressor)\n    \n    cv = metrics.average_precision_score(y_test, multioutput_regressor.predict(x_test).clip(0.0,1.0))\n    \n    cvs.append(cv)\n    \nprint(cvs)\nprint(np.mean(cvs))","metadata":{"papermill":{"duration":0.031609,"end_time":"2023-04-16T22:41:26.019237","exception":false,"start_time":"2023-04-16T22:41:25.987628","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-01T09:54:00.998987Z","iopub.execute_input":"2023-06-01T09:54:01.000070Z","iopub.status.idle":"2023-06-01T09:54:01.020764Z","shell.execute_reply.started":"2023-06-01T09:54:01.000011Z","shell.execute_reply":"2023-06-01T09:54:01.019089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(path.join(root, 'sample_submission.csv'))\nsubmission = []\n\nfor f in test:\n    df = pd.read_csv(f)\n    df.set_index('Time', drop=True, inplace=True)\n\n    df['Id'] = f.split('/')[-1].split('.')[0]\n\n    dataset = Path(f).parts[-2]\n        \n#     if dataset == 'tdcsfog':\n#         df.AccV = df.AccV / 9.80665\n#         df.AccML = df.AccML / 9.80665\n#         df.AccAP = df.AccAP / 9.80665\n            \n    df['Time_frac']=(df.index/df.index.max()).values\n    df = pd.merge(df, tasks[['Id','t_group']], how='left', on='Id').fillna(-1)\n\n    df = pd.merge(df, metadata_w_subjects[['Id','Subject', 'Visit','Test','Medication','s_group']], how='left', on='Id').fillna(-1)\n    df_feats = fc.calculate(df, return_df=True, include_final_window=True, approve_sparsity=True, window_idx=\"begin\")\n    df = df.merge(df_feats, how=\"left\", left_index=True, right_index=True)\n    df.fillna(method=\"ffill\", inplace=True)\n\n#     # stride\n#     df[\"Stride\"] = df[\"AccV\"] + df[\"AccML\"] + df[\"AccAP\"]\n\n#     # step\n#     df[\"Step\"] = np.sqrt(abs(df[\"Stride\"]))\n        \n    res_vals = []\n    \n    for i_fold in range(5):\n        \n        pred = regs[i_fold].predict(df[cols]).clip(0.0,1.0)\n        res_vals.append(np.expand_dims(np.round(pred, 3), axis = 2))\n        \n    res_vals = np.mean(np.concatenate(res_vals, axis = 2), axis = 2)\n    res = pd.DataFrame(res_vals, columns=pcols)\n    \n    df = pd.concat([df,res], axis=1)\n    df['Id'] = df['Id'].astype(str) + '_' + df.index.astype(str)\n    submission.append(df[scols])\n    \nsubmission = pd.concat(submission)\nsubmission = pd.merge(sub[['Id']], submission, how='left', on='Id').fillna(0.0)\nsubmission[scols].to_csv('submission.csv', index=False)","metadata":{"papermill":{"duration":0.03679,"end_time":"2023-04-16T22:41:26.068174","exception":false,"start_time":"2023-04-16T22:41:26.031384","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-01T09:54:01.022990Z","iopub.execute_input":"2023-06-01T09:54:01.023548Z","iopub.status.idle":"2023-06-01T09:54:01.043784Z","shell.execute_reply.started":"2023-06-01T09:54:01.023495Z","shell.execute_reply":"2023-06-01T09:54:01.042199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"papermill":{"duration":0.060364,"end_time":"2023-04-16T22:41:26.141472","exception":false,"start_time":"2023-04-16T22:41:26.081108","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-01T09:54:01.045380Z","iopub.execute_input":"2023-06-01T09:54:01.045882Z","iopub.status.idle":"2023-06-01T09:54:01.093204Z","shell.execute_reply.started":"2023-06-01T09:54:01.045840Z","shell.execute_reply":"2023-06-01T09:54:01.091363Z"},"trusted":true},"execution_count":null,"outputs":[]}]}