{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#!conda install pytorch --yes","metadata":{"execution":{"iopub.status.busy":"2023-05-26T13:27:56.619654Z","iopub.execute_input":"2023-05-26T13:27:56.620130Z","iopub.status.idle":"2023-05-26T13:27:56.626169Z","shell.execute_reply.started":"2023-05-26T13:27:56.620088Z","shell.execute_reply":"2023-05-26T13:27:56.624736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Thanks to great works from \n01000010_01000010: https://www.kaggle.com/code/bernardbr/lgbm-with-hyperparams-and-tsflex-lb-0-28\n\n&\n\n此般浅薄: https://www.kaggle.com/code/xzj19013742/groupkfold-cross-validation-tsflex\n\n## Contribution:\n- manual feature creation in time, frequency (maybe in f(t) coming from wavelets in the future)\n- denoising using empirical wavelet transform","metadata":{}},{"cell_type":"code","source":"import os\nimport random\nimport itertools\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\n#import common_fog\nfrom tqdm import tqdm\nfrom scipy.signal import butter, filtfilt\nfrom scipy.stats import skew, kurtosis, entropy\nfrom scipy.fft import fft, ifft\nfrom utility_common import process_defog, process_tdcsfog, ewt_filter, butter_lowpass_filter, create_time_domain_features, create_time_domain_window_features,fourier_transform, FOGModel\n\nDEBUG = False #Important to train faster, SET TO FALSE WHEN TRAINING SERIOUSLY\nSTRONG_SUPERVISED = False # True: select only strongly supervised data, False: train on all data\n\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-26T13:27:56.695490Z","iopub.execute_input":"2023-05-26T13:27:56.696052Z","iopub.status.idle":"2023-05-26T13:27:56.705848Z","shell.execute_reply.started":"2023-05-26T13:27:56.695995Z","shell.execute_reply":"2023-05-26T13:27:56.704456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_seed = 666\nnp.random.seed(random_seed)\ntorch.manual_seed(random_seed)\ntorch.cuda.manual_seed(random_seed)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T13:27:56.750582Z","iopub.execute_input":"2023-05-26T13:27:56.750982Z","iopub.status.idle":"2023-05-26T13:27:56.759955Z","shell.execute_reply.started":"2023-05-26T13:27:56.750947Z","shell.execute_reply":"2023-05-26T13:27:56.758487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Pipeline\n### Data pre-processing:\n1. third order Butterworth low-pass filter < 20Hz <br>\n2. Feature creation <br>\n    2.1 Time domain features (Velocity , Center of gravity?, angle?, angular velocity?) <br>\n    2.2 Create Time windows <br>\n        2.2.1 Create Time domain features on time windows\n        2.2.2 FFT\n            2.2.2.1 Create Frequency domain features\n            \n            \nBuilding data processing pipeline:","metadata":{}},{"cell_type":"code","source":"train_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train'\n\nprocessed_files = []\nfiles_defog = os.listdir(os.path.join(train_path, 'defog'))\nfiles_tdcsfog = os.listdir(os.path.join(train_path, 'tdcsfog'))\n\ndefog_sampling_frequency = 100\ntdcsfog_sampling_frequency = 128\n\n#for index, df in tqdm(enumerate(processed_defog)):\ndef extract_defog_features(file, path):\n    # load file\n    df = process_defog([file], path)[0]\n    \n    # lowpass filter \n    df = butter_lowpass_filter(df, 20, defog_sampling_frequency)\n    #df = ewt_filter(df, defog_sampling_frequency, window_period_sec = 4, window_overlap_sec=2)\n    df = create_time_domain_features(df, defog_sampling_frequency)\n    df_dict = create_time_domain_window_features(df, defog_sampling_frequency, window_period_sec = 2, window_overlap_sec=1)\n    df = fourier_transform(df_dict, df, defog_sampling_frequency, window_period_sec = 2, window_overlap_sec=1)\n    #processed_defog[index] = df\n    if STRONG_SUPERVISED:\n        df = df[(df['Task'] == True) & (case['Valid'] == True)]\n    #data = torch.Tensor(df.drop(columns=['StartHesitation', 'Turn', 'Walking', 'Task', 'Valid', 'Id']).to_numpy())\n    #labels = torch.Tensor(df[['StartHesitation', 'Turn', 'Walking']].to_numpy())\n    return df\n    \n#for index, df in tqdm(enumerate(processed_tdcsfog)):\ndef extract_tdcsfog_features(file, path):\n    df = process_tdcsfog([file], path)[0]\n\n    df = butter_lowpass_filter(df, 20, tdcsfog_sampling_frequency)\n    #df = ewt_filter(df, tdcsfog_sampling_frequency, window_period_sec = 4, window_overlap_sec=2)\n    df = create_time_domain_features(df, tdcsfog_sampling_frequency)\n    df_dict = create_time_domain_window_features(df, tdcsfog_sampling_frequency, window_period_sec = 2, window_overlap_sec=1)\n    df = fourier_transform(df_dict, df, tdcsfog_sampling_frequency, window_period_sec = 2, window_overlap_sec=1)\n    if STRONG_SUPERVISED:\n        df = df[(df['Task'] == True) & (case['Valid'] == True)]\n    #data = torch.Tensor(df.drop(columns=['StartHesitation', 'Turn', 'Walking', 'Task', 'Valid', 'Id']).to_numpy())\n    #labels = torch.Tensor(df[['StartHesitation', 'Turn', 'Walking']].to_numpy())\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-26T13:28:03.043504Z","iopub.execute_input":"2023-05-26T13:28:03.043923Z","iopub.status.idle":"2023-05-26T13:28:03.128254Z","shell.execute_reply.started":"2023-05-26T13:28:03.043878Z","shell.execute_reply":"2023-05-26T13:28:03.126909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Extracting data (may take a while)","metadata":{}},{"cell_type":"code","source":"files_defog = files_defog[0:50] if DEBUG else files_defog\nfiles_tdcsfog = files_tdcsfog[0:50] if DEBUG else files_tdcsfog\ntrain = []\n\nfor file in tqdm(files_defog):\n    train.append(extract_defog_features(file, train_path))\n\nfor file in tqdm(files_tdcsfog):\n    train.append(extract_tdcsfog_features(file, train_path))\n    \ntrain = pd.concat(train).reindex()\ntrain = train.fillna(0)\ntrain = train.reset_index(drop=True)\nprint(train.shape)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T13:28:04.043745Z","iopub.execute_input":"2023-05-26T13:28:04.044467Z","iopub.status.idle":"2023-05-26T13:41:34.484012Z","shell.execute_reply.started":"2023-05-26T13:28:04.044426Z","shell.execute_reply":"2023-05-26T13:41:34.482108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training of XGB","metadata":{}},{"cell_type":"code","source":"import xgboost as xgb\nimport lightgbm as lgb\nfrom sklearn.multioutput import MultiOutputRegressor\nfrom sklearn.model_selection import RandomizedSearchCV, train_test_split\nfrom sklearn.metrics import average_precision_score, make_scorer\nfrom sklearn.metrics import mean_squared_error, r2_score\nfrom scipy.stats import uniform, randint","metadata":{"execution":{"iopub.status.busy":"2023-05-26T13:42:44.102908Z","iopub.execute_input":"2023-05-26T13:42:44.103461Z","iopub.status.idle":"2023-05-26T13:42:47.054399Z","shell.execute_reply.started":"2023-05-26T13:42:44.103417Z","shell.execute_reply":"2023-05-26T13:42:47.052988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"best_params_ = {'estimator__colsample_bytree': 0.5282057895135501, \n 'estimator__learning_rate': 0.22659963168004743, \n 'estimator__max_depth': 8, \n 'estimator__min_child_weight': 3.1233911067827616, \n 'estimator__n_estimators': 291, \n 'estimator__subsample': 0.9961057796456088}\nbest_params_ = {kk: v for k, v in best_params_.items() for kk in k.split('__')}; del best_params_['estimator']","metadata":{"execution":{"iopub.status.busy":"2023-05-26T13:42:48.717879Z","iopub.execute_input":"2023-05-26T13:42:48.718402Z","iopub.status.idle":"2023-05-26T13:42:48.726446Z","shell.execute_reply.started":"2023-05-26T13:42:48.718359Z","shell.execute_reply":"2023-05-26T13:42:48.724895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.base import clone\n\ndef custom_average_precision(y_true, y_pred):\n    score = average_precision_score(y_true, y_pred)\n    return 'average_precision', score, True\n\nclass LGBMMultiOutputRegressor(MultiOutputRegressor):\n    def fit(self, X, y, eval_set=None, **fit_params):\n        self.estimators_ = [clone(self.estimator) for _ in range(y.shape[1])]\n        \n        for i, estimator in enumerate(self.estimators_):\n            if eval_set:\n                fit_params['eval_set'] = [(eval_set[0], eval_set[1][:, i])]\n            estimator.fit(X, y[:, i], **fit_params)\n        \n        return self","metadata":{"execution":{"iopub.status.busy":"2023-05-26T13:42:49.414549Z","iopub.execute_input":"2023-05-26T13:42:49.414981Z","iopub.status.idle":"2023-05-26T13:42:49.425060Z","shell.execute_reply.started":"2023-05-26T13:42:49.414944Z","shell.execute_reply":"2023-05-26T13:42:49.423586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# columns to train on: \ncols = [c for c in train.columns if c not in ['Id', 'Time', 'StartHesitation', 'Turn' , 'Walking', 'Valid', 'Task','Event']]\n\n#columns to predict:\npcols = ['StartHesitation', 'Turn' , 'Walking']\n\n#columns for submission\nscols = ['Id', 'StartHesitation', 'Turn' , 'Walking']","metadata":{"execution":{"iopub.status.busy":"2023-05-26T13:42:49.941615Z","iopub.execute_input":"2023-05-26T13:42:49.942079Z","iopub.status.idle":"2023-05-26T13:42:49.950887Z","shell.execute_reply.started":"2023-05-26T13:42:49.942023Z","shell.execute_reply":"2023-05-26T13:42:49.949135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom sklearn.model_selection import GroupKFold\n\nN_FOLDS=5\nepochs = 5\nkfold = GroupKFold(N_FOLDS)\n#group_var = train.Subject\ngroups=kfold.split(train, groups=train.Id)\nregs=[]\ncvs=[]\nfor fold, (tr_idx, te_idx)  in enumerate(tqdm(groups, total=N_FOLDS, desc=\"folds\")):\n    tr_idx=pd.Series(tr_idx).sample(n=2000000,random_state=42).values #2000000\n    \n    # Create a base XGBoost regressor with the common parameters\n    base_regressor = lgb.LGBMRegressor(**best_params_)\n\n    # Wrap the base regressor with the MultiOutputRegressor\n    multioutput_regressor = LGBMMultiOutputRegressor(base_regressor)\n\n    x_tr,y_tr=train.loc[tr_idx,cols].to_numpy(),train.loc[tr_idx,pcols].to_numpy()\n    x_te,y_te=train.loc[te_idx,cols].to_numpy(),train.loc[te_idx,pcols].to_numpy()\n\n    multioutput_regressor.fit(\n    x_tr,y_tr,\n    eval_set=(x_te,y_te),\n    eval_metric=custom_average_precision,\n    early_stopping_rounds=25\n    )\n    regs.append(multioutput_regressor)\n    cv=average_precision_score(y_te, multioutput_regressor.predict(x_te).clip(0.0,1.0))\n    cvs.append(cv)\nprint(cvs)\n","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-05-26T13:42:50.455516Z","iopub.execute_input":"2023-05-26T13:42:50.455995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#multioutput_regressor.save_model(\"xgb_1.txt\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"['Time', 'AccV', 'AccML', 'AccAP', 'Id', 'velocity', 'AccV_mean',\n       'AccV_std', 'AccV_min', 'AccV_max', 'AccV_median', 'AccV_rms',\n       'AccV_skew', 'AccV_kurtosis', 'AccV_entropy', 'AccV_avg_velocity',\n       'AccML_mean', 'AccML_std', 'AccML_min', 'AccML_max', 'AccML_median',\n       'AccML_rms', 'AccML_skew', 'AccML_kurtosis', 'AccML_entropy',\n       'AccML_avg_velocity', 'AccAP_mean', 'AccAP_std', 'AccAP_min',\n       'AccAP_max', 'AccAP_median', 'AccAP_rms', 'AccAP_skew',\n       'AccAP_kurtosis', 'AccAP_entropy', 'AccAP_avg_velocity',\n       'AccV_freq_magn_mean', 'AccV_freq_phase_mean', 'AccV_freeze_index',\n       'AccV_power_index', 'AccML_freq_magn_mean', 'AccML_freq_phase_mean',\n       'AccML_freeze_index', 'AccML_power_index', 'AccAP_freq_magn_mean',\n       'AccAP_freq_phase_mean', 'AccAP_freeze_index', 'AccAP_power_index']","metadata":{}},{"cell_type":"code","source":"test_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test'\n\nprocessed_files = []\nfiles_defog = os.listdir(os.path.join(test_path, 'defog'))\nfiles_tdcsfog = os.listdir(os.path.join(test_path, 'tdcsfog'))\n\nsubmission = []\nfor file in files_defog + files_tdcsfog:\n    if file in files_defog:\n        df = extract_defog_features(file, test_path)\n    elif file in files_tdcsfog:\n        df = extract_tdcsfog_features(file, test_path)\n    df = df.drop(columns=[\"Time\", \"Id\"])\n    df.fillna(method=\"ffill\", inplace=True)\n    \n    res_vals=[]\n    for i_fold in range(N_FOLDS):\n        res_val=np.round(regs[i_fold].predict(df[cols]).clip(0.0,1.0),3)\n        res_vals.append(np.expand_dims(res_val,axis=2))\n    res_vals=np.mean(np.concatenate(res_vals,axis=2),axis=2)\n    res = pd.DataFrame(res_vals, columns=pcols)    \n    df = pd.concat([df,res], axis=1)\n    df['Id'] = file.split('/')[-1].split('.')[0]\n    df['Id'] = df['Id'].astype(str) + '_' + df.index.astype(str)\n    submission.append(df[scols])\n    \n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.concat(submission)\nsubmission[scols].to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}