{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\nGradient boosting models are becoming popular because of their effectiveness at classifying complex datasets. Nowadays there are  most powerful gradient boosting models XGBoost, LightGBM, CatBoots. \n\nThe result of this research is to find the most perspective gradient boosting models for FoG prediction.\n\nP.S. We will train models without any hyperparameters to avoid distortion. And we'll use f1 score to assess the quality of the models. ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport glob\n\nfrom xgboost import XGBClassifier\nfrom catboost import CatBoostClassifier\nfrom lightgbm import LGBMClassifier\n\nfrom sklearn.metrics import f1_score\nfrom sklearn.metrics import accuracy_score\n\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2023-06-10T23:08:12.390434Z","iopub.execute_input":"2023-06-10T23:08:12.390841Z","iopub.status.idle":"2023-06-10T23:08:15.978062Z","shell.execute_reply.started":"2023-06-10T23:08:12.390806Z","shell.execute_reply":"2023-06-10T23:08:15.976286Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Loading the data¶","metadata":{}},{"cell_type":"code","source":"data_dir = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/'\n# \ndaily_metadata_file = f'{data_dir}daily_metadata.csv'\ndefog_metadata_file = f'{data_dir}defog_metadata.csv'\ntdcsfog_metadata_file = f'{data_dir}tdcsfog_metadata.csv'\nevents_data_file = f'{data_dir}events.csv'\nsubjects_data_file = f'{data_dir}subjects.csv'\ntasks_data_file = f'{data_dir}tasks.csv'\n\n# Read the meta data\ndaily_metadata = pd.read_csv(daily_metadata_file)\ndefog_metadata = pd.read_csv(defog_metadata_file)\ntdcsfog_metadata = pd.read_csv(tdcsfog_metadata_file)\nfull_metadata = pd.concat([tdcsfog_metadata, defog_metadata])\n\nevents_data = pd.read_csv(events_data_file)\nsubjects_data = pd.read_csv(subjects_data_file)\ntasks_data = pd.read_csv(tasks_data_file)","metadata":{"execution":{"iopub.status.busy":"2023-06-10T23:08:15.980760Z","iopub.execute_input":"2023-06-10T23:08:15.981269Z","iopub.status.idle":"2023-06-10T23:08:16.062612Z","shell.execute_reply.started":"2023-06-10T23:08:15.981226Z","shell.execute_reply":"2023-06-10T23:08:16.061392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the data and append the Id\ndef read_data(path):\n    df = pd.read_csv(path)\n    df['Id'] = path.split(\"/\")[-1].split(\".\")[0]\n    \n    return df\n\n\n# Read and concatenate all files of the train data from specified dataset\ndef create_full_data(dataset_name):\n    paths = glob.glob(data_dir + f'train/{dataset_name}/*')\n    final_df = pd.concat([read_data(p) for p in paths])\n    final_df['dataset'] = dataset_name\n    \n    return final_df","metadata":{"execution":{"iopub.status.busy":"2023-06-10T23:08:17.562175Z","iopub.execute_input":"2023-06-10T23:08:17.562563Z","iopub.status.idle":"2023-06-10T23:08:17.571005Z","shell.execute_reply.started":"2023-06-10T23:08:17.562532Z","shell.execute_reply":"2023-06-10T23:08:17.569090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's take a defog dataset for our goal.","metadata":{}},{"cell_type":"code","source":"# Create a dataframe with all defog data\ndefog_data = create_full_data('defog')\ndefog_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-10T23:08:20.666959Z","iopub.execute_input":"2023-06-10T23:08:20.667358Z","iopub.status.idle":"2023-06-10T23:08:48.459874Z","shell.execute_reply.started":"2023-06-10T23:08:20.667327Z","shell.execute_reply":"2023-06-10T23:08:48.458810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_data_valid = defog_data.loc[ \\\n                        (defog_data.Valid == True) & (defog_data.Task == True)].copy()\n\ndefog_data_valid.reset_index(drop=True, inplace=True)\ndefog_data_valid.drop(['Valid', 'Task'], axis=1, inplace=True)\n\ndefog_data_valid.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-10T23:08:48.461629Z","iopub.execute_input":"2023-06-10T23:08:48.461943Z","iopub.status.idle":"2023-06-10T23:08:50.385416Z","shell.execute_reply.started":"2023-06-10T23:08:48.461916Z","shell.execute_reply":"2023-06-10T23:08:50.384317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature engineering¶\n\n\n\n\n\n\n\n\n\n\n\n","metadata":{}},{"cell_type":"markdown","source":"Feature engineering process is based on an existing (<a href=\"https://www.kaggle.com/code/averkovanika/mutual-information-feature-selection?scriptVersionId=128628966\">Notebook link</a>)","metadata":{}},{"cell_type":"code","source":"# Create new features based on accelerometer data columns and returns the updated dataframe\ndef create_features(df):\n    acc_cols = ['AccV', 'AccML', 'AccAP']\n\n    for acc in acc_cols:\n        df[f'{acc}_lag_2'] = df.groupby('Id')[acc].shift(2).fillna(method=\"backfill\")\n        df[f'{acc}_lag_3'] = df.groupby('Id')[acc].shift(3).fillna(method=\"backfill\")\n        df[f'{acc}_lag_4'] = df.groupby('Id')[acc].shift(4).fillna(method=\"backfill\")\n        df[f'{acc}_lag_5'] = df.groupby('Id')[acc].shift(5).fillna(method=\"backfill\")\n\n        df[f'{acc}_cumsum'] = (df[acc]).groupby(df['Id']).cumsum()\n\n        df[f'{acc}_first_value'] = df.groupby('Id')[acc].transform('first')\n        df[f'{acc}_last_value'] = df.groupby('Id')[acc].transform('last')\n\n        df[f'{acc}_mean'] = df.groupby('Id')[acc].transform('mean')\n        df[f'{acc}_median'] = df.groupby('Id')[acc].transform('median')\n        df[f'{acc}_std'] = df.groupby('Id')[acc].transform('std')\n\n        df[f'{acc}_min'] = df.groupby('Id')[acc].transform('min')\n        df[f'{acc}_max'] = df.groupby('Id')[acc].transform('max')\n\n        df[f'{acc}_delta'] = df[f'{acc}_max'] - df[f'{acc}_min']\n        for lag in [1,2,3]:\n            df[f'ma_{lag}_{acc}'] = df[acc].rolling(lag).mean().fillna(method=\"backfill\")\n            df[f'ma{lag}_{acc}'] = df[acc].rolling(lag).mean().fillna(method=\"backfill\")\n            df[f'ma_{lag}_{acc}'] = df[acc].rolling(lag).mean().fillna(method=\"backfill\")\n        \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-06-10T23:08:50.386819Z","iopub.execute_input":"2023-06-10T23:08:50.387318Z","iopub.status.idle":"2023-06-10T23:08:50.402504Z","shell.execute_reply.started":"2023-06-10T23:08:50.387271Z","shell.execute_reply":"2023-06-10T23:08:50.400870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply the above function to our dataset\ndata_sample = create_features(defog_data_valid)","metadata":{"execution":{"iopub.status.busy":"2023-06-10T23:08:50.405603Z","iopub.execute_input":"2023-06-10T23:08:50.406000Z","iopub.status.idle":"2023-06-10T23:09:11.420239Z","shell.execute_reply.started":"2023-06-10T23:08:50.405968Z","shell.execute_reply":"2023-06-10T23:09:11.418891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a function to reduce memory usage\ndef reduce_mem_usage(df):\n    \n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n\n    for col in df.columns:\n        col_type = df[col].dtype.name\n\n        if col_type not in ['object', 'category', 'datetime64[ns, UTC]']:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n   \n    return df ","metadata":{"execution":{"iopub.status.busy":"2023-06-10T23:09:11.422148Z","iopub.execute_input":"2023-06-10T23:09:11.422992Z","iopub.status.idle":"2023-06-10T23:09:11.436926Z","shell.execute_reply.started":"2023-06-10T23:09:11.422955Z","shell.execute_reply":"2023-06-10T23:09:11.435679Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# Apply the above function to our dataset\nfull_data = reduce_mem_usage(data_sample)","metadata":{"execution":{"iopub.status.busy":"2023-06-10T23:09:11.438710Z","iopub.execute_input":"2023-06-10T23:09:11.439740Z","iopub.status.idle":"2023-06-10T23:09:14.551245Z","shell.execute_reply.started":"2023-06-10T23:09:11.439705Z","shell.execute_reply":"2023-06-10T23:09:14.549906Z"},"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create features and targets\ndrop_cols = ['Time', 'StartHesitation', 'Turn', 'Walking', 'Id', 'dataset']\n\nX = full_data.drop(drop_cols, axis=1)\ny = full_data[\"Turn\"]\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=42, stratify=y)\n\nX_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-10T23:09:14.552477Z","iopub.execute_input":"2023-06-10T23:09:14.552802Z","iopub.status.idle":"2023-06-10T23:09:24.439463Z","shell.execute_reply.started":"2023-06-10T23:09:14.552776Z","shell.execute_reply":"2023-06-10T23:09:24.438339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So our training dataset has 2_863_371 samples. Let's go further.","metadata":{}},{"cell_type":"markdown","source":"### Let's train our models","metadata":{}},{"cell_type":"code","source":"# Сreate function for testing models\ndef test_model(algorithm, X_train, y_train, X_test, y_test):\n    \n    model = algorithm()\n    model.fit(X_train, y_train)\n    y_pred = model.predict(X_test)\n    f1 = f1_score(y_test, y_pred)\n    accuracy = accuracy_score(y_test, y_pred)\n\n        \n    return f1, accuracy","metadata":{"execution":{"iopub.status.busy":"2023-06-10T23:09:24.440882Z","iopub.execute_input":"2023-06-10T23:09:24.441247Z","iopub.status.idle":"2023-06-10T23:09:24.447968Z","shell.execute_reply.started":"2023-06-10T23:09:24.441219Z","shell.execute_reply":"2023-06-10T23:09:24.446642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### XGBoost","metadata":{"execution":{"iopub.status.busy":"2023-06-10T21:40:08.116902Z","iopub.execute_input":"2023-06-10T21:40:08.117336Z","iopub.status.idle":"2023-06-10T21:40:08.124165Z","shell.execute_reply.started":"2023-06-10T21:40:08.117303Z","shell.execute_reply":"2023-06-10T21:40:08.122719Z"}}},{"cell_type":"code","source":"%%time\n\nf1_xgb, accuracy_xgb = test_model(XGBClassifier, X_train, y_train, X_test, y_test)\n\nprint(f\"XGBClassifier | F1 score - {f1_xgb}, | Accuracy - {accuracy_xgb}\")","metadata":{"execution":{"iopub.status.busy":"2023-06-10T23:09:24.449969Z","iopub.execute_input":"2023-06-10T23:09:24.450792Z","iopub.status.idle":"2023-06-10T23:27:44.577859Z","shell.execute_reply.started":"2023-06-10T23:09:24.450748Z","shell.execute_reply":"2023-06-10T23:27:44.576486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### LightGBM","metadata":{}},{"cell_type":"code","source":"%%time\n\nf1_lgbm, accuracy_lgbm = test_model(LGBMClassifier, X_train, y_train, X_test, y_test)\n\nprint(f\"LGBMClassifier | F1 score - {f1_lgbm}, | Accuracy - {accuracy_lgbm}\") ","metadata":{"execution":{"iopub.status.busy":"2023-06-10T23:27:44.582837Z","iopub.execute_input":"2023-06-10T23:27:44.583649Z","iopub.status.idle":"2023-06-10T23:28:58.829954Z","shell.execute_reply.started":"2023-06-10T23:27:44.583601Z","shell.execute_reply":"2023-06-10T23:28:58.828620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### CatBoostClassifier","metadata":{}},{"cell_type":"code","source":"%%time\n\nf1_cb, accuracy_cb = test_model(CatBoostClassifier, X_train, y_train, X_test, y_test)\n\nprint(f\"CatBoostClassifier | F1 score - {f1_cb}, | Accuracy - {accuracy_cb}\")","metadata":{"execution":{"iopub.status.busy":"2023-06-10T23:28:58.831282Z","iopub.execute_input":"2023-06-10T23:28:58.831617Z","iopub.status.idle":"2023-06-10T23:44:09.952521Z","shell.execute_reply.started":"2023-06-10T23:28:58.831574Z","shell.execute_reply":"2023-06-10T23:44:09.951315Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.dummy import DummyClassifier\n\ndummy_clf = DummyClassifier(strategy=\"most_frequent\", random_state=42)\ndummy_clf.fit(X_train, y_train)\ndummy_clf.predict(X_train)\naccuracy_dummy = dummy_clf.score(X_train, y_train)\n\nprint(\"DummyClassifier | accuracy\" , accuracy_dummy)","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-06-10T23:44:09.954033Z","iopub.execute_input":"2023-06-10T23:44:09.954377Z","iopub.status.idle":"2023-06-10T23:44:10.290422Z","shell.execute_reply.started":"2023-06-10T23:44:09.954346Z","shell.execute_reply":"2023-06-10T23:44:10.288943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create summary table\nindex = [\"CatBoostClassifier\",\n         \"XGBClassifier\", \n         \"LGBMClassifier\",\n         \"DummyClassifier\",\n        ]\n\ndata = {'F1 score':[f1_cb.round(5),\n                    f1_xgb.round(5),\n                    f1_lgbm.round(5),\n                    \"-\"],\n        'Accuracy':[accuracy_cb.round(5),\n                    accuracy_xgb.round(5),\n                    accuracy_lgbm.round(5),\n                    accuracy_dummy.round(5)],\n        \n        'Learning time':[\"CPU times: user 51min 13s\",\n                         \"CPU times: user 1h 10min 33s\",\n                         \"CPU times: user 4min 28s\",\n                         \"-\"],\n        }\n\npd.DataFrame(data=data, index=index)","metadata":{"execution":{"iopub.status.busy":"2023-06-10T23:44:47.731231Z","iopub.execute_input":"2023-06-10T23:44:47.731662Z","iopub.status.idle":"2023-06-10T23:44:47.753292Z","shell.execute_reply.started":"2023-06-10T23:44:47.731621Z","shell.execute_reply":"2023-06-10T23:44:47.751990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have a winner! This is CatBoostClassifier. Algorithm CatBoostClassifier model works better (for this data at least).","metadata":{}}]}