{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np \nimport pandas as pd\nimport os\nfrom tqdm import tqdm\nfrom glob import glob\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn import ensemble, linear_model\nfrom sklearn.metrics import average_precision_score\n\nROOT = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/\"\nFEATURES = [\"AccV\", \"AccML\", \"AccAP\"]\nLABELS = [\"StartHesitation\", \"Turn\", \"Walking\"]\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-15T23:28:47.056046Z","iopub.execute_input":"2023-04-15T23:28:47.056710Z","iopub.status.idle":"2023-04-15T23:28:48.624911Z","shell.execute_reply.started":"2023-04-15T23:28:47.056679Z","shell.execute_reply":"2023-04-15T23:28:48.623469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## https://www.kaggle.com/code/arjanso/reducing-dataframe-memory-size-by-65\ndef reduce_memory_usage(df):\n    \n    start_mem = df.memory_usage().sum() / 1024 ** 2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024 ** 2 \n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-04-15T23:28:48.630508Z","iopub.execute_input":"2023-04-15T23:28:48.630811Z","iopub.status.idle":"2023-04-15T23:28:48.646804Z","shell.execute_reply.started":"2023-04-15T23:28:48.630774Z","shell.execute_reply":"2023-04-15T23:28:48.645236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_data(dataset, datatype):\n    \n    metadata = pd.read_csv(os.path.join(ROOT,dataset + \"_metadata.csv\"))\n    \n    file_path = os.path.join(ROOT, datatype, dataset)\n    \n    df_res = pd.DataFrame()\n    for _, _, files in os.walk(file_path):\n        for name in files:\n            f = os.path.join(file_path, name)\n            csv_file = pd.read_csv(f)\n            csv_file[\"file\"] = name.replace(\".csv\", \"\")\n            df_res = pd.concat([df_res,csv_file])\n    \n    df_res = metadata.merge(df_res, how = 'inner', left_on = 'Id', right_on = 'file')\n    df_res = df_res.drop([\"file\"], axis = 1)\n\n    df_res = reduce_memory_usage(df_res)\n        \n    return df_res\n\n","metadata":{"execution":{"iopub.status.busy":"2023-04-15T23:28:48.648548Z","iopub.execute_input":"2023-04-15T23:28:48.649402Z","iopub.status.idle":"2023-04-15T23:28:48.662706Z","shell.execute_reply.started":"2023-04-15T23:28:48.649333Z","shell.execute_reply":"2023-04-15T23:28:48.661437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_defog = read_data('defog','train')\ndf_train_tdcsfog = read_data('tdcsfog', 'train')","metadata":{"execution":{"iopub.status.busy":"2023-04-16T00:37:18.946317Z","iopub.execute_input":"2023-04-16T00:37:18.946677Z","iopub.status.idle":"2023-04-16T00:40:46.103166Z","shell.execute_reply.started":"2023-04-16T00:37:18.946647Z","shell.execute_reply":"2023-04-16T00:40:46.101708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train_defog)\nprint(df_train_tdcsfog)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T00:46:51.844848Z","iopub.execute_input":"2023-04-16T00:46:51.845266Z","iopub.status.idle":"2023-04-16T00:46:51.870668Z","shell.execute_reply.started":"2023-04-16T00:46:51.845230Z","shell.execute_reply":"2023-04-16T00:46:51.869737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_defog.AccV = df_train_defog.AccV * 9.80665\ndf_train_defog.AccML = df_train_defog.AccML * 9.80665\ndf_train_defog.AccAP = df_train_defog.AccAP * 9.80665","metadata":{"execution":{"iopub.status.busy":"2023-04-16T00:47:02.331398Z","iopub.execute_input":"2023-04-16T00:47:02.332361Z","iopub.status.idle":"2023-04-16T00:47:02.798079Z","shell.execute_reply.started":"2023-04-16T00:47:02.332310Z","shell.execute_reply":"2023-04-16T00:47:02.797080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train_defog)\nprint(df_train_tdcsfog)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T00:47:03.850317Z","iopub.execute_input":"2023-04-16T00:47:03.852577Z","iopub.status.idle":"2023-04-16T00:47:03.876249Z","shell.execute_reply.started":"2023-04-16T00:47:03.852535Z","shell.execute_reply":"2023-04-16T00:47:03.875097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.concat([df_train_defog, df_train_tdcsfog])","metadata":{"execution":{"iopub.status.busy":"2023-04-16T00:47:27.173132Z","iopub.execute_input":"2023-04-16T00:47:27.173529Z","iopub.status.idle":"2023-04-16T00:47:28.909674Z","shell.execute_reply.started":"2023-04-16T00:47:27.173499Z","shell.execute_reply":"2023-04-16T00:47:28.908326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = train_test_split(df_train[FEATURES], df_train[LABELS], test_size=0.30, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T00:47:28.911805Z","iopub.execute_input":"2023-04-16T00:47:28.912299Z","iopub.status.idle":"2023-04-16T00:47:33.768493Z","shell.execute_reply.started":"2023-04-16T00:47:28.912259Z","shell.execute_reply":"2023-04-16T00:47:33.767284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import xgboost as xgb\ndtrain_reg = xgb.DMatrix(X_train, y_train, enable_categorical=True)\ndtest_reg = xgb.DMatrix(X_valid, y_valid, enable_categorical=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T00:47:33.770728Z","iopub.execute_input":"2023-04-16T00:47:33.771199Z","iopub.status.idle":"2023-04-16T00:47:36.585849Z","shell.execute_reply.started":"2023-04-16T00:47:33.771160Z","shell.execute_reply":"2023-04-16T00:47:36.584730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {\"objective\": \"reg:squarederror\", \"tree_method\": \"gpu_hist\"}\nn = 100\nmodel = xgb.train(\n    params=params,\n    dtrain=dtrain_reg,\n    num_boost_round=n,\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T00:48:37.873912Z","iopub.execute_input":"2023-04-16T00:48:37.875033Z","iopub.status.idle":"2023-04-16T00:48:51.760453Z","shell.execute_reply.started":"2023-04-16T00:48:37.874993Z","shell.execute_reply":"2023-04-16T00:48:51.759124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(average_precision_score(y_valid, model.predict(dtest_reg).clip(0.0,1.0)))\n","metadata":{"execution":{"iopub.status.busy":"2023-04-16T00:48:51.762631Z","iopub.execute_input":"2023-04-16T00:48:51.762998Z","iopub.status.idle":"2023-04-16T00:49:28.049277Z","shell.execute_reply.started":"2023-04-16T00:48:51.762971Z","shell.execute_reply":"2023-04-16T00:49:28.047941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(os.path.join(ROOT,'sample_submission.csv'))\ntest = glob(os.path.join(ROOT,'test/**/**'))\n\nsample_submission['sample'] = 0\nsubmission = []\nfor file in test:\n    df = pd.read_csv(file)\n    df['Id'] = file.split('/')[-1].split('.')[0]\n    df = df.fillna(0).reset_index(drop=True)\n    res = pd.DataFrame(np.round(model.predict(xgb.DMatrix(df[FEATURES])),3), columns=LABELS)\n    df = pd.concat([df,res], axis=1)\n    df['Id'] = df['Id'].astype(str) + '_' + df['Time'].astype(str)\n    submission.append(df[['Id','StartHesitation', 'Turn' , 'Walking']])\nsubmission = pd.concat(submission)\nsubmission = pd.merge(sample_submission[['Id','sample']], submission, how='left', on='Id').fillna(0.0)\nsubmission[['Id','StartHesitation', 'Turn' , 'Walking']].to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T00:49:50.675496Z","iopub.execute_input":"2023-04-16T00:49:50.675898Z","iopub.status.idle":"2023-04-16T00:49:53.859764Z","shell.execute_reply.started":"2023-04-16T00:49:50.675863Z","shell.execute_reply":"2023-04-16T00:49:53.858397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}