{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# import library\nimport datetime\na=datetime.datetime.now()\nimport os\nimport random\nimport cv2\nimport pandas as pd\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-03-30T05:58:01.757251Z","iopub.execute_input":"2023-03-30T05:58:01.758293Z","iopub.status.idle":"2023-03-30T05:58:01.986021Z","shell.execute_reply.started":"2023-03-30T05:58:01.758251Z","shell.execute_reply":"2023-03-30T05:58:01.985094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reduce Memory Usage\n# reference : https://www.kaggle.com/code/arjanso/reducing-dataframe-memory-size-by-65 @ARJANGROEN\n\ndef reduce_memory_usage(df):\n    \n    \n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024**2 \n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-03-30T05:58:01.988149Z","iopub.execute_input":"2023-03-30T05:58:01.988533Z","iopub.status.idle":"2023-03-30T05:58:02.001607Z","shell.execute_reply.started":"2023-03-30T05:58:01.988480Z","shell.execute_reply":"2023-03-30T05:58:02.000557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#reference: https://www.kaggle.com/code/ghrangel/read-data-and-merge\n\nDATA_ROOT_DEFOG = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/'\ndefog = pd.DataFrame()\nfor root, dirs, files in os.walk(DATA_ROOT_DEFOG):\n    for name in files:       \n        f = os.path.join(root, name)\n        df_list= pd.read_csv(f)\n        words = name.split('.')[0]\n        df_list['file']= name.split('.')[0]\n        defog = pd.concat([defog, df_list], axis=0)\n\ndefog\n       ","metadata":{"execution":{"iopub.status.busy":"2023-03-30T05:58:02.003376Z","iopub.execute_input":"2023-03-30T05:58:02.004112Z","iopub.status.idle":"2023-03-30T05:58:50.959124Z","shell.execute_reply.started":"2023-03-30T05:58:02.004072Z","shell.execute_reply":"2023-03-30T05:58:50.958128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog = reduce_memory_usage(defog)","metadata":{"execution":{"iopub.status.busy":"2023-03-30T05:58:50.961863Z","iopub.execute_input":"2023-03-30T05:58:50.962232Z","iopub.status.idle":"2023-03-30T05:58:53.081891Z","shell.execute_reply.started":"2023-03-30T05:58:50.962196Z","shell.execute_reply":"2023-03-30T05:58:53.080590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> #### 💬 we reduced memory usage from 954MB to 335MB","metadata":{}},{"cell_type":"code","source":"defog = defog[(defog['Task']==1)&(defog['Valid']==1)]","metadata":{"execution":{"iopub.status.busy":"2023-03-30T05:58:53.083568Z","iopub.execute_input":"2023-03-30T05:58:53.084245Z","iopub.status.idle":"2023-03-30T05:58:53.532132Z","shell.execute_reply.started":"2023-03-30T05:58:53.084200Z","shell.execute_reply":"2023-03-30T05:58:53.531097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> #### 💬 As I mentioned above, We are going to use valid data only.","metadata":{}},{"cell_type":"code","source":"print('the shape of defog dataset is {}'.format(defog.shape))","metadata":{"execution":{"iopub.status.busy":"2023-03-30T05:58:53.533526Z","iopub.execute_input":"2023-03-30T05:58:53.534533Z","iopub.status.idle":"2023-03-30T05:58:53.540579Z","shell.execute_reply.started":"2023-03-30T05:58:53.534493Z","shell.execute_reply":"2023-03-30T05:58:53.539389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> #### 💬 Now it's time to combine it with metadata.","metadata":{}},{"cell_type":"code","source":"defog_metadata = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv\")\ndefog_metadata","metadata":{"execution":{"iopub.status.busy":"2023-03-30T05:58:53.542007Z","iopub.execute_input":"2023-03-30T05:58:53.543097Z","iopub.status.idle":"2023-03-30T05:58:53.568020Z","shell.execute_reply.started":"2023-03-30T05:58:53.543059Z","shell.execute_reply":"2023-03-30T05:58:53.567091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_m= defog_metadata.merge(defog, how = 'inner', left_on = 'Id', right_on = 'file')\ndefog_m.drop(['file','Valid','Task'], axis = 1, inplace = True)\ndefog_m","metadata":{"execution":{"iopub.status.busy":"2023-03-30T05:58:53.569204Z","iopub.execute_input":"2023-03-30T05:58:53.569522Z","iopub.status.idle":"2023-03-30T05:58:55.416621Z","shell.execute_reply.started":"2023-03-30T05:58:53.569487Z","shell.execute_reply":"2023-03-30T05:58:55.415557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# summary table function\ndef summary(df):\n    print(f'data shape: {df.shape}')\n    summ = pd.DataFrame(df.dtypes, columns=['data type'])\n    summ['#missing'] = df.isnull().sum().values * 100\n    summ['%missing'] = df.isnull().sum().values / len(df)\n    summ['#unique'] = df.nunique().values\n    desc = pd.DataFrame(df.describe(include='all').transpose())\n    summ['min'] = desc['min'].values\n    summ['max'] = desc['max'].values\n    summ['first value'] = df.loc[0].values\n    summ['second value'] = df.loc[1].values\n    summ['third value'] = df.loc[2].values\n    \n    return summ","metadata":{"execution":{"iopub.status.busy":"2023-03-30T05:58:55.418276Z","iopub.execute_input":"2023-03-30T05:58:55.418653Z","iopub.status.idle":"2023-03-30T05:58:55.426501Z","shell.execute_reply.started":"2023-03-30T05:58:55.418613Z","shell.execute_reply":"2023-03-30T05:58:55.425371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> #### 💬 Let's look at the summary table for defog dataset (data from subjects' home)","metadata":{}},{"cell_type":"code","source":"summary(defog_m)","metadata":{"execution":{"iopub.status.busy":"2023-03-30T05:58:55.432378Z","iopub.execute_input":"2023-03-30T05:58:55.432752Z","iopub.status.idle":"2023-03-30T05:58:59.360360Z","shell.execute_reply.started":"2023-03-30T05:58:55.432714Z","shell.execute_reply":"2023-03-30T05:58:59.359114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# garbage collection for memory\nimport gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-30T05:58:59.362075Z","iopub.execute_input":"2023-03-30T05:58:59.362779Z","iopub.status.idle":"2023-03-30T05:58:59.454110Z","shell.execute_reply.started":"2023-03-30T05:58:59.362736Z","shell.execute_reply":"2023-03-30T05:58:59.452931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> #### 💬 prepare tdcsfog dataset for modeling (data collected from the lab🥼)","metadata":{}},{"cell_type":"code","source":"DATA_ROOT_TDCSFOG = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/'\ntdcsfog = pd.DataFrame()\nfor root, dirs, files in os.walk(DATA_ROOT_TDCSFOG):\n    for name in files:       \n        f = os.path.join(root, name)\n        df_list= pd.read_csv(f)\n        words = name.split('.')[0]\n        df_list['file']= name.split('.')[0]\n        tdcsfog = pd.concat([tdcsfog, df_list], axis=0)\ntdcsfog","metadata":{"execution":{"iopub.status.busy":"2023-03-30T05:58:59.455606Z","iopub.execute_input":"2023-03-30T05:58:59.456219Z","iopub.status.idle":"2023-03-30T06:01:08.446749Z","shell.execute_reply.started":"2023-03-30T05:58:59.456177Z","shell.execute_reply":"2023-03-30T06:01:08.445732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog = reduce_memory_usage(tdcsfog)","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:01:08.448255Z","iopub.execute_input":"2023-03-30T06:01:08.449111Z","iopub.status.idle":"2023-03-30T06:01:09.467436Z","shell.execute_reply.started":"2023-03-30T06:01:08.449037Z","shell.execute_reply":"2023-03-30T06:01:09.466239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> #### 💬 we reduced memory usage from 484MB to 154MB","metadata":{}},{"cell_type":"code","source":"tdcsfog_metadata = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv\")\ntdcsfog_metadata","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:01:09.468891Z","iopub.execute_input":"2023-03-30T06:01:09.469563Z","iopub.status.idle":"2023-03-30T06:01:09.490899Z","shell.execute_reply.started":"2023-03-30T06:01:09.469520Z","shell.execute_reply":"2023-03-30T06:01:09.489714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_m= tdcsfog_metadata.merge(tdcsfog, how = 'inner', left_on = 'Id', right_on = 'file')\ntdcsfog_m.drop(['file'], axis = 1, inplace = True)\ntdcsfog_m","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:01:09.492317Z","iopub.execute_input":"2023-03-30T06:01:09.492772Z","iopub.status.idle":"2023-03-30T06:01:12.622022Z","shell.execute_reply.started":"2023-03-30T06:01:09.492731Z","shell.execute_reply":"2023-03-30T06:01:12.621036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# garbage collection for memory\nimport gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:01:12.623668Z","iopub.execute_input":"2023-03-30T06:01:12.624095Z","iopub.status.idle":"2023-03-30T06:01:12.715014Z","shell.execute_reply.started":"2023-03-30T06:01:12.624038Z","shell.execute_reply":"2023-03-30T06:01:12.713723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### <p style=\"font-family:JetBrains Mono; font-weight:bold; letter-spacing: 2px; color:#006600; font-size:140%; text-align:left;padding: 0px; border-bottom: 3px solid #003300\">✅ Feature engineering and modeling</p>","metadata":{}},{"cell_type":"markdown","source":"<div style=\"border-radius:10px; border:#DEB887 solid; padding: 15px; background-color: #FFFAF0; font-size:100%; text-align:left\">\n\n<h3 align=\"left\"><font color='#DEB887'>💡 Notes:</font></h3>\n\n* For illustrative purpose, we will develop very simple multi-classification model with LGBM.\n\n* This is time series data, therefore we should creat time-related varaible so that it could reflect the change along with the time.\n    \n* On this notebook, I wiil skip time series feature engineering process for now.\n    \n* In the ground truth, only one event class has a non-zero value for each Id, but there is no restriction on the values of predicted scores. -> multi-class task!","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import KFold, StratifiedKFold, train_test_split, GridSearchCV\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:01:12.716856Z","iopub.execute_input":"2023-03-30T06:01:12.717304Z","iopub.status.idle":"2023-03-30T06:01:13.079218Z","shell.execute_reply.started":"2023-03-30T06:01:12.717263Z","shell.execute_reply":"2023-03-30T06:01:13.078234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conditions = [\n    (defog_m['StartHesitation'] == 1),\n    (defog_m['Turn'] == 1),\n    (defog_m['Walking'] == 1)]\nchoices = ['StartHesitation', 'Turn', 'Walking']\ndefog_m['event'] = np.select(conditions, choices, default='Normal')","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:01:13.080531Z","iopub.execute_input":"2023-03-30T06:01:13.081406Z","iopub.status.idle":"2023-03-30T06:01:14.019975Z","shell.execute_reply.started":"2023-03-30T06:01:13.081360Z","shell.execute_reply":"2023-03-30T06:01:14.018370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_m['event'].value_counts().to_frame().style.background_gradient()","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:01:14.025786Z","iopub.execute_input":"2023-03-30T06:01:14.026213Z","iopub.status.idle":"2023-03-30T06:01:14.672329Z","shell.execute_reply.started":"2023-03-30T06:01:14.026171Z","shell.execute_reply":"2023-03-30T06:01:14.671106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> #### 💬 Turn is the most frequently occured event while StartHesitation rarely occurs...","metadata":{}},{"cell_type":"code","source":"train_df = defog_m[['AccV','AccML','AccAP','event']]","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:01:14.673796Z","iopub.execute_input":"2023-03-30T06:01:14.674888Z","iopub.status.idle":"2023-03-30T06:01:15.296496Z","shell.execute_reply.started":"2023-03-30T06:01:14.674843Z","shell.execute_reply":"2023-03-30T06:01:15.295453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> #### 💬 As I mentioned on the notes, I will skip feature engineering process and just use three sensor data as inputs of the model. This model does not consider time-related effect.","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()\n\ntrain_df['target'] = le.fit_transform(train_df['event'])","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:01:15.298165Z","iopub.execute_input":"2023-03-30T06:01:15.298549Z","iopub.status.idle":"2023-03-30T06:01:16.117263Z","shell.execute_reply.started":"2023-03-30T06:01:15.298511Z","shell.execute_reply":"2023-03-30T06:01:16.116171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_df.drop(['event','target'], axis=1)\ny = train_df['target']","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:01:16.119176Z","iopub.execute_input":"2023-03-30T06:01:16.119602Z","iopub.status.idle":"2023-03-30T06:01:16.139988Z","shell.execute_reply.started":"2023-03-30T06:01:16.119540Z","shell.execute_reply":"2023-03-30T06:01:16.139081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> #### 💬 train simple LGBM model without hyper-parameter tuning. the size of dataset is quite huge, it might take a lot of time for traing a decent model.","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb\n\n\n# split dataset into training and test set\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=1004)\n\n#Converting the dataset in proper LGB format\nd_train=lgb.Dataset(X_train, label=y_train)\n#setting up the parameters\nparams={}\nparams['learning_rate']=0.03\nparams['boosting_type']='gbdt' #GradientBoostingDecisionTree\nparams['objective']='multiclass' #Multi-class target feature\nparams['metric']='multi_logloss' #metric for multi-class\nparams['max_depth']=7\nparams['num_class']=4 #no.of unique values in the target class not inclusive of the end value\nparams['verbose']=-1\n#training the model\nclf=lgb.train(params,d_train,1000)  #training the model on 1,000 epocs\n#prediction on the test dataset\ny_pred_1=clf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:01:16.141771Z","iopub.execute_input":"2023-03-30T06:01:16.142169Z","iopub.status.idle":"2023-03-30T06:13:42.222741Z","shell.execute_reply.started":"2023-03-30T06:01:16.142129Z","shell.execute_reply":"2023-03-30T06:13:42.221856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> #### 💬 Let's look at what it predicts","metadata":{}},{"cell_type":"code","source":"y_pred_1[:1]","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:13:42.226874Z","iopub.execute_input":"2023-03-30T06:13:42.227628Z","iopub.status.idle":"2023-03-30T06:13:42.235653Z","shell.execute_reply.started":"2023-03-30T06:13:42.227593Z","shell.execute_reply":"2023-03-30T06:13:42.234755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 'macro' option is to calculate metrics for each label, and find their unweighted mean. \n# This does not take label imbalance into account.\nfrom sklearn.metrics import precision_score\nprecision_score(y_test, np.argmax(y_pred_1, axis=-1), average='macro')","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:13:42.237325Z","iopub.execute_input":"2023-03-30T06:13:42.238085Z","iopub.status.idle":"2023-03-30T06:13:42.406192Z","shell.execute_reply.started":"2023-03-30T06:13:42.238013Z","shell.execute_reply":"2023-03-30T06:13:42.405122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"> #### 💬 Creat inference table (test dataset) and make prediction","metadata":{}},{"cell_type":"code","source":"test_defog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog/02ab235146.csv'\ntest_defog = pd.read_csv(test_defog_path)\nname = os.path.basename(test_defog_path)\nid_value = name.split('.')[0]\ntest_defog['Id_value'] = id_value\ntest_defog['Id'] = test_defog['Id_value'].astype(str) + '_' + test_defog['Time'].astype(str)\ntest_defog = test_defog[['Id','AccV','AccML','AccAP']]\ntest_defog.set_index('Id',inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:13:42.407851Z","iopub.execute_input":"2023-03-30T06:13:42.408274Z","iopub.status.idle":"2023-03-30T06:13:43.048611Z","shell.execute_reply.started":"2023-03-30T06:13:42.408233Z","shell.execute_reply":"2023-03-30T06:13:43.047569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predict event probability\ntest_defog_pred=clf.predict(test_defog)\ntest_defog['event'] = np.argmax(test_defog_pred, axis=-1)","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:13:43.049977Z","iopub.execute_input":"2023-03-30T06:13:43.050388Z","iopub.status.idle":"2023-03-30T06:14:39.423352Z","shell.execute_reply.started":"2023-03-30T06:13:43.050342Z","shell.execute_reply":"2023-03-30T06:14:39.422447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# expand event column it to three columns\ntest_defog['StartHesitation'] = np.where(test_defog['event']==1, 1, 0)\ntest_defog['Turn'] = np.where(test_defog['event']==2, 1, 0)\ntest_defog['Walking'] = np.where(test_defog['event']==3, 1, 0)","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:14:39.429609Z","iopub.execute_input":"2023-03-30T06:14:39.430196Z","iopub.status.idle":"2023-03-30T06:14:39.441319Z","shell.execute_reply.started":"2023-03-30T06:14:39.430156Z","shell.execute_reply":"2023-03-30T06:14:39.440210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_defog.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:14:39.442597Z","iopub.execute_input":"2023-03-30T06:14:39.443691Z","iopub.status.idle":"2023-03-30T06:14:39.458707Z","shell.execute_reply.started":"2023-03-30T06:14:39.443651Z","shell.execute_reply":"2023-03-30T06:14:39.457460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> #### 💬 apply the same process for tdcsfog dataset, but I am not going to train another model for tdcsfog. Instead, I will just use the same model trained from defog dataset. I recommend you to develop two different model because the data distribution is quite different.","metadata":{}},{"cell_type":"code","source":"test_tdcsfog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog/003f117e14.csv'\ntest_tdcsfog = pd.read_csv(test_tdcsfog_path)\nname = os.path.basename(test_tdcsfog_path)\nid_value = name.split('.')[0]\ntest_tdcsfog['Id_value'] = id_value\ntest_tdcsfog['Id'] = test_tdcsfog['Id_value'].astype(str) + '_' + test_tdcsfog['Time'].astype(str)\ntest_tdcsfog = test_tdcsfog[['Id','AccV','AccML','AccAP']]\ntest_tdcsfog.set_index('Id',inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:14:39.460333Z","iopub.execute_input":"2023-03-30T06:14:39.460697Z","iopub.status.idle":"2023-03-30T06:14:39.493964Z","shell.execute_reply.started":"2023-03-30T06:14:39.460662Z","shell.execute_reply":"2023-03-30T06:14:39.493082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tdcsfog_pred=clf.predict(test_tdcsfog)\ntest_tdcsfog['event'] = np.argmax(test_tdcsfog_pred, axis=-1)","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:14:39.495297Z","iopub.execute_input":"2023-03-30T06:14:39.495612Z","iopub.status.idle":"2023-03-30T06:14:40.314820Z","shell.execute_reply.started":"2023-03-30T06:14:39.495578Z","shell.execute_reply":"2023-03-30T06:14:40.314000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tdcsfog['StartHesitation'] = np.where(test_tdcsfog['event']==1, 1, 0)\ntest_tdcsfog['Turn'] = np.where(test_tdcsfog['event']==2, 1, 0)\ntest_tdcsfog['Walking'] = np.where(test_tdcsfog['event']==3, 1, 0)\ntest_tdcsfog.reset_index('Id', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:14:40.318727Z","iopub.execute_input":"2023-03-30T06:14:40.320940Z","iopub.status.idle":"2023-03-30T06:14:40.330205Z","shell.execute_reply.started":"2023-03-30T06:14:40.320904Z","shell.execute_reply":"2023-03-30T06:14:40.329268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tdcsfog.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:14:40.333477Z","iopub.execute_input":"2023-03-30T06:14:40.333754Z","iopub.status.idle":"2023-03-30T06:14:40.349447Z","shell.execute_reply.started":"2023-03-30T06:14:40.333728Z","shell.execute_reply":"2023-03-30T06:14:40.348249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = pd.concat([test_tdcsfog,test_defog])\nsubmit = submit[['Id', 'StartHesitation', 'Turn','Walking']]","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:14:40.351133Z","iopub.execute_input":"2023-03-30T06:14:40.351473Z","iopub.status.idle":"2023-03-30T06:14:40.377815Z","shell.execute_reply.started":"2023-03-30T06:14:40.351439Z","shell.execute_reply":"2023-03-30T06:14:40.376768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:14:40.379428Z","iopub.execute_input":"2023-03-30T06:14:40.379857Z","iopub.status.idle":"2023-03-30T06:14:40.393330Z","shell.execute_reply.started":"2023-03-30T06:14:40.379815Z","shell.execute_reply":"2023-03-30T06:14:40.392112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> #### 💬 Let's compare it with sample submission data.","metadata":{}},{"cell_type":"code","source":"sample = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/sample_submission.csv')\nsample.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:14:40.394800Z","iopub.execute_input":"2023-03-30T06:14:40.395208Z","iopub.status.idle":"2023-03-30T06:14:41.034791Z","shell.execute_reply.started":"2023-03-30T06:14:40.395170Z","shell.execute_reply":"2023-03-30T06:14:41.033720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample.head(10)\nb=datetime.datetime.now()\nprint(b-a)","metadata":{"execution":{"iopub.status.busy":"2023-03-30T06:14:41.036067Z","iopub.execute_input":"2023-03-30T06:14:41.036716Z","iopub.status.idle":"2023-03-30T06:14:41.043373Z","shell.execute_reply.started":"2023-03-30T06:14:41.036670Z","shell.execute_reply":"2023-03-30T06:14:41.042272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"in progress....","metadata":{}}]}