{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Parkinsons' Freezing of Gait Rectangular Data\n\nThe purpose of this notebook is to show how to use the rectangular dataset builder object for the testing dataset.","metadata":{}},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import utilityfogrectangulardataset as fr\nimport pandas as pd\npd.set_option('display.max_columns', None)\nimport numpy as np\nfrom tqdm import tqdm\nfrom hmmlearn import hmm\nfrom sklearn.preprocessing import LabelEncoder","metadata":{"execution":{"iopub.status.busy":"2023-06-13T13:53:25.445705Z","iopub.execute_input":"2023-06-13T13:53:25.446894Z","iopub.status.idle":"2023-06-13T13:53:27.611766Z","shell.execute_reply.started":"2023-06-13T13:53:25.446852Z","shell.execute_reply":"2023-06-13T13:53:27.610230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Helper Functions","metadata":{}},{"cell_type":"markdown","source":"## Cleanup + Feature Engineering","metadata":{}},{"cell_type":"code","source":"PREDS_COLS = ['AccV', 'AccML', 'AccAP', 'Medication', 'Age', 'isMale', 'YearsSinceDx', 'AccV_lag1', 'AccML_lag1', 'AccAP_lag1']","metadata":{"execution":{"iopub.status.busy":"2023-06-13T13:53:27.615270Z","iopub.execute_input":"2023-06-13T13:53:27.616242Z","iopub.status.idle":"2023-06-13T13:53:27.623029Z","shell.execute_reply.started":"2023-06-13T13:53:27.616195Z","shell.execute_reply":"2023-06-13T13:53:27.621255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def Clean_Dataset(tdf, dtype_train=True):\n    fdf = tdf.copy(deep=True)\n\n    # convert to numericals\n    sex_map = {'M': 1, 'F': 0}\n    fdf['isMale'] = fdf['Sex'].map(sex_map)\n\n    # make the Id column into the trial id and the time stamp\n    fdf['Id'] = fdf['Id'].astype(str) + '_' + fdf['Time'].astype(str)\n\n    # specify acc data for only tdcs data\n    if fdf['Dataset'].iloc[0] == 'tdcsfog':\n        fdf = fdf[fdf['Dataset'] == 'tdcsfog']\n        fdf['AccV'] /= 9.8\n        fdf['AccML'] /= 9.8\n        fdf['AccAP'] /= 9.8\n\n    # apply the lag to accelerometer data\n    lag_cols = ['AccV', 'AccML', 'AccAP']\n    for col in lag_cols:\n        fdf[f'{col}_lag1'] = fdf[col].shift(1)\n\n    # create the \"FOG_Events\" column\n    if dtype_train:\n        mapping = {\n            'StartHesitation': 'S',\n            'Turn': 'T',\n            'Walking': 'W',\n            'None': 'N',\n        }\n        y_col = fdf[['StartHesitation', 'Turn', 'Walking', 'None']].idxmax(axis=1).map(mapping)\n        fdf['FOG_Events'] = pd.DataFrame({'FOG_Events': y_col}).astype(object)\n\n    # keep only desired columns\n    target_cols = ['FOG_Events', 'ValidationSet']\n    predictor_cols = ['Id', 'AccV', 'AccML', 'AccAP', 'Medication', 'Age', 'isMale', 'YearsSinceDx'] + [f'{col}_lag1' for col in lag_cols]\n    keep_cols = target_cols + predictor_cols if dtype_train else predictor_cols\n    fdf = fdf[keep_cols]\n    fdf = fdf.fillna(0)\n    return fdf","metadata":{"execution":{"iopub.status.busy":"2023-06-13T13:53:27.624845Z","iopub.execute_input":"2023-06-13T13:53:27.625458Z","iopub.status.idle":"2023-06-13T13:53:27.641398Z","shell.execute_reply.started":"2023-06-13T13:53:27.625425Z","shell.execute_reply":"2023-06-13T13:53:27.640184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Downsample","metadata":{}},{"cell_type":"code","source":"from sklearn.utils import resample\n\ndef Balance(fdf):\n    # Count the number of samples in each class\n    class_counts = fdf['FOG_Events'].value_counts()\n\n    # Find the smallest class\n    smallest_class = class_counts.index[-1]\n\n    # Downsample the other classes to match the size of the smallest class\n    balanced_fdf = pd.concat([\n        resample(fdf[fdf['FOG_Events'] == c], n_samples=class_counts[smallest_class], replace=False)\n        for c in class_counts.index if c != smallest_class\n    ])\n\n    # Add the smallest class back in\n    balanced_fdf = pd.concat([balanced_fdf,fdf[fdf['FOG_Events'] == smallest_class]])\n\n    # Shuffle the dataframe\n    balanced_fdf = balanced_fdf.sample(frac=1).reset_index(drop=True)\n    \n    return balanced_fdf","metadata":{"execution":{"iopub.status.busy":"2023-06-13T13:53:27.642753Z","iopub.execute_input":"2023-06-13T13:53:27.643188Z","iopub.status.idle":"2023-06-13T13:53:27.656296Z","shell.execute_reply.started":"2023-06-13T13:53:27.643158Z","shell.execute_reply":"2023-06-13T13:53:27.655273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Random Forest","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import GridSearchCV\n\ndef GetRandomForest(fdf):\n    scaler = StandardScaler()\n    rfc = RandomForestClassifier()\n    \n    # Split on the training set\n    X_train = fdf[PREDS_COLS]\n    y_train = fdf['FOG_Events']\n\n    # Scale the train data\n    X_train_scaled = scaler.fit_transform(X_train)\n\n    # Make a random tree classifier\n    param_grid = {\n        'n_estimators': [100],\n        'max_depth': [25]\n    }\n    grid_search = GridSearchCV(rfc, param_grid=param_grid, cv=5, scoring='precision_weighted')\n    grid_search.fit(X_train_scaled, y_train)\n    model_RF = grid_search.best_estimator_\n    return model_RF","metadata":{"execution":{"iopub.status.busy":"2023-06-13T13:53:27.659548Z","iopub.execute_input":"2023-06-13T13:53:27.660193Z","iopub.status.idle":"2023-06-13T13:53:27.803241Z","shell.execute_reply.started":"2023-06-13T13:53:27.660157Z","shell.execute_reply":"2023-06-13T13:53:27.801997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Matrices","metadata":{}},{"cell_type":"code","source":"def TrainValidationSplit(fdf):\n    t = fdf[fdf['ValidationSet']==1].reset_index(drop=True)\n    v = fdf[fdf['ValidationSet']==0].reset_index(drop=True)\n    return t, v","metadata":{"execution":{"iopub.status.busy":"2023-06-13T13:53:27.805364Z","iopub.execute_input":"2023-06-13T13:53:27.806301Z","iopub.status.idle":"2023-06-13T13:53:27.813447Z","shell.execute_reply.started":"2023-06-13T13:53:27.806253Z","shell.execute_reply":"2023-06-13T13:53:27.812053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def GetTransitionMatrix(fdf):\n    TM = pd.crosstab(fdf['FOG_Events'], fdf['FOG_Events'].shift(-1), normalize='index')\n    return TM.fillna(0)\n\ndef GetEmissionsMatrix(fdf, RF_model, round_num=1):\n    le = LabelEncoder()\n    RF_preds = RF_model.predict(fdf[PREDS_COLS].values)\n    RF_obs = le.fit_transform(RF_preds)\n    e_mat = pd.crosstab(fdf['FOG_Events'], RF_obs)\n    return e_mat","metadata":{"execution":{"iopub.status.busy":"2023-06-13T13:53:27.815544Z","iopub.execute_input":"2023-06-13T13:53:27.816030Z","iopub.status.idle":"2023-06-13T13:53:27.826295Z","shell.execute_reply.started":"2023-06-13T13:53:27.815991Z","shell.execute_reply":"2023-06-13T13:53:27.825140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## HMM","metadata":{}},{"cell_type":"code","source":"def GetHiddenMarkov(fdf, RF_model):\n\n    trans_mat = GetTransitionMatrix(fdf)\n    e_mat = GetEmissionsMatrix(fdf, RF_model)\n    e_mat = np.round((np.array(e_mat) / np.sum(np.array(e_mat),axis=0))).T\n    pi = [1,0,0,0]\n\n    # Fitting the HMM Model\n    model_HMM = hmm.CategoricalHMM(n_components=4)\n\n    # Setting the model parameters\n    model_HMM.transmat_ = np.array(trans_mat)\n    model_HMM.emissionprob_ = e_mat\n    model_HMM.startprob_ = np.array(pi)\n\n    return model_HMM","metadata":{"execution":{"iopub.status.busy":"2023-06-13T13:53:27.828318Z","iopub.execute_input":"2023-06-13T13:53:27.828777Z","iopub.status.idle":"2023-06-13T13:53:27.843330Z","shell.execute_reply.started":"2023-06-13T13:53:27.828738Z","shell.execute_reply":"2023-06-13T13:53:27.842165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ConvertIntegerToClass(fdf, hmm_preds, col):\n    fdf[col] = np.where(hmm_preds == 1, \"N\",\n                        np.where(hmm_preds == 2, \"S\",\n                            np.where(hmm_preds == 3, \"T\",\"W\"))).copy()\n\n\ndef GetHMMPredictions(fdf, HMM_model, RF_model):\n    \n    # If this dataset has more than one datapoint in it\n    if fdf.shape[0] > 0:\n        le = LabelEncoder()\n\n        # Appending the RF observations of the datasets\n        RF_obs = RF_model.predict(fdf[PREDS_COLS].values)\n        fdf['RF_Obs'] = le.fit_transform(RF_obs).copy()\n\n        hmm_preds = HMM_model.decode(fdf['RF_Obs'].values.reshape(-1,1)) #Outputs 0,1,2,or 3\n        ConvertIntegerToClass(fdf, hmm_preds, 'HMM')","metadata":{"execution":{"iopub.status.busy":"2023-06-13T13:53:27.845023Z","iopub.execute_input":"2023-06-13T13:53:27.845712Z","iopub.status.idle":"2023-06-13T13:53:27.855194Z","shell.execute_reply.started":"2023-06-13T13:53:27.845673Z","shell.execute_reply":"2023-06-13T13:53:27.854006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Clean, Train, and Submit","metadata":{}},{"cell_type":"markdown","source":"## 1. Create Class Object","metadata":{}},{"cell_type":"code","source":"# Create a Freezing of Gait Rectangular Dataset Builder class object\nfrobj = fr.FOGRect()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T13:53:27.856923Z","iopub.execute_input":"2023-06-13T13:53:27.857643Z","iopub.status.idle":"2023-06-13T13:54:40.412000Z","shell.execute_reply.started":"2023-06-13T13:53:27.857601Z","shell.execute_reply":"2023-06-13T13:54:40.410400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Get The Datasets","metadata":{}},{"cell_type":"code","source":"# Obtain the training through downloading the datasets\ntrain_defog_df = pd.concat([chunk for chunk in tqdm(pd.read_csv(\"/kaggle/input/fog-train-defog/train_defog.csv\", low_memory=False, chunksize=1000), desc='Downloading Train Defog', total=13526)])\ntrain_tcds_df = pd.concat([chunk for chunk in tqdm(pd.read_csv(\"/kaggle/input/fog-train-tcds/train_tcds.csv\", low_memory=False, chunksize=1000), desc='Downloading Train TCDS', total=7102)])\n\n# Obtain the testing through the rectangular dataset builder\ntest_defog_df = frobj.Get_DEFOG_Rectangular_Dataset(frobj.test_defog_csv_list)\ntest_tcds_df = frobj.Get_TCDS_Rectangular_Dataset(frobj.test_tcds_csv_list)","metadata":{"execution":{"iopub.status.busy":"2023-06-13T13:54:40.413759Z","iopub.execute_input":"2023-06-13T13:54:40.414149Z","iopub.status.idle":"2023-06-13T13:57:54.000711Z","shell.execute_reply.started":"2023-06-13T13:54:40.414117Z","shell.execute_reply":"2023-06-13T13:57:53.999585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Preprocess & Concatenate Both Datasets","metadata":{}},{"cell_type":"markdown","source":"Process Training Dataset","metadata":{}},{"cell_type":"code","source":"train_defog_df = Clean_Dataset(train_defog_df)\ntrain_tcds_df = Clean_Dataset(train_tcds_df)","metadata":{"execution":{"iopub.status.busy":"2023-06-13T13:57:54.002190Z","iopub.execute_input":"2023-06-13T13:57:54.002599Z","iopub.status.idle":"2023-06-13T13:59:53.575045Z","shell.execute_reply.started":"2023-06-13T13:57:54.002568Z","shell.execute_reply":"2023-06-13T13:59:53.573473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.concat([train_defog_df, train_tcds_df])\n# train.to_csv(\"train.csv\", index=False)\ndel train_defog_df, train_tcds_df","metadata":{"execution":{"iopub.status.busy":"2023-06-13T13:59:53.577144Z","iopub.execute_input":"2023-06-13T13:59:53.577513Z","iopub.status.idle":"2023-06-13T13:59:55.318906Z","shell.execute_reply.started":"2023-06-13T13:59:53.577479Z","shell.execute_reply":"2023-06-13T13:59:55.318032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Process Testing Dataset","metadata":{}},{"cell_type":"code","source":"test_defog_df = Clean_Dataset(test_defog_df, dtype_train=False)\ntest_tcds_df = Clean_Dataset(test_tcds_df, dtype_train=False)","metadata":{"execution":{"iopub.status.busy":"2023-06-13T13:59:55.324223Z","iopub.execute_input":"2023-06-13T13:59:55.324775Z","iopub.status.idle":"2023-06-13T13:59:55.891708Z","shell.execute_reply.started":"2023-06-13T13:59:55.324744Z","shell.execute_reply":"2023-06-13T13:59:55.890534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.concat([test_defog_df, test_tcds_df])\n# test.to_csv(\"test.csv\", index=False)\ndel test_defog_df, test_tcds_df","metadata":{"execution":{"iopub.status.busy":"2023-06-13T13:59:55.893097Z","iopub.execute_input":"2023-06-13T13:59:55.893477Z","iopub.status.idle":"2023-06-13T13:59:55.926577Z","shell.execute_reply.started":"2023-06-13T13:59:55.893446Z","shell.execute_reply":"2023-06-13T13:59:55.925563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Display both","metadata":{}},{"cell_type":"code","source":"display(train.head(2))\ndisplay(test.head(2))","metadata":{"execution":{"iopub.status.busy":"2023-06-13T13:59:55.928030Z","iopub.execute_input":"2023-06-13T13:59:55.928503Z","iopub.status.idle":"2023-06-13T13:59:55.973490Z","shell.execute_reply.started":"2023-06-13T13:59:55.928472Z","shell.execute_reply":"2023-06-13T13:59:55.972419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Split by 'Medication'","metadata":{}},{"cell_type":"code","source":"train['Medication'].value_counts() / len(train) * 100","metadata":{"execution":{"iopub.status.busy":"2023-06-13T13:59:55.974975Z","iopub.execute_input":"2023-06-13T13:59:55.975428Z","iopub.status.idle":"2023-06-13T13:59:56.177507Z","shell.execute_reply.started":"2023-06-13T13:59:55.975389Z","shell.execute_reply":"2023-06-13T13:59:56.176302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"onMed_train = train[train['Medication']==1]\noffMed_train = train[train['Medication']==0]\n\nonMed_test = test[test['Medication']==1]\noffMed_test = test[test['Medication']==0]","metadata":{"execution":{"iopub.status.busy":"2023-06-13T13:59:56.179608Z","iopub.execute_input":"2023-06-13T13:59:56.179975Z","iopub.status.idle":"2023-06-13T14:00:00.945938Z","shell.execute_reply.started":"2023-06-13T13:59:56.179946Z","shell.execute_reply":"2023-06-13T14:00:00.944920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Clean Up To Save RAM\n\nDoing so clears up space for only the `train` and `test` dataframes.","metadata":{}},{"cell_type":"code","source":"# Delete original rectangular datasets\ndel train","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:00:00.947230Z","iopub.execute_input":"2023-06-13T14:00:00.947576Z","iopub.status.idle":"2023-06-13T14:00:01.198824Z","shell.execute_reply.started":"2023-06-13T14:00:00.947548Z","shell.execute_reply":"2023-06-13T14:00:01.197578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6. Down Sample","metadata":{}},{"cell_type":"code","source":"onMed_balanced_train = Balance(onMed_train)\noffMed_balanced_train = Balance(offMed_train)","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:00:01.200517Z","iopub.execute_input":"2023-06-13T14:00:01.200848Z","iopub.status.idle":"2023-06-13T14:00:21.503714Z","shell.execute_reply.started":"2023-06-13T14:00:01.200822Z","shell.execute_reply":"2023-06-13T14:00:21.502732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"onMed_balanced_train['FOG_Events'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:00:21.504906Z","iopub.execute_input":"2023-06-13T14:00:21.505230Z","iopub.status.idle":"2023-06-13T14:00:21.577330Z","shell.execute_reply.started":"2023-06-13T14:00:21.505201Z","shell.execute_reply":"2023-06-13T14:00:21.576131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"offMed_balanced_train['FOG_Events'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:00:21.578860Z","iopub.execute_input":"2023-06-13T14:00:21.579204Z","iopub.status.idle":"2023-06-13T14:00:21.672268Z","shell.execute_reply.started":"2023-06-13T14:00:21.579172Z","shell.execute_reply":"2023-06-13T14:00:21.671108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 7. Random Forest","metadata":{}},{"cell_type":"code","source":"import joblib\n\n# onMed_RF_model = GetRandomForest(onMed_balanced_train)\n# offMed_RF_model = GetRandomForest(offMed_balanced_train)\noffMed_RF_model = joblib.load('/kaggle/input/random-forest-modelpkl/offMed_RF_model.pkl')\nonMed_RF_model = joblib.load('/kaggle/input/random-forest-modelpkl/onMed_RF_model (1).pkl')","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:00:21.673892Z","iopub.execute_input":"2023-06-13T14:00:21.674373Z","iopub.status.idle":"2023-06-13T14:00:44.507619Z","shell.execute_reply.started":"2023-06-13T14:00:21.674309Z","shell.execute_reply":"2023-06-13T14:00:44.506361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 8. HMM","metadata":{}},{"cell_type":"code","source":"# Train the HMM Model\nonMed_model_HMM = GetHiddenMarkov(onMed_train, onMed_RF_model)\noffMed_model_HMM = GetHiddenMarkov(offMed_train, offMed_RF_model)","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:00:44.512277Z","iopub.execute_input":"2023-06-13T14:00:44.512714Z","iopub.status.idle":"2023-06-13T14:05:02.234680Z","shell.execute_reply.started":"2023-06-13T14:00:44.512684Z","shell.execute_reply":"2023-06-13T14:05:02.233385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## onMed Emissions Matrix Discovery\n\nI discovered that the onMed Emission Matrix is always assuming \"N\" or \"No FOG Event\" so I went with it. Since I can't run an HMM model with this emissions matrix (it's missing three prediction classes) I just said \"if the client is on medication, then we're going to assume no FOG event.\"","metadata":{}},{"cell_type":"code","source":"le = LabelEncoder()\nonMed_RF_preds = onMed_RF_model.predict(onMed_train[PREDS_COLS].values)\nonMed_RF_obs = le.fit_transform(onMed_RF_preds)\nonMed_e_mat = pd.crosstab(onMed_train['FOG_Events'], onMed_RF_obs)\nonMed_e_mat","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:05:02.236111Z","iopub.execute_input":"2023-06-13T14:05:02.236742Z","iopub.status.idle":"2023-06-13T14:06:46.562671Z","shell.execute_reply.started":"2023-06-13T14:05:02.236704Z","shell.execute_reply":"2023-06-13T14:06:46.561437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"le = LabelEncoder()\noffMed_RF_preds = offMed_RF_model.predict(offMed_train[PREDS_COLS].values)\noffMed_RF_obs = le.fit_transform(offMed_RF_preds)\noffMed_e_mat = pd.crosstab(offMed_train['FOG_Events'], offMed_RF_obs)\noffMed_e_mat","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:06:46.564126Z","iopub.execute_input":"2023-06-13T14:06:46.564521Z","iopub.status.idle":"2023-06-13T14:09:21.065818Z","shell.execute_reply.started":"2023-06-13T14:06:46.564489Z","shell.execute_reply":"2023-06-13T14:09:21.064495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 10. Test Dataset","metadata":{}},{"cell_type":"code","source":"# Get predictions for the offMed Test\nGetHMMPredictions(offMed_test, offMed_model_HMM, offMed_RF_model)","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:09:21.067491Z","iopub.execute_input":"2023-06-13T14:09:21.067818Z","iopub.status.idle":"2023-06-13T14:09:24.358233Z","shell.execute_reply.started":"2023-06-13T14:09:21.067791Z","shell.execute_reply":"2023-06-13T14:09:24.356753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"offMed_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:09:24.359759Z","iopub.execute_input":"2023-06-13T14:09:24.360277Z","iopub.status.idle":"2023-06-13T14:09:24.380954Z","shell.execute_reply.started":"2023-06-13T14:09:24.360239Z","shell.execute_reply":"2023-06-13T14:09:24.379750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"offMed_test[['StartHesitation', 'Turn', 'Walking']] = offMed_test['HMM'].apply(lambda x: pd.Series({'StartHesitation': 1 if x == 'S' else 0, 'Turn': 1 if x == 'T' else 0, 'Walking': 1 if x == 'W' else 0}))\noffMed_test = offMed_test[['Id', 'StartHesitation', 'Turn', 'Walking']]","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:09:24.382517Z","iopub.execute_input":"2023-06-13T14:09:24.382871Z","iopub.status.idle":"2023-06-13T14:11:10.345802Z","shell.execute_reply.started":"2023-06-13T14:09:24.382842Z","shell.execute_reply":"2023-06-13T14:11:10.344590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/sample_submission.csv\").head()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:11:10.347576Z","iopub.execute_input":"2023-06-13T14:11:10.347909Z","iopub.status.idle":"2023-06-13T14:11:10.629285Z","shell.execute_reply.started":"2023-06-13T14:11:10.347882Z","shell.execute_reply":"2023-06-13T14:11:10.627964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since we can't use the HMM model above for onMed (since the emissions matrix is the wrong size) we will instead just assume that if the client is on medication that they are not experiencing a FOG event.","metadata":{}},{"cell_type":"code","source":"onMed_test[['StartHesitation', 'Turn', 'Walking']] = 0\nonMed_test = onMed_test[['Id', 'StartHesitation', 'Turn', 'Walking']]","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:11:10.630845Z","iopub.execute_input":"2023-06-13T14:11:10.631330Z","iopub.status.idle":"2023-06-13T14:11:10.641189Z","shell.execute_reply.started":"2023-06-13T14:11:10.631290Z","shell.execute_reply":"2023-06-13T14:11:10.639947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.concat([offMed_test, onMed_test])\n\n# get the index order of test data\nindex_order = test.index\n\n# reindex submission using the index order of test\nsubmission = submission.reindex(index_order)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:11:10.642798Z","iopub.execute_input":"2023-06-13T14:11:10.643545Z","iopub.status.idle":"2023-06-13T14:11:10.675433Z","shell.execute_reply.started":"2023-06-13T14:11:10.643513Z","shell.execute_reply":"2023-06-13T14:11:10.674199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6. Submit Predictions","metadata":{}},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:11:10.676764Z","iopub.execute_input":"2023-06-13T14:11:10.677397Z","iopub.status.idle":"2023-06-13T14:11:11.626098Z","shell.execute_reply.started":"2023-06-13T14:11:10.677365Z","shell.execute_reply":"2023-06-13T14:11:11.624942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:11:12.335554Z","iopub.execute_input":"2023-06-13T14:11:12.335905Z","iopub.status.idle":"2023-06-13T14:11:12.344132Z","shell.execute_reply.started":"2023-06-13T14:11:12.335876Z","shell.execute_reply":"2023-06-13T14:11:12.342995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:11:14.004758Z","iopub.execute_input":"2023-06-13T14:11:14.005555Z","iopub.status.idle":"2023-06-13T14:11:14.013324Z","shell.execute_reply.started":"2023-06-13T14:11:14.005507Z","shell.execute_reply":"2023-06-13T14:11:14.012122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[['Id']].head()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T14:12:56.972305Z","iopub.execute_input":"2023-06-13T14:12:56.972786Z","iopub.status.idle":"2023-06-13T14:12:56.991495Z","shell.execute_reply.started":"2023-06-13T14:12:56.972754Z","shell.execute_reply":"2023-06-13T14:12:56.990047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}