{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":41880,"databundleVersionId":5677426,"sourceType":"competition"}],"dockerImageVersionId":30626,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Data Science for Neuroscience - Final Project - Group 3\n> ## Parkinson - FOG Classification\n> ### Approach: Random Forest ","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os\nimport seaborn as sns\nimport warnings\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nfrom sklearn.metrics import accuracy_score, classification_report\n\nimport lightgbm as lgb\n\n!pip install xgboost\nimport xgboost as xgb\n\nwarnings.filterwarnings(action = \"ignore\", category = DeprecationWarning ) \nwarnings.filterwarnings(action = \"ignore\", category = FutureWarning ) ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-04T10:41:26.011633Z","iopub.execute_input":"2024-03-04T10:41:26.012068Z","iopub.status.idle":"2024-03-04T10:42:06.656015Z","shell.execute_reply.started":"2024-03-04T10:41:26.012028Z","shell.execute_reply":"2024-03-04T10:42:06.654983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. Data Exploration\n> ### Load Data\n> (Further Data Exploration is in the other code from the LSTM group)","metadata":{}},{"cell_type":"code","source":"def load_training_data(train_path):\n    \"\"\"\n    Loads training data from CSV files located in the specified directory.\n\n    Parameters:\n    - train_path (str): Path to the directory containing CSV files.\n\n    Returns:\n    - train_data (DataFrame): Concatenated DataFrame containing training data from all CSV files.\n    \"\"\"\n    train_list = []\n\n    file_paths = [os.path.join(train_path, file_name) for file_name in os.listdir(train_path) if file_name.endswith('.csv')]\n\n    train_list = [pd.read_csv(file_path) for file_path in file_paths]\n\n    train_data = pd.concat(train_list, axis=0)\n    \n    return train_data","metadata":{"execution":{"iopub.status.busy":"2024-03-04T10:42:06.658270Z","iopub.execute_input":"2024-03-04T10:42:06.658564Z","iopub.status.idle":"2024-03-04T10:42:06.664932Z","shell.execute_reply.started":"2024-03-04T10:42:06.658536Z","shell.execute_reply":"2024-03-04T10:42:06.663844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load tdcsfog data \ntdcsfog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog'\n\ntdcsfog = load_training_data(tdcsfog_path)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T10:42:06.666916Z","iopub.execute_input":"2024-03-04T10:42:06.667227Z","iopub.status.idle":"2024-03-04T10:42:27.249218Z","shell.execute_reply.started":"2024-03-04T10:42:06.667188Z","shell.execute_reply":"2024-03-04T10:42:27.247949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load defog data\ndefog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog'\n\ndefog = load_training_data(defog_path)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T10:42:27.250832Z","iopub.execute_input":"2024-03-04T10:42:27.251197Z","iopub.status.idle":"2024-03-04T10:42:53.221917Z","shell.execute_reply.started":"2024-03-04T10:42:27.251145Z","shell.execute_reply":"2024-03-04T10:42:53.221135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# exclude not usable rows\ndefog = defog[(defog['Task'] == 1) & (defog['Valid'] == 1)]\ndefog = defog.iloc[:, :7]","metadata":{"execution":{"iopub.status.busy":"2024-03-04T10:42:53.224905Z","iopub.execute_input":"2024-03-04T10:42:53.225287Z","iopub.status.idle":"2024-03-04T10:42:53.462786Z","shell.execute_reply.started":"2024-03-04T10:42:53.225246Z","shell.execute_reply":"2024-03-04T10:42:53.461960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# combine data \nmerged = pd.concat([tdcsfog, defog], axis = 0)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T10:42:53.463941Z","iopub.execute_input":"2024-03-04T10:42:53.464277Z","iopub.status.idle":"2024-03-04T10:42:53.687962Z","shell.execute_reply.started":"2024-03-04T10:42:53.464251Z","shell.execute_reply":"2024-03-04T10:42:53.687215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. First Model - Fog Identification\n> In order to detect fog events, such as start hesitation, turning or walking, within the dataset, we first construct a model that is designed to detect the occurrence of any fog event. This task is essentially binary, where the model is trained to detect fog events over nonfog events (indicated by a value of 1 in either start hesitation, turning or walking) using a random forest algorithm, specifically LightGBM in our case.","metadata":{}},{"cell_type":"markdown","source":"### a. Data preparation\n- adding column 'IsFOG' in the data set (fog event == 1, no fog event == 0)\n- x and y values for model: x = AccV, AccML and AccAP, y = IsFOG\n- split data in train and test ","metadata":{}},{"cell_type":"code","source":"# add column for binary classification: Fog event or nonfog \nmerged['IsFOG'] = merged[['StartHesitation', 'Walking', 'Turn']].any(axis='columns')","metadata":{"execution":{"iopub.status.busy":"2024-03-04T10:42:53.689075Z","iopub.execute_input":"2024-03-04T10:42:53.689442Z","iopub.status.idle":"2024-03-04T10:42:53.833774Z","shell.execute_reply.started":"2024-03-04T10:42:53.689409Z","shell.execute_reply":"2024-03-04T10:42:53.832983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define x and y values for model \nx = merged[['AccV','AccML','AccAP']]\ny = merged['IsFOG']","metadata":{"execution":{"iopub.status.busy":"2024-03-04T10:42:53.834788Z","iopub.execute_input":"2024-03-04T10:42:53.835035Z","iopub.status.idle":"2024-03-04T10:42:53.926444Z","shell.execute_reply.started":"2024-03-04T10:42:53.835014Z","shell.execute_reply":"2024-03-04T10:42:53.925619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, Y_train, Y_test = train_test_split(x, y, test_size = 0.1, random_state = 1)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T10:42:53.927668Z","iopub.execute_input":"2024-03-04T10:42:53.928013Z","iopub.status.idle":"2024-03-04T10:42:55.612832Z","shell.execute_reply.started":"2024-03-04T10:42:53.927983Z","shell.execute_reply":"2024-03-04T10:42:55.611868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### b. Model Building and Training","metadata":{}},{"cell_type":"code","source":"# Split the data into training and validation sets\nx_train, x_val, y_train, y_val = train_test_split(X_train, Y_train, test_size=0.1, random_state=2)\n\n# Create LightGBM Datasets for training and validation\ntrain_data = lgb.Dataset(x_train, label=y_train)\ntest_data = lgb.Dataset(x_val, label=y_val, reference=train_data)\n\n# Define hyperparameters and objective for LightGBM\nfog_params = {\n    'objective': 'binary',\n    'metric': 'auc', \n    'boosting_type': 'gbdt', \n    'num_leaves': 150, \n    'learning_rate': 0.1, \n    'max_depth': 15,\n    'feature_fraction': 0.9, \n} \n\n# Train the LightGBM model\nnum_round = 200  \n\n# Train the model\nfog_model = lgb.train(fog_params, train_data, num_round, valid_sets=[test_data])\n\n# Make predictions\ny_pred = fog_model.predict(x_val, num_iteration=fog_model.best_iteration)\n\n# Convert probabilities to binary predictions\ny_pred_binary = (y_pred > 0.5).astype(int)\n\n# Evaluate the model\naccuracy = metrics.accuracy_score(y_val, y_pred_binary)\nprint(f\"Accuracy: {accuracy}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-03-04T10:42:55.614102Z","iopub.execute_input":"2024-03-04T10:42:55.614486Z","iopub.status.idle":"2024-03-04T10:44:24.161174Z","shell.execute_reply.started":"2024-03-04T10:42:55.614455Z","shell.execute_reply":"2024-03-04T10:44:24.160233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ### Our model achieves an accuracy of 79,1% detecting fog events over nonfog events in tdcsfog and defog data combined.","metadata":{}},{"cell_type":"markdown","source":"## 3. Feature Engineering\n> Since the acceleration data is the most the only data we train our models so far, we added some additional features based on the acceleration data. As we are working with time series data, statistics can be calculated based on a sliding window that moves along the data range. \n\n#### The features are: \n- cum_sum: The cumulative sum of Acc features\n- rolling_sum\n- rolling_min\n- rolling_max\n- rolling_mean\n- rolling_std\n- rolling_delta: The difference between the rolling maximum and the rolling minimum\n- acc_lead_diff: The time difference between the current data point and the previous data point\n- acc_lag_diff:The time difference between the current data point and the next data point\n\n\n\nThis code is inspired by https://www.kaggle.com/code/hadisraad/towards-parkinson-s-freezing-of-gait-prediction.\n\n","metadata":{}},{"cell_type":"code","source":"def process_data(data):\n    \"\"\"\n    Processes the given data by calculating additional features related to acceleration.\n\n    Parameters:\n    - data (DataFrame): Input DataFrame containing acceleration data.\n\n    Returns:\n    - processed_data (DataFrame): DataFrame with additional features calculated for each acceleration feature.\n    \"\"\"\n    try:\n        acc_columns = [col for col in data.columns if 'Acc' in col]\n        for acc in acc_columns:\n            data[f'{acc}_cumsum'] = data[acc].cumsum()\n            data[f'{acc}_rolling_sum'] = data[acc].rolling(window=len(data), min_periods=1).sum()\n            data[f'{acc}_rolling_min'] = data[acc].rolling(window=len(data), min_periods=1).min()\n            data[f'{acc}_rolling_max'] = data[acc].rolling(window=len(data), min_periods=1).max()\n            data[f'{acc}_rolling_mean'] = data[acc].rolling(window=len(data), min_periods=1).mean()\n            data[f'{acc}_rolling_std'] = data[acc].rolling(window=len(data), min_periods=1).std()\n            data[f'{acc}_rolling_delta'] = data[f'{acc}_rolling_max'] - data[f'{acc}_rolling_min']\n            data[f'{acc}_EWMA_02'] = data[acc].ewm(alpha=0.2).mean()\n            data[f\"{acc}_lead_diff\"] = data[acc].diff(-1)\n            data[f\"{acc}_lag_diff\"] = data[acc].diff()\n\n        data.fillna(method='backfill', inplace=True)\n        \n        return data\n    except Exception as e:\n        print(f\"An error occurred: {e}\")\n        return None\n","metadata":{"execution":{"iopub.status.busy":"2024-03-04T11:44:43.850095Z","iopub.execute_input":"2024-03-04T11:44:43.850531Z","iopub.status.idle":"2024-03-04T11:44:43.861791Z","shell.execute_reply.started":"2024-03-04T11:44:43.850501Z","shell.execute_reply":"2024-03-04T11:44:43.860724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged_features = process_data(merged)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T10:45:33.198153Z","iopub.execute_input":"2024-03-04T10:45:33.198556Z","iopub.status.idle":"2024-03-04T10:45:41.146084Z","shell.execute_reply.started":"2024-03-04T10:45:33.198527Z","shell.execute_reply":"2024-03-04T10:45:41.145222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged_features","metadata":{"execution":{"iopub.status.busy":"2024-03-04T10:45:43.961970Z","iopub.execute_input":"2024-03-04T10:45:43.962348Z","iopub.status.idle":"2024-03-04T10:45:49.126815Z","shell.execute_reply.started":"2024-03-04T10:45:43.962319Z","shell.execute_reply":"2024-03-04T10:45:49.125857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Reduce memory storage\n> Since the data set is relatively large, a function is used to reduce memory usage. ","metadata":{}},{"cell_type":"code","source":"# Create a function to reduce memory usage\ndef reduce_mem_usage(df):\n    \"\"\"\n    Reduces memory usage of a DataFrame by optimizing data types.\n\n    Parameters:\n    - df (DataFrame): Input DataFrame to optimize memory usage.\n\n    Returns:\n    - df (DataFrame): DataFrame with optimized memory usage.\n    \"\"\"\n\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n\n    for col in df.columns:\n        col_type = df[col].dtype.name\n\n        if col_type not in ['object', 'category', 'datetime64[ns, UTC]']:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n   \n    return df ","metadata":{"execution":{"iopub.status.busy":"2024-03-04T11:29:43.863776Z","iopub.execute_input":"2024-03-04T11:29:43.864156Z","iopub.status.idle":"2024-03-04T11:29:43.876732Z","shell.execute_reply.started":"2024-03-04T11:29:43.864128Z","shell.execute_reply":"2024-03-04T11:29:43.875749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged_features = reduce_mem_usage(merged_features)","metadata":{"execution":{"iopub.status.busy":"2024-02-29T14:43:54.909318Z","iopub.execute_input":"2024-02-29T14:43:54.910084Z","iopub.status.idle":"2024-02-29T14:44:10.368354Z","shell.execute_reply.started":"2024-02-29T14:43:54.910053Z","shell.execute_reply":"2024-02-29T14:44:10.367350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Second Model - FOG Classification\n> Now, we build another model, which classifies the three different fog events. This time a XGBoost algorithm is used.","metadata":{}},{"cell_type":"markdown","source":"### a. Data Preparation\n- split data into the three different fog events: start hesitation, turn and walking\n- balance data \n- split data in train and test sets","metadata":{}},{"cell_type":"code","source":"targets = [\"StartHesitation\", \"Turn\", 'Walking', 'IsFOG']","metadata":{"execution":{"iopub.status.busy":"2024-03-04T11:17:23.654434Z","iopub.execute_input":"2024-03-04T11:17:23.654768Z","iopub.status.idle":"2024-03-04T11:17:23.659046Z","shell.execute_reply.started":"2024-03-04T11:17:23.654743Z","shell.execute_reply":"2024-03-04T11:17:23.658202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X values for on which the model is trained\nX_merged = merged_features.drop(targets, axis=1)  # features\n\n# split data into the three \ny1 = merged_features['StartHesitation']\ny2 = merged_features['Turn']\ny3 = merged_features['Walking']","metadata":{"execution":{"iopub.status.busy":"2024-03-04T11:22:26.571134Z","iopub.execute_input":"2024-03-04T11:22:26.571519Z","iopub.status.idle":"2024-03-04T11:22:26.956634Z","shell.execute_reply.started":"2024-03-04T11:22:26.571488Z","shell.execute_reply":"2024-03-04T11:22:26.955564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def balance_labels(X, y):\n    \"\"\"\n    Balances the labels in the target variable by randomly sampling the same number of samples \n    for each class where y == 1.\n\n    Parameters:\n    - X (DataFrame): Input feature matrix.\n    - y (Series): Target variable.\n\n    Returns:\n    - X_balanced (DataFrame): Feature matrix with balanced labels.\n    - y_balanced (Series): Balanced target variable.\n    \"\"\"\n    \n    # Find the positions of y where it equals 0 and 1.\n    y_zeros = np.where(y == 0)[0]\n    y_ones = np.where(y == 1)[0]\n\n    # Choose the same number of samples with y == 1 as there are with y == 0.\n    num_ones = (y == 1).sum()\n    np.random.seed(42)\n    y_zeros_balanced = np.random.choice(y_zeros, size=num_ones, replace = True)\n\n    # Combine the positions of y == 0 and y == 1.\n    y_balanced_idxs = np.sort(np.concatenate([y_zeros_balanced, y_ones]))\n\n    # Use the balanced indices to get the corresponding rows of X and y.\n    X_balanced = X.iloc[y_balanced_idxs]\n    y_balanced = y.iloc[y_balanced_idxs]\n    \n    return X_balanced, y_balanced\n\n# Balance the labels for each dataset\nX1_balanced, y1_balanced = balance_labels(X_merged, y1)\nX2_balanced, y2_balanced = balance_labels(X_merged, y2)\nX3_balanced, y3_balanced = balance_labels(X_merged, y3)\n\n# train test split\nX1_train, X1_test, y1_train, y1_test = train_test_split(X1_balanced, y1_balanced, test_size = 0.1, random_state = 42)\nX2_train, X2_test, y2_train, y2_test = train_test_split(X2_balanced, y2_balanced, test_size = 0.1, random_state = 42)\nX3_train, X3_test, y3_train, y3_test = train_test_split(X3_balanced, y3_balanced, test_size = 0.1, random_state = 42)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T11:22:27.243288Z","iopub.execute_input":"2024-03-04T11:22:27.243600Z","iopub.status.idle":"2024-03-04T11:22:32.198862Z","shell.execute_reply.started":"2024-03-04T11:22:27.243574Z","shell.execute_reply":"2024-03-04T11:22:32.197744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### b. Model Building and Training \n> The parameters are a mix of try and error and grid search (can be found in previous versions).","metadata":{}},{"cell_type":"code","source":"def train_xgboost_model(X_train, y_train, X_test, y_test, xgboost_params, num_boost_round):\n    \"\"\"\n    Trains an XGBoost model on the given training data and evaluates it on the test data.\n\n    Parameters:\n    - X_train (DataFrame): Training feature matrix.\n    - y_train (Series): Training target variable.\n    - X_test (DataFrame): Test feature matrix.\n    - y_test (Series): Test target variable.\n    - xgboost_params (dict): Parameters for XGBoost model training.\n    - num_boost_round (int): Number of boosting rounds.\n\n    Returns:\n    - model (XGBModel): Trained XGBoost model.\n    - accuracy (float): Accuracy of the trained model on the test data.\n    \"\"\"\n    \n    # Convert data to DMatrix format\n    dtrain = xgb.DMatrix(X_train, label=y_train)\n    dtest = xgb.DMatrix(X_test, label=y_test)\n\n    # Train the XGBoost model\n    evals = [(dtest, 'eval')]\n    model = xgb.train(params=xgboost_params, dtrain=dtrain, num_boost_round=num_boost_round, evals=evals)\n    \n    # Predict on the test set\n    y_pred = model.predict(dtest)\n\n    # Calculate accuracy\n    accuracy = accuracy_score(y_test, y_pred.round())  # For binary classification\n\n    return model, accuracy\n\n# Define XGBoost parameters\nxgboost_params = {\n    'objective': 'binary:logistic',  \n    'eval_metric': 'auc', \n    'colsample_bytree': 0.5282057895135501,\n    'learning_rate': 0.22659963168004743,\n    'max_depth': 15,\n    'min_child_weight': 3.1233911067827616,\n    'n_estimators': 291,\n    'subsample': 0.9961057796456088,\n    'num_leaves': 150,\n}","metadata":{"execution":{"iopub.status.busy":"2024-03-04T11:22:34.425123Z","iopub.execute_input":"2024-03-04T11:22:34.425530Z","iopub.status.idle":"2024-03-04T11:22:34.433416Z","shell.execute_reply.started":"2024-03-04T11:22:34.425501Z","shell.execute_reply":"2024-03-04T11:22:34.432391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_boost_round = 200\nmodel1, accuracy1 = train_xgboost_model(X1_train, y1_train, X1_test, y1_test, xgboost_params, num_boost_round)\nprint(\"Accuracy StartHesitation:\", accuracy1)\nmodel2, accuracy2 = train_xgboost_model(X2_train, y2_train, X2_test, y2_test, xgboost_params, num_boost_round)\nprint(\"Accuracy Turn:\", accuracy2)\nmodel3, accuracy3 = train_xgboost_model(X3_train, y3_train, X3_test, y3_test, xgboost_params, num_boost_round)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-04T11:22:37.509910Z","iopub.execute_input":"2024-03-04T11:22:37.510389Z","iopub.status.idle":"2024-03-04T11:26:46.408436Z","shell.execute_reply.started":"2024-03-04T11:22:37.510357Z","shell.execute_reply":"2024-03-04T11:26:46.407569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Test Model for Submission","metadata":{}},{"cell_type":"code","source":"tdcsfog_test_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog'\n        \ntdcsfog_test_list = [\n    pd.read_csv(os.path.join(tdcsfog_test_path, file_name)).assign(Id=lambda df: file_name[:-4] + '_' + df['Time'].astype(str))\n    for file_name in os.listdir(tdcsfog_test_path)\n    if file_name.endswith('.csv')\n]        \n\ntdcsfog_test = pd.concat(tdcsfog_test_list, axis = 0)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-03-04T11:29:01.490335Z","iopub.execute_input":"2024-03-04T11:29:01.491053Z","iopub.status.idle":"2024-03-04T11:29:01.529992Z","shell.execute_reply.started":"2024-03-04T11:29:01.491017Z","shell.execute_reply":"2024-03-04T11:29:01.529203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_test_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog'\n\ndefog_test_list = [\n    pd.read_csv(os.path.join(defog_test_path, file_name)).assign(Id=lambda df: file_name[:-4] + '_' + df['Time'].astype(str))\n    for file_name in os.listdir(defog_test_path)\n    if file_name.endswith('.csv')\n]        \n\ndefog_test = pd.concat(defog_test_list, axis = 0)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-03-04T11:29:01.766959Z","iopub.execute_input":"2024-03-04T11:29:01.767801Z","iopub.status.idle":"2024-03-04T11:29:02.440450Z","shell.execute_reply.started":"2024-03-04T11:29:01.767769Z","shell.execute_reply":"2024-03-04T11:29:02.439568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.concat([tdcsfog_test, defog_test], axis = 0).reset_index(drop = True)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T11:29:02.554020Z","iopub.execute_input":"2024-03-04T11:29:02.554765Z","iopub.status.idle":"2024-03-04T11:29:02.573028Z","shell.execute_reply.started":"2024-03-04T11:29:02.554731Z","shell.execute_reply":"2024-03-04T11:29:02.572139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### a. FOG Classification","metadata":{}},{"cell_type":"code","source":"X_test = test.iloc[:, 1:4]\ntest_pred_fog = fog_model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T11:29:09.382461Z","iopub.execute_input":"2024-03-04T11:29:09.383259Z","iopub.status.idle":"2024-03-04T11:29:10.484989Z","shell.execute_reply.started":"2024-03-04T11:29:09.383227Z","shell.execute_reply":"2024-03-04T11:29:10.484225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['FogProb'] = test_pred_fog\ntest","metadata":{"execution":{"iopub.status.busy":"2024-03-04T11:29:21.953058Z","iopub.execute_input":"2024-03-04T11:29:21.953946Z","iopub.status.idle":"2024-03-04T11:29:21.969620Z","shell.execute_reply.started":"2024-03-04T11:29:21.953909Z","shell.execute_reply":"2024-03-04T11:29:21.968590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_test = process_data(test)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T11:29:25.888228Z","iopub.execute_input":"2024-03-04T11:29:25.888602Z","iopub.status.idle":"2024-03-04T11:29:26.101366Z","shell.execute_reply.started":"2024-03-04T11:29:25.888572Z","shell.execute_reply":"2024-03-04T11:29:26.100549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_test = reduce_mem_usage(features_test)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T11:29:53.657931Z","iopub.execute_input":"2024-03-04T11:29:53.658315Z","iopub.status.idle":"2024-03-04T11:29:53.780382Z","shell.execute_reply.started":"2024-03-04T11:29:53.658285Z","shell.execute_reply":"2024-03-04T11:29:53.779327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_test","metadata":{"execution":{"iopub.status.busy":"2024-03-04T11:30:54.196597Z","iopub.execute_input":"2024-03-04T11:30:54.197280Z","iopub.status.idle":"2024-03-04T11:30:54.332353Z","shell.execute_reply.started":"2024-03-04T11:30:54.197233Z","shell.execute_reply":"2024-03-04T11:30:54.331436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = ['FogProb', 'Id']\n\nX_test = features_test.drop(targets, axis=1)\n\n# Convert test data to DMatrix format\ndtest = xgb.DMatrix(X_test)\n\n# Get the predictions for the three models on the test data.\nY1_pred = model1.predict(dtest)\nY2_pred = model2.predict(dtest)\nY3_pred = model3.predict(dtest)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T11:31:51.574545Z","iopub.execute_input":"2024-03-04T11:31:51.574903Z","iopub.status.idle":"2024-03-04T11:31:53.938090Z","shell.execute_reply.started":"2024-03-04T11:31:51.574876Z","shell.execute_reply":"2024-03-04T11:31:53.937301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_test['StartHesitation'] = np.sqrt(Y1_pred * test_pred_fog)\nfeatures_test['Turn'] = np.sqrt(Y2_pred * test_pred_fog)\nfeatures_test['Walking'] = np.sqrt(Y3_pred * test_pred_fog)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T11:32:01.869559Z","iopub.execute_input":"2024-03-04T11:32:01.869932Z","iopub.status.idle":"2024-03-04T11:32:01.880036Z","shell.execute_reply.started":"2024-03-04T11:32:01.869906Z","shell.execute_reply":"2024-03-04T11:32:01.879033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = features_test[['Id','StartHesitation','Turn','Walking']]\nsubmission = submission.fillna(0.0)\nsubmission.to_csv(\"submission.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T11:32:22.316010Z","iopub.execute_input":"2024-03-04T11:32:22.316549Z","iopub.status.idle":"2024-03-04T11:32:24.730199Z","shell.execute_reply.started":"2024-03-04T11:32:22.316513Z","shell.execute_reply":"2024-03-04T11:32:24.729368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2024-03-04T11:32:26.686672Z","iopub.execute_input":"2024-03-04T11:32:26.687461Z","iopub.status.idle":"2024-03-04T11:32:26.700836Z","shell.execute_reply.started":"2024-03-04T11:32:26.687425Z","shell.execute_reply":"2024-03-04T11:32:26.699878Z"},"trusted":true},"execution_count":null,"outputs":[]}]}