{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":41880,"databundleVersionId":5677426,"sourceType":"competition"}],"dockerImageVersionId":30746,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfrom scipy import signal\nfrom sklearn.model_selection import train_test_split\nimport lightgbm as lgb\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix\nimport lightgbm as lgb\nfrom sklearn.metrics import average_precision_score\n\n","metadata":{"execution":{"iopub.status.busy":"2024-08-28T13:59:18.648274Z","iopub.execute_input":"2024-08-28T13:59:18.648668Z","iopub.status.idle":"2024-08-28T13:59:22.361398Z","shell.execute_reply.started":"2024-08-28T13:59:18.648636Z","shell.execute_reply":"2024-08-28T13:59:22.360374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n\n\ntdcsfog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog'\n\ntdcsfog_list = []\n\n# iterate over each file in the directory\nfor file_name in os.listdir(tdcsfog_path):\n    # exclude this file because it is also in the test set\n    if file_name.endswith('.csv') and file_name != '003f117e14.csv': \n        file_path = os.path.join(tdcsfog_path, file_name)\n        df = pd.read_csv(file_path)\n        \n        # adding ID column (name of each recording)\n        df['Id'] = file_name[:-4]\n        tdcsfog_list.append(df)\n\ntdcsfog = pd.concat(tdcsfog_list, ignore_index=True)\n\n# adding \"IsFOG\" column - if there was any type of FOG or not\ntdcsfog['IsFOG'] = tdcsfog[['StartHesitation', 'Walking', 'Turn']].any(axis='columns')\nprint(tdcsfog)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-08-28T14:18:05.603246Z","iopub.execute_input":"2024-08-28T14:18:05.603956Z","iopub.status.idle":"2024-08-28T14:18:23.152072Z","shell.execute_reply.started":"2024-08-28T14:18:05.603918Z","shell.execute_reply":"2024-08-28T14:18:23.150881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Re-load the metadata file and count the unique number of subjects\nmetadata_tdcs = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv')\nunique_subjects = metadata_tdcs['Subject'].nunique()\n\n# Merge the tdcsfog dataframe with metadata based on 'ID'\ntdcsfog = tdcsfog.merge(metadata_tdcs[['Id', 'Subject']], on='Id', how='left')\n\n# Merge the tdcsfog dataframe with defog metadata based on 'ID'\ntdcsfog = tdcsfog.merge(metadata_tdcs[['Id', 'Medication']], on='Id', how='left')\n\nprint(unique_subjects)\nprint(tdcsfog['Subject'].nunique())\nprint(tdcsfog)","metadata":{"execution":{"iopub.status.busy":"2024-08-28T14:19:51.544954Z","iopub.execute_input":"2024-08-28T14:19:51.545428Z","iopub.status.idle":"2024-08-28T14:19:56.809662Z","shell.execute_reply.started":"2024-08-28T14:19:51.545392Z","shell.execute_reply":"2024-08-28T14:19:56.808418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading defog data same as tdcs data\ndefog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog'\ndefog_list = []\n\n# iterate over each file in the directory\nfor file_name in os.listdir(defog_path):\n    # exclude this file because it is also in the test set\n    if file_name.endswith('.csv') and file_name != '003f117e14.csv': \n        file_path = os.path.join(defog_path, file_name)\n        df = pd.read_csv(file_path)\n        \n        # add a new column with the file name without the .csv extension\n        df['Id'] = file_name[:-4]\n        df = df.drop(columns = ['Valid', 'Task'])\n        defog_list.append(df)\n        \ndefog = pd.concat(defog_list, ignore_index=True)\n\n# create 'IsFOG' column based on any non-zero value in 'StartHesitation', 'Walking', 'Turn' columns\ndefog['IsFOG'] = defog[['StartHesitation', 'Walking', 'Turn']].any(axis='columns')\nprint(defog)","metadata":{"execution":{"iopub.status.busy":"2024-08-28T14:20:03.726843Z","iopub.execute_input":"2024-08-28T14:20:03.727267Z","iopub.status.idle":"2024-08-28T14:20:21.853592Z","shell.execute_reply.started":"2024-08-28T14:20:03.727232Z","shell.execute_reply":"2024-08-28T14:20:21.852361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Re-load the metadata file and count the unique number of subjects\nmetadata_defog = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv')\nunique_subjects = metadata_defog['Subject'].nunique()\n\n# Merge the tdcsfog dataframe with metadata based on 'ID'\ndefog = defog.merge(metadata_defog[['Id', 'Subject']], on='Id', how='left')\n\n# Merge the tdcsfog dataframe with defog metadata based on 'ID'\ndefog = defog.merge(metadata_defog[['Id', 'Medication']], on='Id', how='left')\n\nprint(unique_subjects)\nprint(defog['Subject'].nunique())\nprint(defog)","metadata":{"execution":{"iopub.status.busy":"2024-08-28T14:20:26.866274Z","iopub.execute_input":"2024-08-28T14:20:26.867416Z","iopub.status.idle":"2024-08-28T14:20:37.698567Z","shell.execute_reply.started":"2024-08-28T14:20:26.867359Z","shell.execute_reply":"2024-08-28T14:20:37.697444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Bandpass Filter**","metadata":{}},{"cell_type":"code","source":"# Apllying bandpass filter to all the Acc \n\ndef apply_bandpass_filter(data, lowcut, highcut, fs, order=2):\n    \"\"\"\n    Apply a bandpass filter to the data.\n    \n    Parameters:\n    - data: numpy array of the signal to be filtered\n    - lowcut: lower cutoff frequency\n    - highcut: higher cutoff frequency\n    - fs: sampling frequency\n    - order: filter order, defines steepness\n    \n    Returns:\n    - filtered_data: numpy array of the filtered signal\n    \"\"\"\n    nyquist = 0.5 * fs  # Nyquist Frequency\n    low = lowcut / nyquist\n    high = highcut / nyquist\n    \n    b, a = signal.butter(order, [low, high], btype='band')\n    filtered_data = signal.lfilter(b, a, data)\n    \n    return filtered_data\n\n# Define the sampling frequency for tDCSFOG and DeFOG data\nsampling_freq_tdcsfog = 128  # For tDCSFOG\nsampling_freq_defog = 100    # For DeFOG\n\n# Apply bandpass filter for each axis (AccV, AccML, AccAP) in the dataset\ntdcsfog['AccV_filtered'] = apply_bandpass_filter(tdcsfog['AccV'].values, lowcut=0.5, highcut=20, fs=sampling_freq_tdcsfog)\ntdcsfog['AccML_filtered'] = apply_bandpass_filter(tdcsfog['AccML'].values, lowcut=0.5, highcut=20, fs=sampling_freq_tdcsfog)\ntdcsfog['AccAP_filtered'] = apply_bandpass_filter(tdcsfog['AccAP'].values, lowcut=0.5, highcut=20, fs=sampling_freq_tdcsfog)\n\n# for DeFOG data:\ndefog['AccV_filtered'] = apply_bandpass_filter(defog['AccV'].values, lowcut=0.5, highcut=20, fs=sampling_freq_defog)\ndefog['AccML_filtered'] = apply_bandpass_filter(defog['AccML'].values, lowcut=0.5, highcut=20, fs=sampling_freq_defog)\ndefog['AccAP_filtered'] = apply_bandpass_filter(defog['AccAP'].values, lowcut=0.5, highcut=20, fs=sampling_freq_defog)\n","metadata":{"execution":{"iopub.status.busy":"2024-08-28T14:20:45.504048Z","iopub.execute_input":"2024-08-28T14:20:45.504471Z","iopub.status.idle":"2024-08-28T14:20:46.991991Z","shell.execute_reply.started":"2024-08-28T14:20:45.504431Z","shell.execute_reply":"2024-08-28T14:20:46.990807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom scipy.signal import welch\n\ndef spectral_entropy(signal, sampling_frequency, nperseg=None):\n    \"\"\"\n    Calculate the spectral entropy of a signal.\n    \n    Parameters:\n    - signal: 1D array-like, the time series data.\n    - sampling_frequency: float, the sampling frequency of the signal.\n    - nperseg: int, optional, length of each segment for Welch's method (default is None).\n    \n    Returns:\n    - entropy: float, the spectral entropy of the signal.\n    \"\"\"\n    # Calculate Power Spectral Density (PSD) using Welch's method\n    frequencies, psd = welch(signal, fs=sampling_frequency, nperseg=nperseg)\n    \n    # Normalize the PSD to get a probability distribution\n    psd_norm = psd / np.sum(psd)\n    \n    # Calculate the spectral entropy\n    entropy = -np.sum(psd_norm * np.log2(psd_norm + np.finfo(float).eps))\n    \n    return entropy\n\ndef calculate_entropy_per_id(df, sampling_frequency, nperseg=None):\n    # Group by 'Id' and calculate entropy for each accelerometer axis\n    entropy_accv = df.groupby('Id')['AccV'].apply(lambda x: spectral_entropy(x, sampling_frequency, nperseg))\n    entropy_accml = df.groupby('Id')['AccML'].apply(lambda x: spectral_entropy(x, sampling_frequency, nperseg))\n    entropy_accap = df.groupby('Id')['AccAP'].apply(lambda x: spectral_entropy(x, sampling_frequency, nperseg))\n\n    # Add the calculated entropy as new columns in the DataFrame\n    df['Entropy_AccV'] = df['Id'].map(entropy_accv)\n    df['Entropy_AccML'] = df['Id'].map(entropy_accml)\n    df['Entropy_AccAP'] = df['Id'].map(entropy_accap)\n\n    return df\n\n# Example usage\nfs_defog = 100 \nfs_tdcs = 128\n\n# Apply the entropy calculation to the tDCSFOG dataset\ntdcsfog = calculate_entropy_per_id(tdcsfog, fs_tdcs)\n\n# Apply the entropy calculation to the DeFOG dataset\ndefog = calculate_entropy_per_id(defog, fs_defog)\n\n# Now, both `tdcsfog` and `defog` DataFrames will have new columns:\n# 'Entropy_AccV', 'Entropy_AccML', and 'Entropy_AccAP'\n\n","metadata":{"execution":{"iopub.status.busy":"2024-08-28T14:20:57.104800Z","iopub.execute_input":"2024-08-28T14:20:57.105294Z","iopub.status.idle":"2024-08-28T14:21:15.278041Z","shell.execute_reply.started":"2024-08-28T14:20:57.105257Z","shell.execute_reply":"2024-08-28T14:21:15.276852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Function to check if a group contains any FOG events\ndef contains_fog(group):\n    return group[['StartHesitation', 'Turn', 'Walking']].any().any()\n\n# Get unique subjects\nunique_subjects_tdcs = tdcsfog['Subject'].unique()\nunique_subjects_defog = defog['Subject'].unique()\n\n\n# Split subjects into train and test while preserving FOG event representation\ntrain_subjects_tdcsfog, test_subjects_tdcsfog = train_test_split(unique_subjects_tdcs, test_size=0.2, random_state=42)\ntrain_subjects_defog, test_subjects_defog = train_test_split(unique_subjects_defog, test_size=0.2, random_state=42)\n\n# Create the train and test splits by filtering the original dataframes\ntrain_tdcsfog = tdcsfog[tdcsfog['Subject'].isin(train_subjects_tdcsfog)]\ntest_tdcsfog = tdcsfog[tdcsfog['Subject'].isin(test_subjects_tdcsfog)]\n\ntrain_defog = defog[defog['Subject'].isin(train_subjects_defog)]\ntest_defog = defog[defog['Subject'].isin(test_subjects_defog)]\n\n# Check FOG event distribution\ntrain_fog = train_tdcsfog.groupby('Subject').apply(contains_fog).reset_index(name='contains_fog')\ntest_fog = test_tdcsfog.groupby('Subject').apply(contains_fog).reset_index(name='contains_fog')\n\n# If test set lacks FOG events, adjust the split\nwhile not test_fog['contains_fog'].any():\n    # Re-split to try and ensure test set contains FOG events\n    train_subjects_tdcsfog, test_subjects_tdcsfog = train_test_split(unique_subjects_tdcsfog, test_size=0.2, random_state=None)\n    train_tdcsfog = tdcsfog[tdcsfog['Subject'].isin(train_subjects_tdcsfog)]\n    test_tdcsfog = tdcsfog[tdcsfog['Subject'].isin(test_subjects_tdcsfog)]\n    test_fog = test_tdcsfog.groupby('Subject').apply(contains_fog).reset_index(name='contains_fog')\n\n# Same process for DeFOG data\ntrain_fog_defog = train_defog.groupby('Subject').apply(contains_fog).reset_index(name='contains_fog')\ntest_fog_defog = test_defog.groupby('Subject').apply(contains_fog).reset_index(name='contains_fog')\n\nwhile not test_fog_defog['contains_fog'].any():\n    train_subjects_defog, test_subjects_defog = train_test_split(unique_subjects_defog, test_size=0.2, random_state=None)\n    train_defog = defog[defog['Subject'].isin(train_subjects_defog)]\n    test_defog = defog[defog['Subject'].isin(test_subjects_defog)]\n    test_fog_defog = test_defog.groupby('Subject').apply(contains_fog).reset_index(name='contains_fog')\n\n# After successful split, print the results\nprint(f\"Unique subjects in train set tdcs: {train_tdcsfog['Subject'].nunique()}\")\nprint(f\"Unique subjects in test set tdcs: {test_tdcsfog['Subject'].nunique()}\")\n\nprint(f\"Unique subjects in train set defog: {train_defog['Subject'].nunique()}\")\nprint(f\"Unique subjects in test set defog: {test_defog['Subject'].nunique()}\")\n\n# Checking the distribution of FOG events in train and test sets\nprint(\"tDCSFOG - Train FOG Distribution:\")\nprint(train_tdcsfog[['StartHesitation', 'Turn', 'Walking']].sum())\n\nprint(\"tDCSFOG - Test FOG Distribution:\")\nprint(test_tdcsfog[['StartHesitation', 'Turn', 'Walking']].sum())\n\nprint(\"DeFOG - Train FOG Distribution:\")\nprint(train_defog[['StartHesitation', 'Turn', 'Walking']].sum())\n\nprint(\"DeFOG - Test FOG Distribution:\")\nprint(test_defog[['StartHesitation', 'Turn', 'Walking']].sum())\n","metadata":{"execution":{"iopub.status.busy":"2024-08-28T14:21:20.882622Z","iopub.execute_input":"2024-08-28T14:21:20.883072Z","iopub.status.idle":"2024-08-28T14:21:36.375822Z","shell.execute_reply.started":"2024-08-28T14:21:20.883038Z","shell.execute_reply":"2024-08-28T14:21:36.374244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check if we have only 38 and not 45 subject as in the meta data becouse of an error\n\n# Get unique IDs from the defog DataFrame\ndefog_s = defog['Subject'].unique()\n\n# Get unique IDs from the metadata\nmetadata_s = metadata_defog['Subject'].unique()\n\n# Check for missing IDs\nmissing_s = [s for s in metadata_s if s not in defog_s]\n\nif len(missing_s) == 0:\n    print(\"All subjects in the metadata have corresponding data in the defog DataFrame.\")\nelse:\n    print(f\"Subjects in the metadata without corresponding data in the defog DataFrame: {len(missing_s)}\")\n    print(\"Missing Subjects:\", missing_s)\n","metadata":{"execution":{"iopub.status.busy":"2024-08-28T14:21:58.760059Z","iopub.execute_input":"2024-08-28T14:21:58.760612Z","iopub.status.idle":"2024-08-28T14:21:59.978156Z","shell.execute_reply.started":"2024-08-28T14:21:58.760569Z","shell.execute_reply":"2024-08-28T14:21:59.976762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model 1 - was there fog or not?\n\n# Define feature columns\nfeature_columns = ['AccV_filtered', 'AccML_filtered', 'AccAP_filtered', 'Entropy_AccAP', 'Entropy_AccML', 'Entropy_AccV']\n\n# tDCSFOG Features and Labels\nX_train_tdcsfog = train_tdcsfog[feature_columns]\ny_train_tdcsfog = train_tdcsfog['IsFOG']\nX_test_tdcsfog = test_tdcsfog[feature_columns]\ny_test_tdcsfog = test_tdcsfog['IsFOG']\n\n# DeFOG Features and Labels\nX_train_defog = train_defog[feature_columns]\ny_train_defog = train_defog['IsFOG']\nX_test_defog = test_defog[feature_columns]\ny_test_defog = test_defog['IsFOG']\n# Initialize the LightGBM model for tDCSFOG\nlgb_model_tdcsfog = lgb.LGBMClassifier(n_estimators=100, random_state=42)\n\n# Initialize the LightGBM model for DeFOG\nlgb_model_defog = lgb.LGBMClassifier(n_estimators=100, random_state=42)\n\n# Train the LightGBM model on tDCSFOG data\nlgb_model_tdcsfog.fit(X_train_tdcsfog, y_train_tdcsfog)\n\n# Train the LightGBM model on DeFOG data\nlgb_model_defog.fit(X_train_defog, y_train_defog)\n# Predictions for tDCSFOG\ny_pred_tdcsfog = lgb_model_tdcsfog.predict(X_test_tdcsfog)\n\n# Predictions for DeFOG\ny_pred_defog = lgb_model_defog.predict(X_test_defog)\n\n# Evaluation for tDCSFOG\nprint(\"tDCSFOG Data - LightGBM Model\")\nprint(\"Accuracy:\", accuracy_score(y_test_tdcsfog, y_pred_tdcsfog))\nprint(\"Classification Report:\\n\", classification_report(y_test_tdcsfog, y_pred_tdcsfog))\nprint(\"Confusion Matrix:\\n\", confusion_matrix(y_test_tdcsfog, y_pred_tdcsfog))\n\n# Evaluation for DeFOG\nprint(\"\\nDeFOG Data - LightGBM Model\")\nprint(\"Accuracy:\", accuracy_score(y_test_defog, y_pred_defog))\nprint(\"Classification Report:\\n\", classification_report(y_test_defog, y_pred_defog))\nprint(\"Confusion Matrix:\\n\", confusion_matrix(y_test_defog, y_pred_defog))\n","metadata":{"execution":{"iopub.status.busy":"2024-08-28T14:06:04.885977Z","iopub.execute_input":"2024-08-28T14:06:04.888189Z","iopub.status.idle":"2024-08-28T14:09:30.707604Z","shell.execute_reply.started":"2024-08-28T14:06:04.888102Z","shell.execute_reply":"2024-08-28T14:09:30.706266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Step 1: Filter the data for FOG events\n#tdcsfog_fog_events = test_tdcsfog[test_tdcsfog['IsFOG'] == 1]\n#defog_fog_events = test_defog[test_defog['IsFOG'] == 1]\n\ntdcsfog_fog_events = test_tdcsfog\ndefog_fog_events = test_defog\n\n# Step 2: Define the feature columns and label columns\nfeature_columns = ['AccV_filtered', 'AccML_filtered', 'AccAP_filtered', 'Time', 'Entropy_AccV', 'Entropy_AccML', 'Entropy_AccAP']\nlabel_columns = ['StartHesitation', 'Turn', 'Walking']\n\n# tDCSFOG Features and Labels\nX_tdcsfog_subtypes = tdcsfog_fog_events[feature_columns]\ny_tdcsfog_subtypes = tdcsfog_fog_events[label_columns]\n\n# DeFOG Features and Labels\nX_defog_subtypes = defog_fog_events[feature_columns]\ny_defog_subtypes = defog_fog_events[label_columns]\n\n# Map \"on\" to 1 and \"off\" to 0\n#X_tdcsfog_subtypes['Medication'] = X_tdcsfog_subtypes['Medication'].map({'on': 1, 'off': 0})\n#X_defog_subtypes['Medication'] = X_defog_subtypes['Medication'].map({'on': 1, 'off': 0})\n\n# Ensure there are no missing values after mapping\n#X_tdcsfog_subtypes['Medication'].fillna(0, inplace=True)  # Handle any unexpected categories\n#X_defog_subtypes['Medication'].fillna(0, inplace=True)  # Handle any unexpected categories\n\n","metadata":{"execution":{"iopub.status.busy":"2024-08-28T14:22:20.279382Z","iopub.execute_input":"2024-08-28T14:22:20.279812Z","iopub.status.idle":"2024-08-28T14:22:20.455223Z","shell.execute_reply.started":"2024-08-28T14:22:20.279777Z","shell.execute_reply":"2024-08-28T14:22:20.454102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Step 4: Initialize and Train LightGBM models for each subtype\n# tDCSFOG\nlgb_model_tdcsfog_SH = lgb.LGBMClassifier(n_estimators=100, random_state=42)\nlgb_model_tdcsfog_Turn = lgb.LGBMClassifier(n_estimators=100, random_state=42)\nlgb_model_tdcsfog_Walking = lgb.LGBMClassifier(n_estimators=100, random_state=42)\n\nlgb_model_tdcsfog_SH.fit(X_tdcsfog_subtypes, y_tdcsfog_subtypes['StartHesitation'])\nlgb_model_tdcsfog_Turn.fit(X_tdcsfog_subtypes, y_tdcsfog_subtypes['Turn'])\nlgb_model_tdcsfog_Walking.fit(X_tdcsfog_subtypes, y_tdcsfog_subtypes['Walking'])\n\n# DeFOG\nlgb_model_defog_SH = lgb.LGBMClassifier(n_estimators=100, random_state=42)\nlgb_model_defog_Turn = lgb.LGBMClassifier(n_estimators=100, random_state=42)\nlgb_model_defog_Walking = lgb.LGBMClassifier(n_estimators=100, random_state=42)\n\nlgb_model_defog_SH.fit(X_defog_subtypes, y_defog_subtypes['StartHesitation'])\nlgb_model_defog_Turn.fit(X_defog_subtypes, y_defog_subtypes['Turn'])\nlgb_model_defog_Walking.fit(X_defog_subtypes, y_defog_subtypes['Walking'])\n","metadata":{"execution":{"iopub.status.busy":"2024-08-28T14:22:24.860756Z","iopub.execute_input":"2024-08-28T14:22:24.861504Z","iopub.status.idle":"2024-08-28T14:23:57.011917Z","shell.execute_reply.started":"2024-08-28T14:22:24.861435Z","shell.execute_reply":"2024-08-28T14:23:57.010656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Step 5: Make Predictions\n# tDCSFOG\ny_pred_tdcsfog_SH = lgb_model_tdcsfog_SH.predict_proba(X_tdcsfog_subtypes)[:,1]\ny_pred_tdcsfog_Turn = lgb_model_tdcsfog_Turn.predict_proba(X_tdcsfog_subtypes)[:,1]\ny_pred_tdcsfog_Walking = lgb_model_tdcsfog_Walking.predict_proba(X_tdcsfog_subtypes)[:,1]\n\n# DeFOG\ny_pred_defog_SH = lgb_model_defog_SH.predict_proba(X_defog_subtypes)[:,1]\ny_pred_defog_Turn = lgb_model_defog_Turn.predict_proba(X_defog_subtypes)[:,1]\ny_pred_defog_Walking = lgb_model_defog_Walking.predict_proba(X_defog_subtypes)[:,1]\n\n","metadata":{"execution":{"iopub.status.busy":"2024-08-28T14:24:05.943462Z","iopub.execute_input":"2024-08-28T14:24:05.943852Z","iopub.status.idle":"2024-08-28T14:24:27.403314Z","shell.execute_reply.started":"2024-08-28T14:24:05.943821Z","shell.execute_reply":"2024-08-28T14:24:27.402015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Step 6: Evaluate the Models and Collect Scores\n\n# tDCSFOG Evaluation and DataFrame\ntdcsfog_results = pd.DataFrame({'Id': tdcsfog_fog_events['Id']\n    + '_' + tdcsfog_fog_events['Time'].astype(str),\n    'StartHesitation_Pred': y_pred_tdcsfog_SH,\n    'Turn_Pred': y_pred_tdcsfog_Turn,\n    'Walking_Pred': y_pred_tdcsfog_Walking\n})\n\n# DeFOG Evaluation and DataFrame\ndefog_results = pd.DataFrame({'Id': defog_fog_events['Id']\n    + '_' + defog_fog_events['Time'].astype(str),\n    'StartHesitation_Pred': y_pred_defog_SH,\n    'Turn_Pred': y_pred_defog_Turn,\n    'Walking_Pred': y_pred_defog_Walking\n})\n\nprint(tdcsfog_results)\nprint(defog_results)","metadata":{"execution":{"iopub.status.busy":"2024-08-28T14:25:03.329131Z","iopub.execute_input":"2024-08-28T14:25:03.330231Z","iopub.status.idle":"2024-08-28T14:25:06.694814Z","shell.execute_reply.started":"2024-08-28T14:25:03.330158Z","shell.execute_reply":"2024-08-28T14:25:06.693640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nSH = average_precision_score(y_tdcsfog_subtypes['StartHesitation'], y_pred_tdcsfog_SH)\nTurn = average_precision_score(y_tdcsfog_subtypes['Turn'], y_pred_tdcsfog_Turn)\nWalking = average_precision_score(y_tdcsfog_subtypes['Walking'], y_pred_tdcsfog_Walking)\n\n\ntotal = np.mean([SH, Turn, Walking])\nprint(total)","metadata":{"execution":{"iopub.status.busy":"2024-08-28T14:25:15.925123Z","iopub.execute_input":"2024-08-28T14:25:15.925674Z","iopub.status.idle":"2024-08-28T14:25:16.688806Z","shell.execute_reply.started":"2024-08-28T14:25:15.925630Z","shell.execute_reply":"2024-08-28T14:25:16.687507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Test evaluation**","metadata":{}},{"cell_type":"code","source":"# loading test file\ntdcsfog_test_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog'\n        \ntdcsfog_test_list = [\n    pd.read_csv(os.path.join(tdcsfog_test_path, file_name)).assign(Id=lambda df: file_name[:-4] + '_' + df['Time'].astype(str))\n    for file_name in os.listdir(tdcsfog_test_path)\n    if file_name.endswith('.csv')\n]        \n\ntdcsfog_test = pd.concat(tdcsfog_test_list, axis = 0)","metadata":{"execution":{"iopub.status.busy":"2024-08-28T14:25:32.281632Z","iopub.execute_input":"2024-08-28T14:25:32.282020Z","iopub.status.idle":"2024-08-28T14:25:32.308463Z","shell.execute_reply.started":"2024-08-28T14:25:32.281991Z","shell.execute_reply":"2024-08-28T14:25:32.307397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_test_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog'\ntdcsfog_list = []\n\n# iterate over each file in the directory\nfor file_name in os.listdir(tdcsfog_test_path):\n    # exclude this file because it is also in the test set\n    if file_name.endswith('.csv'): \n        file_path = os.path.join(tdcsfog_test_path, file_name)\n        df = pd.read_csv(file_path)\n        \n        # add a new column with the file name without the .csv extension\n        df['Id'] = file_name[:-4]\n        tdcsfog_list.append(df)\n\ntdcsfog = pd.concat(tdcsfog_list, ignore_index=True)\n\nprint(tdcsfog)\n","metadata":{"execution":{"iopub.status.busy":"2024-08-28T14:26:58.054129Z","iopub.execute_input":"2024-08-28T14:26:58.054561Z","iopub.status.idle":"2024-08-28T14:26:58.077435Z","shell.execute_reply.started":"2024-08-28T14:26:58.054527Z","shell.execute_reply":"2024-08-28T14:26:58.076275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_test_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog'\ndefog_list = []\n\n# iterate over each file in the directory\nfor file_name in os.listdir(defog_test_path):\n    # exclude this file because it is also in the test set\n    if file_name.endswith('.csv'): \n        file_path = os.path.join(defog_test_path, file_name)\n        df = pd.read_csv(file_path)\n        \n        # add a new column with the file name without the .csv extension\n        df['Id'] = file_name[:-4]\n        defog_list.append(df)\n\ndefog = pd.concat(defog_list, ignore_index=True)\n\nprint(defog)","metadata":{"execution":{"iopub.status.busy":"2024-08-28T14:26:53.841317Z","iopub.execute_input":"2024-08-28T14:26:53.841755Z","iopub.status.idle":"2024-08-28T14:26:54.106998Z","shell.execute_reply.started":"2024-08-28T14:26:53.841712Z","shell.execute_reply":"2024-08-28T14:26:54.105672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Define the sampling frequency for tDCSFOG and DeFOG data\nsampling_freq_tdcsfog = 128  # For tDCSFOG\nsampling_freq_defog = 100    # For DeFOG\n\n# Apply bandpass filter for each axis (AccV, AccML, AccAP) in the dataset\ntdcsfog['AccV_filtered'] = apply_bandpass_filter(tdcsfog['AccV'].values, lowcut=0.5, highcut=30, fs=sampling_freq_tdcsfog)\ntdcsfog['AccML_filtered'] = apply_bandpass_filter(tdcsfog['AccML'].values, lowcut=0.5, highcut=30, fs=sampling_freq_tdcsfog)\ntdcsfog['AccAP_filtered'] = apply_bandpass_filter(tdcsfog['AccAP'].values, lowcut=0.5, highcut=30, fs=sampling_freq_tdcsfog)\n\n# for DeFOG data:\ndefog['AccV_filtered'] = apply_bandpass_filter(defog['AccV'].values, lowcut=0.5, highcut=30, fs=sampling_freq_defog)\ndefog['AccML_filtered'] = apply_bandpass_filter(defog['AccML'].values, lowcut=0.5, highcut=30, fs=sampling_freq_defog)\ndefog['AccAP_filtered'] = apply_bandpass_filter(defog['AccAP'].values, lowcut=0.5, highcut=30, fs=sampling_freq_defog)\n\n\n# Apply the entropy calculation to the tDCSFOG dataset\ntdcsfog = calculate_entropy_per_id(tdcsfog, sampling_freq_tdcsfog)\n\n# Apply the entropy calculation to the DeFOG dataset\ndefog = calculate_entropy_per_id(defog, sampling_freq_defog)","metadata":{"execution":{"iopub.status.busy":"2024-08-28T14:27:04.289437Z","iopub.execute_input":"2024-08-28T14:27:04.289866Z","iopub.status.idle":"2024-08-28T14:27:04.512461Z","shell.execute_reply.started":"2024-08-28T14:27:04.289834Z","shell.execute_reply":"2024-08-28T14:27:04.511256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Assuming defog is your DataFrame\n# Select the relevant columns\nfeature_columns = ['AccV_filtered', 'AccML_filtered', 'AccAP_filtered', 'Time', 'Entropy_AccAP', 'Entropy_AccML', 'Entropy_AccV']\n\n# Create a new DataFrame with the selected columns\ndf_defog_filtered = defog[feature_columns].copy()\ndf_tdcsfog_filtered = tdcsfog[feature_columns].copy()\n\n# Add a new 'Medication' column with the value 0.5\n#df_defog_filtered['Medication'] = 0.5 #the odds for madications or not are equal- we dont know\n#df_tdcsfog_filtered['Medication'] = 0.5\n\n# Now df_defog_filtered has the features you need\nprint(df_defog_filtered.head())\nprint(df_tdcsfog_filtered.head())\n","metadata":{"execution":{"iopub.status.busy":"2024-08-28T14:27:09.436165Z","iopub.execute_input":"2024-08-28T14:27:09.437250Z","iopub.status.idle":"2024-08-28T14:27:09.476025Z","shell.execute_reply.started":"2024-08-28T14:27:09.437183Z","shell.execute_reply":"2024-08-28T14:27:09.474789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tDCSFOG\ny_pred_tdcsfog_SH = lgb_model_tdcsfog_SH.predict_proba(df_tdcsfog_filtered)[:,1]\ny_pred_tdcsfog_Turn = lgb_model_tdcsfog_Turn.predict_proba(df_tdcsfog_filtered)[:,1]\ny_pred_tdcsfog_Walking = lgb_model_tdcsfog_Walking.predict_proba(df_tdcsfog_filtered)[:,1]\n\n# DeFOG\ny_pred_defog_SH = lgb_model_defog_SH.predict_proba(df_defog_filtered)[:,1]\ny_pred_defog_Turn = lgb_model_defog_Turn.predict_proba(df_defog_filtered)[:,1]\ny_pred_defog_Walking = lgb_model_defog_Walking.predict_proba(df_defog_filtered)[:,1]\n","metadata":{"execution":{"iopub.status.busy":"2024-08-28T14:27:17.692858Z","iopub.execute_input":"2024-08-28T14:27:17.693330Z","iopub.status.idle":"2024-08-28T14:27:19.189721Z","shell.execute_reply.started":"2024-08-28T14:27:17.693290Z","shell.execute_reply":"2024-08-28T14:27:19.188577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# tDCSFOG Evaluation and DataFrame\ntdcsfog_results = pd.DataFrame({'Id': tdcsfog['Id']\n    + '_' + tdcsfog['Time'].astype(str),\n    'StartHesitation_Pred': y_pred_tdcsfog_SH,\n    'Turn_Pred': y_pred_tdcsfog_Turn,\n    'Walking_Pred': y_pred_tdcsfog_Walking\n})\n\n# DeFOG Evaluation and DataFrame\ndefog_results = pd.DataFrame({'Id': defog['Id']\n    + '_' + defog['Time'].astype(str),\n    'StartHesitation_Pred': y_pred_defog_SH,\n    'Turn_Pred': y_pred_defog_Turn,\n    'Walking_Pred': y_pred_defog_Walking\n})\n\nprint(tdcsfog_results)\nprint(defog_results)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-08-28T14:27:23.548288Z","iopub.execute_input":"2024-08-28T14:27:23.549405Z","iopub.status.idle":"2024-08-28T14:27:23.951107Z","shell.execute_reply.started":"2024-08-28T14:27:23.549363Z","shell.execute_reply":"2024-08-28T14:27:23.949883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the DataFrame to a CSV file\n# Concatenate the two DataFrames\ncombined_results = pd.concat([tdcsfog_results, defog_results], axis=0)\n\n# Save the combined DataFrame to a CSV file\ncombined_results.to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-08-28T14:27:31.241549Z","iopub.execute_input":"2024-08-28T14:27:31.241975Z","iopub.status.idle":"2024-08-28T14:27:33.387223Z","shell.execute_reply.started":"2024-08-28T14:27:31.241940Z","shell.execute_reply":"2024-08-28T14:27:33.385805Z"},"trusted":true},"execution_count":null,"outputs":[]}]}