{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":41880,"databundleVersionId":5677426,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# importing all the libraries\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport warnings\nimport os\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.model_selection import train_test_split\n\n#to remove warnings\nwarnings.filterwarnings(action = \"ignore\", category = DeprecationWarning ) \nwarnings.filterwarnings(action = \"ignore\", category = FutureWarning ) \nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os\nimport seaborn as sns\nimport warnings\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nfrom sklearn.metrics import accuracy_score, classification_report\n\nimport lightgbm as lgb\n\nimport pywt\n\nwarnings.filterwarnings(action = \"ignore\", category = DeprecationWarning ) \nwarnings.filterwarnings(action = \"ignore\", category = FutureWarning ) ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-08-24T11:36:55.033215Z","iopub.execute_input":"2024-08-24T11:36:55.034291Z","iopub.status.idle":"2024-08-24T11:36:59.665815Z","shell.execute_reply.started":"2024-08-24T11:36:55.034243Z","shell.execute_reply":"2024-08-24T11:36:59.664437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog'\ntdcsfog_list = []\n\n# Iterate over each file in the directory\nfor file_name in os.listdir(tdcsfog_path):\n    if file_name.endswith('.csv') and file_name != '003f117e14.csv':  # Exclude the specific file\n        file_path = os.path.join(tdcsfog_path, file_name)\n        df = pd.read_csv(file_path)\n        \n        # Add a new column with the file name without the .csv extension\n        df['file_name'] = file_name[:-4]\n        \n        tdcsfog_list.append(df)\n\n# Concatenate all DataFrames in the list into a single DataFrame\ntdcsfog = pd.concat(tdcsfog_list, ignore_index=True)\n\n# Create the 'IsFOG' column based on any non-zero value in 'StartHesitation', 'Walking', 'Turn' columns\ntdcsfog['IsFOG'] = tdcsfog[['StartHesitation', 'Walking', 'Turn']].any(axis='columns')\n\n# Display the first few rows of the DataFrame\ntdcsfog.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-08-24T11:37:03.001106Z","iopub.execute_input":"2024-08-24T11:37:03.001545Z","iopub.status.idle":"2024-08-24T11:37:25.660550Z","shell.execute_reply.started":"2024-08-24T11:37:03.001505Z","shell.execute_reply":"2024-08-24T11:37:25.658959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_metadata = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv')\n# Merge the dataframes based on matching 'file_name' in tdcsfog and 'id' in tdcsfog_metadata\n# Assuming 'file_name' and 'id' are the column names in the respective dataframes\n# and that the 'id' in tdcsfog_metadata corresponds to 'file_name' in tdcsfog\ntdcsfog = tdcsfog.merge(tdcsfog_metadata[['Id', 'Subject']], left_on='file_name', right_on='Id', how='left')\n\n# Rename the 'Subject' column from tdcsfog_metadata to 'subject' in tdcsfog\ntdcsfog = tdcsfog.rename(columns={'Subject': 'subject'})\n# Drop the now unnecessary 'id' column from the merge\ntdcsfog = tdcsfog.drop(columns=['Id'])\n\nprint(tdcsfog.head())\n# Count the number of unique values in each column\nunique_counts = tdcsfog.nunique()\n\n# Display the number of unique values in each column\nprint(unique_counts)","metadata":{"execution":{"iopub.status.busy":"2024-08-24T11:37:29.114935Z","iopub.execute_input":"2024-08-24T11:37:29.115383Z","iopub.status.idle":"2024-08-24T11:37:40.146004Z","shell.execute_reply.started":"2024-08-24T11:37:29.115347Z","shell.execute_reply":"2024-08-24T11:37:40.144812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog'\ndefog_list = []\n\n# Iterate over each file in the directory\nfor file_name in os.listdir(defog_path):\n    if file_name.endswith('.csv') and file_name != '003f117e14.csv':  # Exclude the specific file\n        file_path = os.path.join(defog_path, file_name)\n        df = pd.read_csv(file_path)\n        \n        # Add a new column with the file name without the .csv extension\n        df['file_name'] = file_name[:-4]\n        \n        defog_list.append(df)\n\n# Concatenate all DataFrames in the list into a single DataFrame\ndefog = pd.concat(defog_list, ignore_index=True)\n\n# Create the 'IsFOG' column based on any non-zero value in 'StartHesitation', 'Walking', 'Turn' columns\ndefog['IsFOG'] = defog[['StartHesitation', 'Walking', 'Turn']].any(axis='columns')\n\n# Display the first few rows of the DataFrame\ndefog.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-24T11:37:45.458702Z","iopub.execute_input":"2024-08-24T11:37:45.459114Z","iopub.status.idle":"2024-08-24T11:38:14.044126Z","shell.execute_reply.started":"2024-08-24T11:37:45.459081Z","shell.execute_reply":"2024-08-24T11:38:14.042916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_metadata = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv')\n# Merge the dataframes based on matching 'file_name' in tdcsfog and 'id' in tdcsfog_metadata\n# Assuming 'file_name' and 'id' are the column names in the respective dataframes\n# and that the 'id' in tdcsfog_metadata corresponds to 'file_name' in tdcsfog\ndefog = defog.merge(defog_metadata[['Id', 'Subject']], left_on='file_name', right_on='Id', how='left')\n\n# Rename the 'Subject' column from tdcsfog_metadata to 'subject' in tdcsfog\ndefog = defog.rename(columns={'Subject': 'subject'})\n# Drop the now unnecessary 'id' column from the merge\ndefog = defog.drop(columns=['Id'])\n\nprint(tdcsfog.head())\n# Count the number of unique values in each column\nunique_counts = defog.nunique()\n\n# Display the number of unique values in each column\nprint(unique_counts)","metadata":{"execution":{"iopub.status.busy":"2024-08-24T11:38:16.558400Z","iopub.execute_input":"2024-08-24T11:38:16.558808Z","iopub.status.idle":"2024-08-24T11:38:34.950223Z","shell.execute_reply.started":"2024-08-24T11:38:16.558778Z","shell.execute_reply":"2024-08-24T11:38:34.948952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.signal import butter, filtfilt\n\ndef lowpass_filter(signal, cutoff_freq, sampling_rate, order=4):\n    # Design a Butterworth lowpass filter\n    nyquist = 0.5 * sampling_rate\n    normal_cutoff = cutoff_freq / nyquist\n    b, a = butter(order, normal_cutoff, btype='low', analog=False)\n    # Apply the filter to the signal\n    filtered_signal = filtfilt(b, a, signal)\n    return filtered_signal","metadata":{"execution":{"iopub.status.busy":"2024-08-24T11:38:42.151266Z","iopub.execute_input":"2024-08-24T11:38:42.151768Z","iopub.status.idle":"2024-08-24T11:38:42.159294Z","shell.execute_reply.started":"2024-08-24T11:38:42.151729Z","shell.execute_reply":"2024-08-24T11:38:42.157548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def morlet_wavelet_features(signal, min_scale_log=3.6, max_scale_log=5.4, sampling_period = 1/100):\n    # Compute scales for the CWT\n    scales = np.exp(np.arange(min_scale_log, max_scale_log, 0.05))\n    wavelet = 'morl'\n    \n    # Perform the CWT\n    coeff, freq = pywt.cwt(signal, scales, wavelet, sampling_period=sampling_period)\n    \n    # Find the index of the max coefficient at each time point\n    max_coeff_indices = np.argmax(np.abs(coeff), axis=0)\n    \n    # Extract the max coefficients along the scale axis\n    max_coeff = coeff[max_coeff_indices, range(coeff.shape[1])]\n    \n    # Extract the corresponding frequencies for the max coefficients\n    max_freq = freq[max_coeff_indices]\n    \n    \n    return max_coeff, max_freq","metadata":{"execution":{"iopub.status.busy":"2024-08-24T11:38:44.434126Z","iopub.execute_input":"2024-08-24T11:38:44.434900Z","iopub.status.idle":"2024-08-24T11:38:44.446648Z","shell.execute_reply.started":"2024-08-24T11:38:44.434843Z","shell.execute_reply":"2024-08-24T11:38:44.445122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sampling_freq_tdcsfog = 128  \nsampling_freq_defog = 100\ncutoff_freq = 50\n\n# Calculate the sampling periods\nsampling_period_tdcsfog = 1 / sampling_freq_tdcsfog\nsampling_period_defog = 1 / sampling_freq_defog","metadata":{"execution":{"iopub.status.busy":"2024-08-24T11:38:46.991405Z","iopub.execute_input":"2024-08-24T11:38:46.991855Z","iopub.status.idle":"2024-08-24T11:38:46.998591Z","shell.execute_reply.started":"2024-08-24T11:38:46.991822Z","shell.execute_reply":"2024-08-24T11:38:46.997382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_results = pd.DataFrame()\n\nfor file_name in tdcsfog['file_name'].unique():\n    subject = tdcsfog[tdcsfog['file_name'] == file_name].copy()\n    \n    # Apply lowpass filter to all three signals\n    subject['AccML_filtered'] = lowpass_filter(subject['AccML'], cutoff_freq, sampling_freq_tdcsfog)\n    subject['AccAP_filtered'] = lowpass_filter(subject['AccAP'], cutoff_freq, sampling_freq_tdcsfog)\n    subject['AccV_filtered'] = lowpass_filter(subject['AccV'], cutoff_freq, sampling_freq_tdcsfog)\n\n    \n    # Extract features for the subject\n    max_coeff, max_freq = morlet_wavelet_features(\n        subject['AccML_filtered'],\n        min_scale_log=3.6, \n        max_scale_log=5.4, \n        sampling_period=sampling_period_tdcsfog)\n    \n    # Prepare a DataFrame to store the results for this subject\n    subject_results = pd.DataFrame({\n        'AccML_filtered': subject['AccML_filtered'],\n        'AccAP_filtered': subject['AccAP_filtered'],\n        'AccV_filtered': subject['AccV_filtered'],\n        'max_coeff': max_coeff,\n        'freq': max_freq,\n    })\n    \n    tdcsfog_results = pd.concat([tdcsfog_results, subject_results], ignore_index=True)\n\nprint(\"Feature extraction completed and saved.\")","metadata":{"execution":{"iopub.status.busy":"2024-08-24T11:38:49.779750Z","iopub.execute_input":"2024-08-24T11:38:49.780172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize an empty DataFrame to store the results\ndefog_results = pd.DataFrame()\n\n# Process each subject in the dataset\nfor file_name in defog['file_name'].unique():\n    subject = defog[defog['file_name'] == file_name].copy()\n    \n    # Apply lowpass filter to all three signals\n    subject['AccML_filtered'] = lowpass_filter(subject['AccML'], cutoff_freq, sampling_freq_tdcsfog)\n    subject['AccAP_filtered'] = lowpass_filter(subject['AccAP'], cutoff_freq, sampling_freq_tdcsfog)\n    subject['AccV_filtered'] = lowpass_filter(subject['AccV'], cutoff_freq, sampling_freq_tdcsfog)\n    \n    # Extract features for the subject\n    max_coeff, max_freq = morlet_wavelet_features(\n        subject['AccML_filtered'],\n        min_scale_log=3.6, \n        max_scale_log=5.4, \n        sampling_period=sampling_period_defog)\n    \n    # Prepare a DataFrame to store the results for this subject\n    subject_results = pd.DataFrame({\n        'AccML_filtered': subject['AccML_filtered'],\n        'AccAP_filtered': subject['AccAP_filtered'],\n        'AccV_filtered': subject['AccV_filtered'],\n        'max_coeff': max_coeff,\n        'freq': max_freq,\n    })\n    \n    defog_results = pd.concat([defog_results, subject_results], ignore_index=True)\n\nprint(\"Feature extraction completed and saved.\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_results = tdcsfog_results[['AccML_filtered', 'AccAP_filtered', 'AccV_filtered', 'max_coeff','freq']]\ndefog_results = defog_results[['AccML_filtered', 'AccAP_filtered', 'AccV_filtered','max_coeff','freq']]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog = pd.concat([defog, defog_results], axis=1)\n\ntdcsfog =pd.concat([tdcsfog, tdcsfog_results], axis=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_rolling_window_features(data, window_size=200, feature_columns=['AccV', 'AccML', 'AccAP']):\n    \"\"\"\n    Add rolling window features to the dataset.\n\n    Parameters:\n    - data: The pandas DataFrame to which the features will be added.\n    - window_size: The size of the rolling window. 2 seconds window for 100Hz sampling rate\n    - feature_columns: The columns to calculate the rolling features for.\n\n    Returns:\n    - The pandas DataFrame with the new rolling window features added.\n    \"\"\"\n    for axis in feature_columns:\n        data[f'{axis}_rolling_mean'] = data[axis].rolling(window=window_size, min_periods=1).mean()\n        data[f'{axis}_rolling_std'] = data[axis].rolling(window=window_size, min_periods=1).std()\n        data[f'{axis}_rolling_max'] = data[axis].rolling(window=window_size, min_periods=1).max()\n        data[f'{axis}_rolling_min'] = data[axis].rolling(window=window_size, min_periods=1).min()\n    \n    data.dropna(inplace=True)\n    return data","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:18:52.817999Z","iopub.execute_input":"2024-05-27T14:18:52.819104Z","iopub.status.idle":"2024-05-27T14:18:52.827353Z","shell.execute_reply.started":"2024-05-27T14:18:52.819072Z","shell.execute_reply":"2024-05-27T14:18:52.826349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog = add_rolling_window_features(tdcsfog)\ndefog = add_rolling_window_features(defog)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:18:52.82883Z","iopub.execute_input":"2024-05-27T14:18:52.829282Z","iopub.status.idle":"2024-05-27T14:19:08.167773Z","shell.execute_reply.started":"2024-05-27T14:18:52.829243Z","shell.execute_reply":"2024-05-27T14:19:08.166716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:19:08.169083Z","iopub.execute_input":"2024-05-27T14:19:08.169394Z","iopub.status.idle":"2024-05-27T14:19:12.308825Z","shell.execute_reply.started":"2024-05-27T14:19:08.169369Z","shell.execute_reply":"2024-05-27T14:19:12.307594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract unique file names\nunique_files_tdcsfog = tdcsfog['subject'].unique()\n\n# Calculate the split index\nsplit_index_tdcsfog  = int(0.8 * len(unique_files_tdcsfog))\n\n# Split the file names into training and test sets\ntrain_files_tdcsfog = unique_files_tdcsfog[:split_index_tdcsfog]\ntest_files_tdcsfog = unique_files_tdcsfog[split_index_tdcsfog:]\n\n# Filter the DataFrame for training and test sets\ntrain_data_tdcsfog = tdcsfog[tdcsfog['subject'].isin(train_files_tdcsfog)]\ntest_data_tdcsfog = tdcsfog[tdcsfog['subject'].isin(test_files_tdcsfog)]\n\n# Print the number of rows in each set to verify\nprint(f'Train data rows: {len(train_data_tdcsfog)}')\nprint(f'Test data rows: {len(test_data_tdcsfog)}')\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:19:12.310516Z","iopub.execute_input":"2024-05-27T14:19:12.311537Z","iopub.status.idle":"2024-05-27T14:19:14.852179Z","shell.execute_reply.started":"2024-05-27T14:19:12.311355Z","shell.execute_reply":"2024-05-27T14:19:14.851029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract unique file names\nunique_files_defog = defog['subject'].unique()\n\n# Calculate the split index\nsplit_index_defog  = int(0.8 * len(unique_files_defog))\n\n# Split the file names into training and test sets\ntrain_files_defog = unique_files_defog[:split_index_defog]\ntest_files_defog = unique_files_defog[split_index_defog:]\n\n# Filter the DataFrame for training and test sets\ntrain_data_defog = defog[defog['subject'].isin(train_files_defog)]\ntest_data_defog = defog[defog['subject'].isin(test_files_defog)]\n\n# Print the number of rows in each set to verify\nprint(f'Train data rows: {len(train_data_defog)}')\nprint(f'Test data rows: {len(test_data_defog)}')\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:19:14.853685Z","iopub.execute_input":"2024-05-27T14:19:14.854387Z","iopub.status.idle":"2024-05-27T14:19:20.124313Z","shell.execute_reply.started":"2024-05-27T14:19:14.854354Z","shell.execute_reply":"2024-05-27T14:19:20.123198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting\nplt.figure(figsize=(10, 6))\n\nplt.plot(train_data_tdcsfog['AccV'], label='AccV')\nplt.plot(train_data_tdcsfog['AccML'], label='AccML')\nplt.plot(train_data_tdcsfog['AccAP'], label='AccAP')\n\nplt.title('Event Progression Over Time')\nplt.xlabel('Time')\nplt.ylabel('Acc')\nplt.legend()\nplt.grid(True)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:19:20.128696Z","iopub.execute_input":"2024-05-27T14:19:20.129316Z","iopub.status.idle":"2024-05-27T14:19:44.639307Z","shell.execute_reply.started":"2024-05-27T14:19:20.129283Z","shell.execute_reply":"2024-05-27T14:19:44.637543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting\nplt.figure(figsize=(10, 6))\n\nplt.plot(test_data_tdcsfog['AccV'], label='AccV')\nplt.plot(test_data_tdcsfog['AccML'], label='AccML')\nplt.plot(test_data_tdcsfog['AccAP'], label='AccAP')\n\nplt.title('Event Progression Over Time')\nplt.xlabel('Time')\nplt.ylabel('Acc')\nplt.legend()\nplt.grid(True)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:19:44.641782Z","iopub.execute_input":"2024-05-27T14:19:44.642806Z","iopub.status.idle":"2024-05-27T14:19:50.580647Z","shell.execute_reply.started":"2024-05-27T14:19:44.642761Z","shell.execute_reply":"2024-05-27T14:19:50.579369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_tdcsfog","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:19:50.582761Z","iopub.execute_input":"2024-05-27T14:19:50.583223Z","iopub.status.idle":"2024-05-27T14:19:54.42402Z","shell.execute_reply.started":"2024-05-27T14:19:50.583186Z","shell.execute_reply":"2024-05-27T14:19:54.422912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming 'train_data' and 'test_data' are pandas DataFrames\ntrain_features_tdcsfog = train_data_tdcsfog[['Time','AccV', 'AccML', 'AccAP', \n                   'AccV_rolling_mean', 'AccV_rolling_std', 'AccV_rolling_max', 'AccV_rolling_min',\n                   'AccML_rolling_mean', 'AccML_rolling_std', 'AccML_rolling_max', 'AccML_rolling_min',\n                   'AccAP_rolling_mean', 'AccAP_rolling_std', 'AccAP_rolling_max', 'AccAP_rolling_min']]\ntest_features_tdcsfog = test_data_tdcsfog[['Time','AccV', 'AccML', 'AccAP', \n                   'AccV_rolling_mean', 'AccV_rolling_std', 'AccV_rolling_max', 'AccV_rolling_min',\n                   'AccML_rolling_mean', 'AccML_rolling_std', 'AccML_rolling_max', 'AccML_rolling_min',\n                   'AccAP_rolling_mean', 'AccAP_rolling_std', 'AccAP_rolling_max', 'AccAP_rolling_min']]\n\n# Select the label column\ntrain_labels_tdcsfog = train_data_tdcsfog['IsFOG']\ntest_labels_tdcsfog = test_data_tdcsfog['IsFOG']\n\n# Create LightGBM datasets\ntrain_dataset_tdcsfog = lgb.Dataset(train_features_tdcsfog, label=train_labels_tdcsfog)\ntest_dataset_tdcsfog = lgb.Dataset(test_features_tdcsfog, label=test_labels_tdcsfog)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:19:54.42574Z","iopub.execute_input":"2024-05-27T14:19:54.426136Z","iopub.status.idle":"2024-05-27T14:19:54.901796Z","shell.execute_reply.started":"2024-05-27T14:19:54.426108Z","shell.execute_reply":"2024-05-27T14:19:54.900832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming 'train_data' and 'test_data' are pandas DataFrames\ntrain_features_defog = train_data_defog[['Time','AccV', 'AccML', 'AccAP', \n                   'AccV_rolling_mean', 'AccV_rolling_std', 'AccV_rolling_max', 'AccV_rolling_min',\n                   'AccML_rolling_mean', 'AccML_rolling_std', 'AccML_rolling_max', 'AccML_rolling_min',\n                   'AccAP_rolling_mean', 'AccAP_rolling_std', 'AccAP_rolling_max', 'AccAP_rolling_min']]\ntest_features_defog = test_data_defog[['Time','AccV', 'AccML', 'AccAP', \n                   'AccV_rolling_mean', 'AccV_rolling_std', 'AccV_rolling_max', 'AccV_rolling_min',\n                   'AccML_rolling_mean', 'AccML_rolling_std', 'AccML_rolling_max', 'AccML_rolling_min',\n                   'AccAP_rolling_mean', 'AccAP_rolling_std', 'AccAP_rolling_max', 'AccAP_rolling_min']]\n\n# Select the label column\ntrain_labels_defog = train_data_defog['IsFOG']\ntest_labels_defog = test_data_defog['IsFOG']\n\n# Create LightGBM datasets\ntrain_dataset_defog = lgb.Dataset(train_features_defog, label=train_labels_defog)\ntest_dataset_defog = lgb.Dataset(test_features_defog, label=test_labels_defog)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:19:54.90315Z","iopub.execute_input":"2024-05-27T14:19:54.903483Z","iopub.status.idle":"2024-05-27T14:19:55.799936Z","shell.execute_reply.started":"2024-05-27T14:19:54.903455Z","shell.execute_reply":"2024-05-27T14:19:55.798939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting\nplt.figure(figsize=(10, 6))\n\nplt.plot(train_features_tdcsfog['AccV'], label='AccV')\nplt.plot(train_features_tdcsfog['AccML'], label='AccML')\nplt.plot(train_features_tdcsfog['AccAP'], label='AccAP')\n\nplt.title('Event Progression Over Time')\nplt.xlabel('Time')\nplt.ylabel('Acc')\nplt.legend()\nplt.grid(True)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:19:55.801205Z","iopub.execute_input":"2024-05-27T14:19:55.801534Z","iopub.status.idle":"2024-05-27T14:20:17.686969Z","shell.execute_reply.started":"2024-05-27T14:19:55.801508Z","shell.execute_reply":"2024-05-27T14:20:17.685878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting\nplt.figure(figsize=(10, 6))\n\nplt.plot(test_features_tdcsfog['AccV'], label='AccV')\nplt.plot(test_features_tdcsfog['AccML'], label='AccML')\nplt.plot(test_features_tdcsfog['AccAP'], label='AccAP')\n\nplt.title('Event Progression Over Time')\nplt.xlabel('Time')\nplt.ylabel('Acc')\nplt.legend()\nplt.grid(True)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:20:17.688298Z","iopub.execute_input":"2024-05-27T14:20:17.688646Z","iopub.status.idle":"2024-05-27T14:20:23.562155Z","shell.execute_reply.started":"2024-05-27T14:20:17.688612Z","shell.execute_reply":"2024-05-27T14:20:23.561032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fog_params={\n    'objective': 'binary', #binary target feature\n    'metric': 'average_precision', \n    'boosting_type': 'gbdt',  #GradientBoostingDecisionTree\n    'learning_rate': 0.18,\n    'verbose': 1,\n    'max_depth': 10,\n    'num_leaves': 80,\n    'is_unbalance':True\n}\n\n# Train the LightGBM model\nnum_round = 200  \n\n# Train the model\nfog_model_tdcsfog = lgb.train(fog_params, train_dataset_tdcsfog, num_round, valid_sets=[test_dataset_tdcsfog])\n\n# Make predictions\ny_pred_tdcsfog = fog_model_tdcsfog.predict(test_features_tdcsfog, num_iteration=fog_model_tdcsfog.best_iteration)\n\n# Convert probabilities to binary predictions\ny_pred_binary_tdcsfog = (y_pred_tdcsfog > 0.5).astype(int)\n\n# Evaluate the model\naccuracy = metrics.accuracy_score(test_labels_tdcsfog, y_pred_binary_tdcsfog)\nprint(f\"Accuracy: {accuracy}\")","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:20:23.563584Z","iopub.execute_input":"2024-05-27T14:20:23.563954Z","iopub.status.idle":"2024-05-27T14:22:17.122413Z","shell.execute_reply.started":"2024-05-27T14:20:23.563922Z","shell.execute_reply":"2024-05-27T14:22:17.119632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fog_params={\n    'objective': 'binary', #binary target feature\n    'metric': 'average_precision', \n    'boosting_type': 'gbdt',  #GradientBoostingDecisionTree\n    'learning_rate': 0.18,\n    'verbose': 1,\n    'max_depth': 10,\n    'num_leaves': 80,\n    'is_unbalance':True\n}\n# Train the LightGBM model\nnum_round = 200  \n\n# Train the model\nfog_model_defog = lgb.train(fog_params, train_dataset_defog, num_round, valid_sets=[test_dataset_defog])\n\n# Make predictions\ny_pred_defog = fog_model_defog.predict(test_features_defog, num_iteration=fog_model_defog.best_iteration)\n\n# Convert probabilities to binary predictions\ny_pred_binary_defog = (y_pred_defog > 0.5).astype(int)\n\n# Evaluate the model\naccuracy_defog = metrics.accuracy_score(test_labels_defog, y_pred_binary_defog)\nprint(f\"Accuracy: {accuracy_defog}\")","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:22:17.125095Z","iopub.execute_input":"2024-05-27T14:22:17.126099Z","iopub.status.idle":"2024-05-27T14:25:38.090449Z","shell.execute_reply.started":"2024-05-27T14:22:17.126034Z","shell.execute_reply":"2024-05-27T14:25:38.089064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = [\"StartHesitation\", \"Turn\", 'Walking', 'IsFOG', 'file_name','subject']","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:25:38.092489Z","iopub.execute_input":"2024-05-27T14:25:38.092946Z","iopub.status.idle":"2024-05-27T14:25:38.09872Z","shell.execute_reply.started":"2024-05-27T14:25:38.092909Z","shell.execute_reply":"2024-05-27T14:25:38.097605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"CLASSIFICATION","metadata":{}},{"cell_type":"code","source":"train_data_tdcsfog","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:25:38.100201Z","iopub.execute_input":"2024-05-27T14:25:38.100531Z","iopub.status.idle":"2024-05-27T14:25:41.612658Z","shell.execute_reply.started":"2024-05-27T14:25:38.100505Z","shell.execute_reply":"2024-05-27T14:25:41.611497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_tdcsfog_cl = train_data_tdcsfog[train_data_tdcsfog['IsFOG'] == True]\ntest_data_tdcsfog_cl = test_data_tdcsfog[test_data_tdcsfog['IsFOG'] == True]","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:25:41.613977Z","iopub.execute_input":"2024-05-27T14:25:41.614284Z","iopub.status.idle":"2024-05-27T14:25:41.947438Z","shell.execute_reply.started":"2024-05-27T14:25:41.614258Z","shell.execute_reply":"2024-05-27T14:25:41.946207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_tdcsfog_cl","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:38:08.875493Z","iopub.execute_input":"2024-05-27T14:38:08.875973Z","iopub.status.idle":"2024-05-27T14:38:09.748404Z","shell.execute_reply.started":"2024-05-27T14:38:08.875938Z","shell.execute_reply":"2024-05-27T14:38:09.747414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_defog_cl = train_data_defog[train_data_defog['IsFOG'] == True]\ntest_data_defog_cl = test_data_defog[test_data_defog['IsFOG'] == True]","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:25:41.951414Z","iopub.execute_input":"2024-05-27T14:25:41.952092Z","iopub.status.idle":"2024-05-27T14:25:42.083141Z","shell.execute_reply.started":"2024-05-27T14:25:41.952061Z","shell.execute_reply":"2024-05-27T14:25:42.082086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X values for on which the model is trained\nX_train_tdcsfog = train_data_tdcsfog_cl.drop(targets, axis=1)  # features\n\n# split data into the three \ny1_train_tdcsfog = train_data_tdcsfog_cl[['StartHesitation', 'Turn','Walking']]\n# X values for on which the model is trained\nX_test_tdcsfog = test_data_tdcsfog_cl.drop(targets, axis=1)  # features\n\n# split data into the three \ny1_test_tdcsfog = test_data_tdcsfog_cl[['StartHesitation', 'Turn','Walking']]","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:25:42.084592Z","iopub.execute_input":"2024-05-27T14:25:42.084977Z","iopub.status.idle":"2024-05-27T14:25:42.290064Z","shell.execute_reply.started":"2024-05-27T14:25:42.084947Z","shell.execute_reply":"2024-05-27T14:25:42.2889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X values for on which the model is trained\nX_train_defog = train_data_defog_cl.drop(targets, axis=1)  # features\n\n# split data into the three \ny1_train_defog = train_data_defog_cl[['StartHesitation', 'Turn','Walking']]\n# X values for on which the model is trained\nX_test_defog = test_data_defog_cl.drop(targets, axis=1)  # features\n\n# split data into the three \ny1_test_defog = test_data_defog_cl[['StartHesitation', 'Turn','Walking']]","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:25:42.2914Z","iopub.execute_input":"2024-05-27T14:25:42.291737Z","iopub.status.idle":"2024-05-27T14:25:42.377907Z","shell.execute_reply.started":"2024-05-27T14:25:42.291705Z","shell.execute_reply":"2024-05-27T14:25:42.37691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting\nplt.figure(figsize=(10, 6))\n\nplt.plot(X_train_tdcsfog['AccV'], label='AccV')\nplt.plot(X_train_tdcsfog['AccML'], label='AccML')\nplt.plot(X_train_tdcsfog['AccAP'], label='AccAP')\n\nplt.title('Event Progression Over Time')\nplt.xlabel('Time')\nplt.ylabel('Acc')\nplt.legend()\nplt.grid(True)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:25:42.379471Z","iopub.execute_input":"2024-05-27T14:25:42.379932Z","iopub.status.idle":"2024-05-27T14:25:46.861278Z","shell.execute_reply.started":"2024-05-27T14:25:42.379891Z","shell.execute_reply":"2024-05-27T14:25:46.860203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from imblearn.over_sampling import SMOTE\ndef apply_smote(X, y):\n    smote = SMOTE(random_state=42)\n    X_smote, y_smote = smote.fit_resample(X, y)\n    return X_smote, y_smote","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:25:46.862786Z","iopub.execute_input":"2024-05-27T14:25:46.863222Z","iopub.status.idle":"2024-05-27T14:25:47.262046Z","shell.execute_reply.started":"2024-05-27T14:25:46.863184Z","shell.execute_reply":"2024-05-27T14:25:47.260771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_tdcsfog = X_train_tdcsfog.values\ny1_train_tdcsfog = y1_train_tdcsfog.values\n\nX_test_tdcsfog = X_test_tdcsfog.values\ny1_test_tdcsfog = y1_test_tdcsfog.values\n\nX_train_defog = X_train_defog.values\ny1_train_defog = y1_train_defog.values\n\nX_test_defog = X_test_defog.values\ny1_test_defog = y1_test_defog.values","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:25:47.263695Z","iopub.execute_input":"2024-05-27T14:25:47.264498Z","iopub.status.idle":"2024-05-27T14:25:48.077348Z","shell.execute_reply.started":"2024-05-27T14:25:47.264454Z","shell.execute_reply":"2024-05-27T14:25:48.075896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_tdcsfog, y1_train_tdcsfog = apply_smote(X_train_tdcsfog, y1_train_tdcsfog)\nX_test_tdcsfog, y1_test_tdcsfog = apply_smote(X_test_tdcsfog, y1_test_tdcsfog)\nX_train_defog, y1_train_defog = apply_smote(X_train_defog, y1_train_defog)\nX_test_defog, y1_test_defog = apply_smote(X_test_defog, y1_test_defog)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:25:48.079053Z","iopub.execute_input":"2024-05-27T14:25:48.079507Z","iopub.status.idle":"2024-05-27T14:34:30.195701Z","shell.execute_reply.started":"2024-05-27T14:25:48.079465Z","shell.execute_reply":"2024-05-27T14:34:30.19451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Plot the AccV column from the numpy arrays and the DataFrame\nplt.figure(figsize=(10, 6))  # Optional: to make the plot larger\n\n# Plotting the AccV column from the numpy arrays\nplt.plot(X_train_tdcsfog[:10000, 1], label='X_train_tdcsfog AccV')\nplt.plot(X_test_tdcsfog[:10000, 1], label='X_test_tdcsfog AccV')\n\n# Plotting the AccV column from the DataFrame\nplt.plot(train_data_tdcsfog['AccV'][:10000], label='train_data_tdcsfog AccV')\n\n# Adding titles and labels\nplt.title('AccV Column from Different Datasets')\nplt.xlabel('Sample Index')\nplt.ylabel('AccV Value')\n\n# Adding a legend\nplt.legend()\n\n# Show the plot\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:34:30.197174Z","iopub.execute_input":"2024-05-27T14:34:30.197523Z","iopub.status.idle":"2024-05-27T14:34:30.594569Z","shell.execute_reply.started":"2024-05-27T14:34:30.197492Z","shell.execute_reply":"2024-05-27T14:34:30.592919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y1_train_tdcsfog = pd.DataFrame(y1_train_tdcsfog)\ny1_test_tdcsfog = pd.DataFrame(y1_test_tdcsfog)\ny1_train_defog = pd.DataFrame(y1_train_defog)\ny1_test_defog = pd.DataFrame(y1_test_defog)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:34:30.596232Z","iopub.execute_input":"2024-05-27T14:34:30.596648Z","iopub.status.idle":"2024-05-27T14:34:30.606551Z","shell.execute_reply.started":"2024-05-27T14:34:30.596604Z","shell.execute_reply":"2024-05-27T14:34:30.605073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y1_train_tdcsfog","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:37:04.364135Z","iopub.execute_input":"2024-05-27T14:37:04.365224Z","iopub.status.idle":"2024-05-27T14:37:04.378376Z","shell.execute_reply.started":"2024-05-27T14:37:04.365179Z","shell.execute_reply":"2024-05-27T14:37:04.377143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_StartHesitation_tdcsfog_train = y1_train_tdcsfog[0]\ny_Turn_tdcsfog_train = y1_train_tdcsfog[1]\ny_Walking_tdcsfog_train = y1_train_tdcsfog[2]\n\ny_StartHesitation_tdcsfog_test = y1_test_tdcsfog[0]\ny_Turn_tdcsfog_test = y1_test_tdcsfog[1]\ny_Walking_tdcsfog_test = y1_test_tdcsfog[2]","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:39:52.213055Z","iopub.execute_input":"2024-05-27T14:39:52.213476Z","iopub.status.idle":"2024-05-27T14:39:52.220109Z","shell.execute_reply.started":"2024-05-27T14:39:52.213443Z","shell.execute_reply":"2024-05-27T14:39:52.218799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_StartHesitation_defog_train = y1_train_defog[0]\ny_Turn_defog_train = y1_train_defog[1]\ny_Walking_defog_train = y1_train_defog[2]","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:40:19.663503Z","iopub.execute_input":"2024-05-27T14:40:19.664179Z","iopub.status.idle":"2024-05-27T14:40:19.670662Z","shell.execute_reply.started":"2024-05-27T14:40:19.664134Z","shell.execute_reply":"2024-05-27T14:40:19.669145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_StartHesitation_defog_test = y1_test_defog[0]\ny_Turn_defog_test = y1_test_defog[1]\ny_Walking_defog_test = y1_test_defog[2]","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:40:32.111754Z","iopub.execute_input":"2024-05-27T14:40:32.112155Z","iopub.status.idle":"2024-05-27T14:40:32.118057Z","shell.execute_reply.started":"2024-05-27T14:40:32.112125Z","shell.execute_reply":"2024-05-27T14:40:32.116676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create LightGBM datasets\ntrain_dataset_defog_SH = lgb.Dataset(X_train_defog, label=y_StartHesitation_defog_train)\ntest_dataset_defog_SH = lgb.Dataset(X_test_defog, label=y_StartHesitation_defog_test)\n\n# Create LightGBM datasets\ntrain_dataset_defog_T = lgb.Dataset(X_train_defog, label=y_Turn_defog_train)\ntest_dataset_defog_T = lgb.Dataset(X_test_defog, label=y_Turn_defog_test)\n\n# Create LightGBM datasets\ntrain_dataset_defog_W = lgb.Dataset(X_train_defog, label=y_Walking_defog_train)\ntest_dataset_defog_W = lgb.Dataset(X_test_defog, label=y_Walking_defog_test)\n\n# Create LightGBM datasets\ntrain_dataset_tdcsfog_SH = lgb.Dataset(X_train_tdcsfog, label=y_StartHesitation_tdcsfog_train)\ntest_dataset_tdcsfog_SH = lgb.Dataset(X_test_tdcsfog, label=y_StartHesitation_tdcsfog_test)\n\n# Create LightGBM datasets\ntrain_dataset_tdcsfog_T = lgb.Dataset(X_train_tdcsfog, label=y_Turn_tdcsfog_train)\ntest_dataset_tdcsfog_T = lgb.Dataset(X_test_tdcsfog, label=y_Turn_tdcsfog_test)\n\n# Create LightGBM datasets\ntrain_dataset_tdcsfog_W = lgb.Dataset(X_train_tdcsfog, label=y_Walking_tdcsfog_train)\ntest_dataset_tdcsfog_W = lgb.Dataset(X_test_tdcsfog, label=y_Walking_tdcsfog_test)\n\n# Define XGBoost parameters\nxgboost_params = {\n    'objective': 'binary',\n    'metric': 'average_precision',\n    'boosting_type': 'gbdt',\n    'learning_rate': 0.03,\n    'verbose': 1,\n    'max_depth': 6,\n    'num_leaves': 50\n}\nnum_round = 300  ","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:40:34.08777Z","iopub.execute_input":"2024-05-27T14:40:34.088559Z","iopub.status.idle":"2024-05-27T14:40:34.097097Z","shell.execute_reply.started":"2024-05-27T14:40:34.088524Z","shell.execute_reply":"2024-05-27T14:40:34.095838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the LightGBM model\nmodel1_defog_SH = lgb.train(xgboost_params, train_dataset_defog_SH, num_round, valid_sets=[test_dataset_defog_SH])","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:40:41.185129Z","iopub.execute_input":"2024-05-27T14:40:41.186507Z","iopub.status.idle":"2024-05-27T14:41:20.665132Z","shell.execute_reply.started":"2024-05-27T14:40:41.186423Z","shell.execute_reply":"2024-05-27T14:41:20.659839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the LightGBM model\nmodel2_defog_T = lgb.train(xgboost_params, train_dataset_defog_T, num_round, valid_sets=[test_dataset_defog_T])","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:43:10.267648Z","iopub.execute_input":"2024-05-27T14:43:10.268257Z","iopub.status.idle":"2024-05-27T14:43:53.051439Z","shell.execute_reply.started":"2024-05-27T14:43:10.268215Z","shell.execute_reply":"2024-05-27T14:43:53.049988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the LightGBM model\nmodel3_defog_W = lgb.train(xgboost_params, train_dataset_defog_W, num_round, valid_sets=[test_dataset_defog_W])","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:44:51.269646Z","iopub.execute_input":"2024-05-27T14:44:51.270727Z","iopub.status.idle":"2024-05-27T14:45:34.443046Z","shell.execute_reply.started":"2024-05-27T14:44:51.270678Z","shell.execute_reply":"2024-05-27T14:45:34.441817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the LightGBM model\nmodel4_tdcsfog_SH = lgb.train(xgboost_params, train_dataset_tdcsfog_SH, num_round, valid_sets=[test_dataset_tdcsfog_SH])","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:46:13.347526Z","iopub.execute_input":"2024-05-27T14:46:13.348085Z","iopub.status.idle":"2024-05-27T14:48:20.433593Z","shell.execute_reply.started":"2024-05-27T14:46:13.348044Z","shell.execute_reply":"2024-05-27T14:48:20.432436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the LightGBM model\nmodel5_tdcsfog_T = lgb.train(xgboost_params, train_dataset_tdcsfog_T, num_round, valid_sets=[test_dataset_tdcsfog_T])","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:48:29.555633Z","iopub.execute_input":"2024-05-27T14:48:29.556908Z","iopub.status.idle":"2024-05-27T14:50:28.941998Z","shell.execute_reply.started":"2024-05-27T14:48:29.55684Z","shell.execute_reply":"2024-05-27T14:50:28.940749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the LightGBM model\nmodel6_tdcsfog_W = lgb.train(xgboost_params, train_dataset_tdcsfog_W, num_round, valid_sets=[test_dataset_tdcsfog_W])","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:50:49.421812Z","iopub.execute_input":"2024-05-27T14:50:49.422789Z","iopub.status.idle":"2024-05-27T14:52:47.495336Z","shell.execute_reply.started":"2024-05-27T14:50:49.422749Z","shell.execute_reply":"2024-05-27T14:52:47.49382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_test_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog'\n        \ntdcsfog_test_list = [\n    pd.read_csv(os.path.join(tdcsfog_test_path, file_name)).assign(Id=lambda df: file_name[:-4] + '_' + df['Time'].astype(str))\n    for file_name in os.listdir(tdcsfog_test_path)\n    if file_name.endswith('.csv')\n]        \n\ntdcsfog_test = pd.concat(tdcsfog_test_list, axis = 0)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:53:02.059146Z","iopub.execute_input":"2024-05-27T14:53:02.059999Z","iopub.status.idle":"2024-05-27T14:53:02.097121Z","shell.execute_reply.started":"2024-05-27T14:53:02.059958Z","shell.execute_reply":"2024-05-27T14:53:02.095594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_test_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog'\n\ndefog_test_list = [\n    pd.read_csv(os.path.join(defog_test_path, file_name)).assign(Id=lambda df: file_name[:-4] + '_' + df['Time'].astype(str))\n    for file_name in os.listdir(defog_test_path)\n    if file_name.endswith('.csv')\n]        \n\ndefog_test = pd.concat(defog_test_list, axis = 0)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:53:04.121078Z","iopub.execute_input":"2024-05-27T14:53:04.121582Z","iopub.status.idle":"2024-05-27T14:53:04.703272Z","shell.execute_reply.started":"2024-05-27T14:53:04.121546Z","shell.execute_reply":"2024-05-27T14:53:04.70214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize an empty DataFrame to store the results for tdcsfog_test\ntdcsfog_test_results = pd.DataFrame()\n\n# Process each subject in the tdcsfog_test dataset\nfor file_name_prefix in tdcsfog_test['Id'].str.split('_', expand=True)[0].unique():\n    # Filter data for the current file_name prefix\n    subject = tdcsfog_test[tdcsfog_test['Id'].str.startswith(file_name_prefix)].copy()\n    \n    # Apply lowpass filter to all three signals\n    subject['AccML_filtered'] = lowpass_filter(subject['AccML'], cutoff_freq, sampling_freq_tdcsfog)\n    subject['AccAP_filtered'] = lowpass_filter(subject['AccAP'], cutoff_freq, sampling_freq_tdcsfog)\n    subject['AccV_filtered'] = lowpass_filter(subject['AccV'], cutoff_freq, sampling_freq_tdcsfog)\n\n    # Extract features for the subject\n    max_coeff, max_freq = morlet_wavelet_features(\n        subject['AccML_filtered'],\n        min_scale_log=3.6, \n        max_scale_log=5.4, \n        sampling_period=sampling_period_tdcsfog)\n    \n    # Prepare a DataFrame to store the results for this subject\n    subject_results = pd.DataFrame({\n        'AccML_filtered': subject['AccML_filtered'],\n        'AccAP_filtered': subject['AccAP_filtered'],\n        'AccV_filtered': subject['AccV_filtered'],\n        'max_coeff': max_coeff,\n        'freq': max_freq,\n    })\n    \n\n    tdcsfog_test_results = pd.concat([tdcsfog_test_results, subject_results], ignore_index=True)\n\nprint(\"Feature extraction completed and saved for tdcsfog_test.\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_test = pd.concat([tdcsfog_test, tdcsfog_test_results], axis=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize an empty DataFrame to store the results for defog_test\ndefog_test_results = pd.DataFrame()\n\n# Process each subject in the defog_test dataset\nfor file_name_prefix in defog_test['Id'].str.split('_', expand=True)[0].unique():\n    # Filter data for the current file_name prefix\n    subject = defog_test[defog_test['Id'].str.startswith(file_name_prefix)].copy()\n    \n    # Apply lowpass filter to all three signals\n    subject['AccML_filtered'] = lowpass_filter(subject['AccML'], cutoff_freq, sampling_freq_defog)\n    subject['AccAP_filtered'] = lowpass_filter(subject['AccAP'], cutoff_freq, sampling_freq_defog)\n    subject['AccV_filtered'] = lowpass_filter(subject['AccV'], cutoff_freq, sampling_freq_defog)\n\n    # Extract features for the subject\n    max_coeff, max_freq = morlet_wavelet_features(\n        subject['AccML_filtered'],\n        min_scale_log=3.6, \n        max_scale_log=5.4, \n        sampling_period=sampling_period_defog)\n    \n    # Prepare a DataFrame to store the results for this subject\n    subject_results = pd.DataFrame({\n        'AccML_filtered': subject['AccML_filtered'],\n        'AccAP_filtered': subject['AccAP_filtered'],\n        'AccV_filtered': subject['AccV_filtered'],\n        'max_coeff': max_coeff,\n        'freq': max_freq,\n    })\n    \n    defog_test_results = pd.concat([defog_test_results, subject_results], ignore_index=True)\n\nprint(\"Feature extraction completed and saved for defog_test.\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_test = pd.concat([defog_test, defog_test_results], axis=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"window_size = 200  # 2 seconds window for 100Hz sampling rate\n\n# Calculating rolling window features for each acceleration axis\nfor axis in ['AccV', 'AccML', 'AccAP']:\n    defog_test[f'{axis}_rolling_mean'] = defog_test[axis].rolling(window=window_size, min_periods=1).mean()\n    defog_test[f'{axis}_rolling_std'] = defog_test[axis].rolling(window=window_size, min_periods=1).std()\n    defog_test[f'{axis}_rolling_max'] = defog_test[axis].rolling(window=window_size, min_periods=1).max()\n    defog_test[f'{axis}_rolling_min'] = defog_test[axis].rolling(window=window_size, min_periods=1).min()","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:53:06.414597Z","iopub.execute_input":"2024-05-27T14:53:06.415718Z","iopub.status.idle":"2024-05-27T14:53:06.556258Z","shell.execute_reply.started":"2024-05-27T14:53:06.415669Z","shell.execute_reply":"2024-05-27T14:53:06.554915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculating rolling window features for each acceleration axis\nfor axis in ['AccV', 'AccML', 'AccAP']:\n    tdcsfog_test[f'{axis}_rolling_mean'] = tdcsfog_test[axis].rolling(window=window_size, min_periods=1).mean()\n    tdcsfog_test[f'{axis}_rolling_std'] = tdcsfog_test[axis].rolling(window=window_size, min_periods=1).std()\n    tdcsfog_test[f'{axis}_rolling_max'] = tdcsfog_test[axis].rolling(window=window_size, min_periods=1).max()\n    tdcsfog_test[f'{axis}_rolling_min'] = tdcsfog_test[axis].rolling(window=window_size, min_periods=1).min()","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:53:08.483644Z","iopub.execute_input":"2024-05-27T14:53:08.484128Z","iopub.status.idle":"2024-05-27T14:53:08.503579Z","shell.execute_reply.started":"2024-05-27T14:53:08.484093Z","shell.execute_reply":"2024-05-27T14:53:08.502318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Updating feature selection with rolling window features\nfeature_columns = ['Time','AccV', 'AccML', 'AccAP', \n                   'AccV_rolling_mean', 'AccV_rolling_std', 'AccV_rolling_max', 'AccV_rolling_min',\n                   'AccML_rolling_mean', 'AccML_rolling_std', 'AccML_rolling_max', 'AccML_rolling_min',\n                   'AccAP_rolling_mean', 'AccAP_rolling_std', 'AccAP_rolling_max', 'AccAP_rolling_min']\n\nX_defog = defog_test[feature_columns]","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:53:10.74715Z","iopub.execute_input":"2024-05-27T14:53:10.748166Z","iopub.status.idle":"2024-05-27T14:53:10.770039Z","shell.execute_reply.started":"2024-05-27T14:53:10.748121Z","shell.execute_reply":"2024-05-27T14:53:10.768556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Updating feature selection with rolling window features\nfeature_columns = ['Time','AccV', 'AccML', 'AccAP', \n                   'AccV_rolling_mean', 'AccV_rolling_std', 'AccV_rolling_max', 'AccV_rolling_min',\n                   'AccML_rolling_mean', 'AccML_rolling_std', 'AccML_rolling_max', 'AccML_rolling_min',\n                   'AccAP_rolling_mean', 'AccAP_rolling_std', 'AccAP_rolling_max', 'AccAP_rolling_min']\n\nX_tdcsfog= tdcsfog_test[feature_columns]","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:53:12.786278Z","iopub.execute_input":"2024-05-27T14:53:12.786741Z","iopub.status.idle":"2024-05-27T14:53:12.79646Z","shell.execute_reply.started":"2024-05-27T14:53:12.786707Z","shell.execute_reply":"2024-05-27T14:53:12.795066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_pred_defog = fog_model_defog.predict(X_defog)\ntest_pred_tdcsfog = fog_model_tdcsfog.predict(X_tdcsfog)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:53:14.959214Z","iopub.execute_input":"2024-05-27T14:53:14.959701Z","iopub.status.idle":"2024-05-27T14:53:16.004392Z","shell.execute_reply.started":"2024-05-27T14:53:14.959657Z","shell.execute_reply":"2024-05-27T14:53:16.003155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Add momentum to defog identification; FOG event continuity\nde_len = len(test_pred_defog)\ndefog_pred = np.insert(test_pred_defog, 0, 0)\ndefog_pred_fog = [(defog_pred[i]/8) + defog_pred[i+1] for i in range(de_len)]","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:53:29.37261Z","iopub.execute_input":"2024-05-27T14:53:29.373734Z","iopub.status.idle":"2024-05-27T14:53:29.509767Z","shell.execute_reply.started":"2024-05-27T14:53:29.373686Z","shell.execute_reply":"2024-05-27T14:53:29.508637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Add momentum to defog identification; FOG event continuity\ntdcsfog_len = len(test_pred_tdcsfog)\ntdcsfog_pred = np.insert(test_pred_tdcsfog, 0, 0)\ntdcsfog_pred_fog = [(tdcsfog_pred[i]/8) + tdcsfog_pred[i+1] for i in range(tdcsfog_len)]","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:53:21.147229Z","iopub.execute_input":"2024-05-27T14:53:21.148375Z","iopub.status.idle":"2024-05-27T14:53:21.156059Z","shell.execute_reply.started":"2024-05-27T14:53:21.148331Z","shell.execute_reply":"2024-05-27T14:53:21.154864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_test['FogProb'] = tdcsfog_pred_fog","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:53:23.488213Z","iopub.execute_input":"2024-05-27T14:53:23.489647Z","iopub.status.idle":"2024-05-27T14:53:23.498339Z","shell.execute_reply.started":"2024-05-27T14:53:23.489601Z","shell.execute_reply":"2024-05-27T14:53:23.496941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_test['FogProb'] = defog_pred_fog","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:53:31.671199Z","iopub.execute_input":"2024-05-27T14:53:31.672317Z","iopub.status.idle":"2024-05-27T14:53:31.799066Z","shell.execute_reply.started":"2024-05-27T14:53:31.672263Z","shell.execute_reply":"2024-05-27T14:53:31.797697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_defog","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:53:33.644779Z","iopub.execute_input":"2024-05-27T14:53:33.645259Z","iopub.status.idle":"2024-05-27T14:53:33.672313Z","shell.execute_reply.started":"2024-05-27T14:53:33.645224Z","shell.execute_reply":"2024-05-27T14:53:33.671073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_defog_SH_pred = model1_defog_SH.predict(X_defog, predict_disable_shape_check='TRUE')\ntest_defog_T_pred = model2_defog_T.predict(X_defog, predict_disable_shape_check='TRUE')\ntest_defog_W_pred = model3_defog_W.predict(X_defog, predict_disable_shape_check='TRUE')","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:53:37.336152Z","iopub.execute_input":"2024-05-27T14:53:37.336616Z","iopub.status.idle":"2024-05-27T14:53:41.266254Z","shell.execute_reply.started":"2024-05-27T14:53:37.336582Z","shell.execute_reply":"2024-05-27T14:53:41.264925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tdcsfog_SH_pred = model4_tdcsfog_SH.predict(X_tdcsfog)\ntest_tdcsfog_T_pred = model5_tdcsfog_T.predict(X_tdcsfog)\ntest_tdcsfog_W_pred = model6_tdcsfog_W.predict(X_tdcsfog)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:53:43.720956Z","iopub.execute_input":"2024-05-27T14:53:43.721927Z","iopub.status.idle":"2024-05-27T14:53:43.81075Z","shell.execute_reply.started":"2024-05-27T14:53:43.721881Z","shell.execute_reply":"2024-05-27T14:53:43.80939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_test['StartHesitation'] = np.sqrt(test_defog_SH_pred * defog_pred_fog)\ndefog_test['Turn'] = np.sqrt(test_defog_T_pred * defog_pred_fog)\ndefog_test['Walking'] = np.sqrt(test_defog_W_pred * defog_pred_fog)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:53:46.286251Z","iopub.execute_input":"2024-05-27T14:53:46.287292Z","iopub.status.idle":"2024-05-27T14:53:46.349773Z","shell.execute_reply.started":"2024-05-27T14:53:46.287247Z","shell.execute_reply":"2024-05-27T14:53:46.3486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_test['StartHesitation'] = np.sqrt(test_tdcsfog_SH_pred * tdcsfog_pred_fog)\ntdcsfog_test['Turn'] = np.sqrt(test_tdcsfog_T_pred * tdcsfog_pred_fog)\ntdcsfog_test['Walking'] = np.sqrt(test_tdcsfog_W_pred * tdcsfog_pred_fog)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:53:48.431269Z","iopub.execute_input":"2024-05-27T14:53:48.431752Z","iopub.status.idle":"2024-05-27T14:53:48.444591Z","shell.execute_reply.started":"2024-05-27T14:53:48.431708Z","shell.execute_reply":"2024-05-27T14:53:48.443478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_defog = defog_test[['Id','StartHesitation','Turn','Walking']]","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:53:50.946094Z","iopub.execute_input":"2024-05-27T14:53:50.947284Z","iopub.status.idle":"2024-05-27T14:53:50.966383Z","shell.execute_reply.started":"2024-05-27T14:53:50.947243Z","shell.execute_reply":"2024-05-27T14:53:50.965141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_tdcsfog = tdcsfog_test[['Id','StartHesitation','Turn','Walking']]","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:53:53.247962Z","iopub.execute_input":"2024-05-27T14:53:53.248422Z","iopub.status.idle":"2024-05-27T14:53:53.255594Z","shell.execute_reply.started":"2024-05-27T14:53:53.24839Z","shell.execute_reply":"2024-05-27T14:53:53.254354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.concat([submission_defog, submission_tdcsfog], ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:53:55.951309Z","iopub.execute_input":"2024-05-27T14:53:55.951778Z","iopub.status.idle":"2024-05-27T14:53:55.96534Z","shell.execute_reply.started":"2024-05-27T14:53:55.951744Z","shell.execute_reply":"2024-05-27T14:53:55.964066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:53:57.827017Z","iopub.execute_input":"2024-05-27T14:53:57.827491Z","iopub.status.idle":"2024-05-27T14:53:59.661963Z","shell.execute_reply.started":"2024-05-27T14:53:57.827457Z","shell.execute_reply":"2024-05-27T14:53:59.660764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:54:01.628675Z","iopub.execute_input":"2024-05-27T14:54:01.629174Z","iopub.status.idle":"2024-05-27T14:54:01.644364Z","shell.execute_reply.started":"2024-05-27T14:54:01.629137Z","shell.execute_reply":"2024-05-27T14:54:01.643228Z"},"trusted":true},"execution_count":null,"outputs":[]}]}