{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":41880,"databundleVersionId":5677426,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-26T03:44:37.112737Z","iopub.execute_input":"2024-11-26T03:44:37.113056Z","iopub.status.idle":"2024-11-26T03:44:40.660946Z","shell.execute_reply.started":"2024-11-26T03:44:37.113030Z","shell.execute_reply":"2024-11-26T03:44:40.659769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport glob\nimport pandas as pd\n\n\ntrain_dir = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog'\ntest_dir = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog'\n\n\ntrain_files = glob.glob(os.path.join(train_dir, '*.csv'))\ntrain_data = pd.concat([pd.read_csv(file) for file in train_files], ignore_index=True)\n\n\ntest_files = glob.glob(os.path.join(test_dir, '*.csv'))\ntest_data = pd.concat([pd.read_csv(file) for file in test_files], ignore_index=True)\n\n\nprint(\"Train Data Info:\")\nprint(train_data.info())\nprint(\"\\nTest Data Info:\")\nprint(test_data.info())\n\n\nprint(\"\\nFirst 5 rows of Train Data:\")\nprint(train_data.head())\n\nprint(\"\\nFirst 5 rows of Test Data:\")\nprint(test_data.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T12:45:50.725670Z","iopub.execute_input":"2024-11-21T12:45:50.726961Z","iopub.status.idle":"2024-11-21T12:46:11.062678Z","shell.execute_reply.started":"2024-11-21T12:45:50.726912Z","shell.execute_reply":"2024-11-21T12:46:11.061684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nmissing_train = train_data.isnull().sum()\nprint(\"\\nMissing Values in Train Data:\")\nprint(missing_train[missing_train > 0])\n\n\nmissing_test = test_data.isnull().sum()\nprint(\"\\nMissing Values in Test Data:\")\nprint(missing_test[missing_test > 0])\n\n\ntrain_data.fillna(train_data.median(), inplace=True)\ntest_data.fillna(test_data.median(), inplace=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:41:18.98445Z","iopub.execute_input":"2024-11-17T10:41:18.984903Z","iopub.status.idle":"2024-11-17T10:41:20.634742Z","shell.execute_reply.started":"2024-11-17T10:41:18.98486Z","shell.execute_reply":"2024-11-17T10:41:20.633534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nplt.figure(figsize=(10, 10))\ntrain_data.hist()\nplt.suptitle(\"Distribution in Train Data\")\nplt.show()\n\nplt.figure(figsize=(12, 10))\ntest_data.hist()\nplt.suptitle(\"Distribution in Test Data\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T11:27:15.128982Z","iopub.execute_input":"2024-11-17T11:27:15.129503Z","iopub.status.idle":"2024-11-17T11:27:20.309449Z","shell.execute_reply.started":"2024-11-17T11:27:15.129457Z","shell.execute_reply":"2024-11-17T11:27:20.308358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 8))\nsns.heatmap(train_data.corr(), annot=True)\nplt.title(\"Correlation Matrix for Train Data\")\nplt.show()\n\nplt.figure(figsize=(12, 8))\nsns.heatmap(test_data.corr(), annot=True)\nplt.title(\"Correlation Matrix for Test Data\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T11:21:34.549642Z","iopub.execute_input":"2024-11-17T11:21:34.550093Z","iopub.status.idle":"2024-11-17T11:21:39.190292Z","shell.execute_reply.started":"2024-11-17T11:21:34.550052Z","shell.execute_reply":"2024-11-17T11:21:39.189076Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in train_data.select_dtypes(include=[np.number]).columns:\n    plt.figure(figsize=(10, 5))\n    sns.boxplot(x=train_data[col])\n    plt.title(f\"Boxplot of {col} in Train Data\")\n    plt.show()\n    \nfor col in test_data.select_dtypes(include=[np.number]).columns:\n    plt.figure(figsize=(10, 5))\n    sns.boxplot(x=test_data[col])\n    plt.title(f\"Boxplot of {col} in Test Data\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:42:26.141455Z","iopub.execute_input":"2024-11-17T10:42:26.141937Z","iopub.status.idle":"2024-11-17T10:42:37.118344Z","shell.execute_reply.started":"2024-11-17T10:42:26.141891Z","shell.execute_reply":"2024-11-17T10:42:37.117094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"skewness_train = train_data.select_dtypes(include=[np.number]).skew()\nskewness_test = test_data.select_dtypes(include=[np.number]).skew()\n\nprint(\"Skewness in Train Data:\")\nprint(skewness_train)\n\nprint(\"\\nSkewness in Test Data:\")\nprint(skewness_test)\n\nplt.figure(figsize=(12, 6))\ntrain_data.select_dtypes(include=[np.number]).skew().plot(kind='bar')\nplt.title(\"Skewness of Numerical Features in Train Data\")\nplt.show()\nplt.figure(figsize=(12, 6))\ntest_data.select_dtypes(include=[np.number]).skew().plot(kind='bar')\nplt.title(\"Skewness of Numerical Features in Test Data\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:57:50.425556Z","iopub.execute_input":"2024-11-17T10:57:50.426957Z","iopub.status.idle":"2024-11-17T10:57:55.196958Z","shell.execute_reply.started":"2024-11-17T10:57:50.426898Z","shell.execute_reply":"2024-11-17T10:57:55.195697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.subplot(3, 1, 1)\nplt.plot(train_data['Time'], train_data['AccV'])\nplt.title('accV - Acceleration in Vertical Axis')\nplt.xlabel('Time')\nplt.ylabel('Acceleration (m/s²)')\n\nplt.subplot(3, 1, 2)\nplt.plot(train_data['Time'], train_data['AccML'])\nplt.title('accML - Acceleration in Medial-Lateral Axis')\nplt.xlabel('Time')\nplt.ylabel('Acceleration (m/s²)')\n\nplt.subplot(3, 1, 3)\nplt.plot(train_data['Time'], train_data['AccAP'])\nplt.title('accAP - Acceleration in Anterior-Posterior Axis')\nplt.xlabel('Time')\nplt.ylabel('Acceleration (m/s²)')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T11:05:26.752572Z","iopub.execute_input":"2024-11-17T11:05:26.753021Z","iopub.status.idle":"2024-11-17T11:05:32.857098Z","shell.execute_reply.started":"2024-11-17T11:05:26.752978Z","shell.execute_reply":"2024-11-17T11:05:32.855852Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sensor_columns = ['AccV', 'AccML', 'AccAP']\nprint(train_data[sensor_columns].describe())\nprint(train_data[sensor_columns].skew())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T11:07:34.642316Z","iopub.execute_input":"2024-11-17T11:07:34.642774Z","iopub.status.idle":"2024-11-17T11:07:37.130105Z","shell.execute_reply.started":"2024-11-17T11:07:34.642731Z","shell.execute_reply":"2024-11-17T11:07:37.128929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sensor_data_corr = train_data[sensor_columns].corr()\nplt.figure(figsize=(6, 5))\nsns.heatmap(sensor_data_corr, annot=True, fmt=\".2f\", cmap='coolwarm', vmin=-1, vmax=1)\nplt.title(\"Correlation Heatmap for Accelerometer Data\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T11:08:15.079358Z","iopub.execute_input":"2024-11-17T11:08:15.08027Z","iopub.status.idle":"2024-11-17T11:08:15.987827Z","shell.execute_reply.started":"2024-11-17T11:08:15.080223Z","shell.execute_reply":"2024-11-17T11:08:15.986586Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(15, 5))\nsns.boxplot(x='Walking', y='AccV', data=train_data)\nplt.title('Walking Impact on Vertical Acceleration')\nplt.show()\n\nsns.boxplot(x='StartHesitation', y='AccML', data=train_data)\nplt.title('Start Hesitation Impact on Medial-Lateral Acceleration')\nplt.show()\n\nsns.boxplot(x='Turn', y='AccAP', data=train_data)\nplt.title('Turn Impact on Anterior-Posterior Acceleration')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T11:12:47.909902Z","iopub.execute_input":"2024-11-17T11:12:47.910385Z","iopub.status.idle":"2024-11-17T11:12:56.266605Z","shell.execute_reply.started":"2024-11-17T11:12:47.910343Z","shell.execute_reply":"2024-11-17T11:12:56.265492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"walking_data = train_data[train_data['Walking'] == 1]\nplt.figure(figsize=(12, 6))\nplt.plot(walking_data['AccV'], label=\"AccV (Vertical)\")\nplt.plot(walking_data['AccML'], label=\"AccML (Medial-Lateral)\")\nplt.plot(walking_data['AccAP'], label=\"AccAP (Anterior-Posterior)\")\nplt.title('Sensor Data during Walking')\nplt.legend()\nplt.show()\n\nhesitation_data = train_data[train_data['StartHesitation'] == 1]\nplt.figure(figsize=(12, 6))\nplt.plot(hesitation_data['AccV'], label=\"AccV (Vertical)\")\nplt.plot(hesitation_data['AccML'], label=\"AccML (Medial-Lateral)\")\nplt.plot(hesitation_data['AccAP'], label=\"AccAP (Anterior-Posterior)\")\nplt.title('Sensor Data during Start Hesitation')\nplt.legend()\nplt.show()\n\nturn_data = train_data[train_data['Turn'] == 1]\nplt.figure(figsize=(12, 6))\nplt.plot(turn_data['AccV'], label=\"AccV (Vertical)\")\nplt.plot(turn_data['AccML'], label=\"AccML (Medial-Lateral)\")\nplt.plot(turn_data['AccAP'], label=\"AccAP (Anterior-Posterior)\")\nplt.title('Sensor Data during Turn')\nplt.legend()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T11:17:56.264986Z","iopub.execute_input":"2024-11-17T11:17:56.26604Z","iopub.status.idle":"2024-11-17T11:18:00.426315Z","shell.execute_reply.started":"2024-11-17T11:17:56.265992Z","shell.execute_reply":"2024-11-17T11:18:00.425236Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"people_df = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/subjects.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T07:59:10.570648Z","iopub.execute_input":"2024-11-21T07:59:10.571099Z","iopub.status.idle":"2024-11-21T07:59:10.607367Z","shell.execute_reply.started":"2024-11-21T07:59:10.571065Z","shell.execute_reply":"2024-11-21T07:59:10.606747Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"people_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T07:59:25.480037Z","iopub.execute_input":"2024-11-21T07:59:25.480402Z","iopub.status.idle":"2024-11-21T07:59:25.503880Z","shell.execute_reply.started":"2024-11-21T07:59:25.480358Z","shell.execute_reply":"2024-11-21T07:59:25.502937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Load the dataset (replace with the actual path if needed)\npeople_df = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/subjects.csv\")\n\n# Basic DataFrame check\nprint(people_df.info())\nprint(people_df.describe())\n\n# Plot 1: Distribution of Age\nplt.figure(figsize=(8, 5))\nsns.histplot(people_df[\"Age\"].dropna(), bins=15, kde=True, color='skyblue')\nplt.title(\"Distribution of Age\")\nplt.xlabel(\"Age\")\nplt.ylabel(\"Frequency\")\nplt.show()\n\n# Plot 2: Gender Distribution\nplt.figure(figsize=(6, 5))\nsns.countplot(x=\"Sex\", data=people_df, palette=\"pastel\")\nplt.title(\"Gender Distribution\")\nplt.xlabel(\"Sex\")\nplt.ylabel(\"Count\")\nplt.show()\n\n# Plot 3: UPDRSIII (On and Off) Correlation\nplt.figure(figsize=(8, 6))\nsns.scatterplot(x=\"UPDRSIII_On\", y=\"UPDRSIII_Off\", data=people_df, color=\"green\")\nplt.title(\"Correlation Between UPDRSIII (On) and UPDRSIII (Off)\")\nplt.xlabel(\"UPDRSIII On\")\nplt.ylabel(\"UPDRSIII Off\")\nplt.show()\n\n# Plot 4: Years Since Diagnosis vs NFOGQ\nplt.figure(figsize=(8, 6))\nsns.scatterplot(x=\"YearsSinceDx\", y=\"NFOGQ\", data=people_df, hue=\"Sex\", palette=\"Set2\")\nplt.title(\"Years Since Diagnosis vs NFOGQ\")\nplt.xlabel(\"Years Since Diagnosis\")\nplt.ylabel(\"NFOGQ\")\nplt.legend(title=\"Sex\")\nplt.show()\n\n# Plot 5: Distribution of Visits\nplt.figure(figsize=(8, 5))\nsns.countplot(x=\"Visit\", data=people_df, palette=\"viridis\")\nplt.title(\"Distribution of Visits\")\nplt.xlabel(\"Visit\")\nplt.ylabel(\"Count\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:00:35.899946Z","iopub.execute_input":"2024-11-21T08:00:35.900836Z","iopub.status.idle":"2024-11-21T08:00:37.804228Z","shell.execute_reply.started":"2024-11-21T08:00:35.900787Z","shell.execute_reply":"2024-11-21T08:00:37.803444Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tasks_df = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tasks.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:03:00.408382Z","iopub.execute_input":"2024-11-21T08:03:00.409175Z","iopub.status.idle":"2024-11-21T08:03:00.424298Z","shell.execute_reply.started":"2024-11-21T08:03:00.409140Z","shell.execute_reply":"2024-11-21T08:03:00.423675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tasks_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:03:02.087316Z","iopub.execute_input":"2024-11-21T08:03:02.087617Z","iopub.status.idle":"2024-11-21T08:03:02.097326Z","shell.execute_reply.started":"2024-11-21T08:03:02.087588Z","shell.execute_reply":"2024-11-21T08:03:02.096467Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\n#Calculate the Task Duration by subtracting Begin Time from End Time\ntasks_df['Task Duration (s)'] = tasks_df['End'] - tasks_df['Begin']\n\n# Task Summary: Mean, Max, Min Task Durations\nmean_duration = tasks_df['Task Duration (s)'].mean()\nmax_duration = tasks_df['Task Duration (s)'].max()\nmin_duration = tasks_df['Task Duration (s)'].min()\n\nprint(f\"Mean Task Duration: {mean_duration} seconds\")\nprint(f\"Max Task Duration: {max_duration} seconds\")\nprint(f\"Min Task Duration: {min_duration} seconds\")\n\n# Plot 1: Distribution of Task Durations\nplt.figure(figsize=(8, 5))\ntasks_df['Task Duration (s)'].plot(kind='hist', bins=20, color='skyblue', edgecolor='black', alpha=0.7)\nplt.axvline(mean_duration, color='red', linestyle='--', label=f'Mean: {mean_duration:.2f}s')\nplt.axvline(max_duration, color='green', linestyle='--', label=f'Max: {max_duration:.2f}s')\nplt.axvline(min_duration, color='orange', linestyle='--', label=f'Min: {min_duration:.2f}s')\nplt.title(\"Distribution of Task Durations\")\nplt.xlabel(\"Task Duration (seconds)\")\nplt.ylabel(\"Frequency\")\nplt.legend()\nplt.show()\n\n# Plot 2: Boxplot for Task Durations\nplt.figure(figsize=(8, 5))\nsns.boxplot(x=tasks_df['Task Duration (s)'], color='lightgreen')\nplt.title(\"Boxplot of Task Durations\")\nplt.xlabel(\"Task Duration (seconds)\")\nplt.show()\n\n# Plot 3: Task Duration vs Task Type\nplt.figure(figsize=(10, 6))\nsns.boxplot(x=\"Task\", y=\"Task Duration (s)\", data=tasks_df, palette=\"Set2\")\nplt.title(\"Task Duration vs Task Type\")\nplt.xlabel(\"Task Type\")\nplt.ylabel(\"Task Duration (seconds)\")\nplt.xticks(rotation=45)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:05:26.377982Z","iopub.execute_input":"2024-11-21T08:05:26.378299Z","iopub.status.idle":"2024-11-21T08:05:27.478596Z","shell.execute_reply.started":"2024-11-21T08:05:26.378271Z","shell.execute_reply":"2024-11-21T08:05:27.477704Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport pywt\nimport numpy as np\n\n# Define the directory containing Defog data\ntrain_dir = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog'\n\n# Define a function to apply DWT\ndef apply_dwt(signal, wavelet='db4', level=4):\n    \"\"\"\n    Apply Discrete Wavelet Transform to a signal.\n    Returns approximation and detail coefficients.\n    \"\"\"\n    coeffs = pywt.wavedec(signal, wavelet=wavelet, level=level)\n    return coeffs  # [A_n, D_n, D_n-1, ..., D_1]\n\n# Initialize a list to store features\nfeatures = []\n\n# Iterate over all CSV files in the directory\nfor file in os.listdir(train_dir):\n    if file.endswith('.csv'):\n        # Load the CSV file\n        file_path = os.path.join(train_dir, file)\n        data = pd.read_csv(file_path)\n        \n        # Extract relevant columns (e.g., acceleration signals)\n        acc_v = data['AccV']  # Vertical acceleration\n        acc_ml = data['AccML']  # Mediolateral acceleration\n        acc_ap = data['AccAP']  # Anteroposterior acceleration\n        \n        # Apply DWT to each signal\n        coeffs_v = apply_dwt(acc_v)\n        coeffs_ml = apply_dwt(acc_ml)\n        coeffs_ap = apply_dwt(acc_ap)\n        \n        # Extract statistical features from the coefficients\n        features_v = [np.mean(c) for c in coeffs_v] + [np.std(c) for c in coeffs_v]\n        features_ml = [np.mean(c) for c in coeffs_ml] + [np.std(c) for c in coeffs_ml]\n        features_ap = [np.mean(c) for c in coeffs_ap] + [np.std(c) for c in coeffs_ap]\n        \n        # Combine all features into a single row\n        feature_row = features_v + features_ml + features_ap\n        features.append(feature_row)\n\n# Convert features to a DataFrame\ncolumns = [f'coeff_{i}' for i in range(len(features[0]))]\nfeatures_df = pd.DataFrame(features, columns=columns)\n\n# Save the extracted features for modeling\nfeatures_df.to_csv('dwt_features.csv', index=False)\nprint(\"DWT features saved as 'dwt_features.csv'\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T03:45:14.645038Z","iopub.execute_input":"2024-11-26T03:45:14.645845Z","iopub.status.idle":"2024-11-26T03:45:39.285630Z","shell.execute_reply.started":"2024-11-26T03:45:14.645805Z","shell.execute_reply":"2024-11-26T03:45:39.284686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T03:45:46.784469Z","iopub.execute_input":"2024-11-26T03:45:46.785058Z","iopub.status.idle":"2024-11-26T03:45:46.809320Z","shell.execute_reply.started":"2024-11-26T03:45:46.785025Z","shell.execute_reply":"2024-11-26T03:45:46.808444Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport pywt\nimport numpy as np\nimport matplotlib.pyplot as plt\n\n# Define the directory containing Defog data\ntrain_dir = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog'\n\n# Define a function to apply DWT\ndef apply_dwt(signal, wavelet='db4', level=4):\n    \"\"\"\n    Apply Discrete Wavelet Transform to a signal.\n    Returns approximation and detail coefficients.\n    \"\"\"\n    coeffs = pywt.wavedec(signal, wavelet=wavelet, level=level)\n    return coeffs  # [A_n, D_n, D_n-1, ..., D_1]\n\n# Load and process one file for visualization\nfile_to_visualize = None\nfor file in os.listdir(train_dir):\n    if file.endswith('.csv'):\n        file_to_visualize = os.path.join(train_dir, file)\n        break\n\n# Read the CSV file\ndata = pd.read_csv(file_to_visualize)\n\n# Extract the vertical acceleration signal\nacc_v = data['AccV']\n\n# Apply DWT\ncoeffs_v = apply_dwt(acc_v)\napproximation = coeffs_v[0]\ndetails = coeffs_v[1:]\n\n# Plot the raw signal and DWT-transformed components\nplt.figure(figsize=(15, 8))\n\n# Plot raw signal\nplt.subplot(3, 1, 1)\nplt.plot(acc_v, label='Raw Signal (AccV)', color='blue')\nplt.title('Raw Acceleration Signal (AccV)')\nplt.xlabel('Time')\nplt.ylabel('Amplitude')\nplt.legend()\n\n# Plot DWT Approximation\nplt.subplot(3, 1, 2)\nplt.plot(np.linspace(0, len(acc_v), len(approximation)), approximation, label='DWT Approximation', color='orange')\nplt.title('DWT Approximation Coefficient')\nplt.xlabel('Time')\nplt.ylabel('Amplitude')\nplt.legend()\n\n# Plot DWT Details\nplt.subplot(3, 1, 3)\nfor i, detail in enumerate(details):\n    plt.plot(np.linspace(0, len(acc_v), len(detail)), detail, label=f'DWT Detail {i+1}', alpha=0.7)\nplt.title('DWT Detail Coefficients')\nplt.xlabel('Time')\nplt.ylabel('Amplitude')\nplt.legend()\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T03:46:22.902229Z","iopub.execute_input":"2024-11-26T03:46:22.902915Z","iopub.status.idle":"2024-11-26T03:46:24.151490Z","shell.execute_reply.started":"2024-11-26T03:46:22.902882Z","shell.execute_reply":"2024-11-26T03:46:24.150613Z"}},"outputs":[],"execution_count":null}]}