{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"![ML](https://raw.githubusercontent.com/AniMilina/Parkinson-s-Freezing-of-Gait-Prediction/main/EDA.jpg)\n","metadata":{}},{"cell_type":"markdown","source":"Our task is to detect freezing of gait (FOG) episodes during walking in the tdcsfog and defog datasets. To accomplish this, we need to use the data of temporal steps, accelerations along three axes, and event types (recorded in the Type column) from these two datasets.\n\nThe defog dataset has two additional columns: Valid and Task. We should only use event annotations where the series is marked as true, and sections marked as false should be treated as unannotated.\n\nThe tdcsfog_metadata, defog_metadata, and events metadata contain information about laboratory visits, tests performed, and medications taken for each subject in the study. We can use this information to analyze the results of our model depending on various factors.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nfrom IPython.display import display\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport seaborn as sns\nfrom sklearn.metrics import mutual_info_score\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:55:50.329685Z","iopub.execute_input":"2023-05-15T16:55:50.330016Z","iopub.status.idle":"2023-05-15T16:55:51.935305Z","shell.execute_reply.started":"2023-05-15T16:55:50.329989Z","shell.execute_reply":"2023-05-15T16:55:51.934548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set formatting option\n\npd.options.display.float_format = '{:.3f}'.format","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:55:51.937701Z","iopub.execute_input":"2023-05-15T16:55:51.938046Z","iopub.status.idle":"2023-05-15T16:55:51.945917Z","shell.execute_reply.started":"2023-05-15T16:55:51.938017Z","shell.execute_reply":"2023-05-15T16:55:51.944970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fill_missing_values(df):\n    \"\"\"\n    Replaces missing values with the median value for each numerical column.\n\n    :param df: pandas DataFrame, the dataset in which missing values need to be replaced\n    :return: pandas DataFrame, the dataset with replaced missing values\n    \"\"\"\n    numeric_cols = df.select_dtypes(include=['float64', 'int64']).columns  # Selecting all numerical columns\n    for col in numeric_cols:\n        median = df[col].median()  # Finding the Median Value of a Column\n        df[col].fillna(median, inplace=True)   # Replacing Missing Values with the Median Value\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:55:51.953650Z","iopub.execute_input":"2023-05-15T16:55:51.953955Z","iopub.status.idle":"2023-05-15T16:55:51.959286Z","shell.execute_reply.started":"2023-05-15T16:55:51.953930Z","shell.execute_reply":"2023-05-15T16:55:51.958319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to Get Data Information\n\ndef explore_dataframe(df):\n    print(\"Shape of dataframe:\", df.shape)\n    display(df.head())\n    print(\"Info of dataframe:\\n\")\n    df.info()\n    print(\"Summary statistics of dataframe:\\n\", df.describe())\n    print(\"Missing values in dataframe:\\n\", df.isnull().sum())\n    print(\"Duplicate rows in dataframe:\", df.duplicated().sum())","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:55:51.960535Z","iopub.execute_input":"2023-05-15T16:55:51.961042Z","iopub.status.idle":"2023-05-15T16:55:51.972789Z","shell.execute_reply.started":"2023-05-15T16:55:51.961018Z","shell.execute_reply":"2023-05-15T16:55:51.972081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking for Missing Values in Each Column\n\ndef check_missing_values(df):\n    \"\"\"\n    Checks the count of missing values in each column of a DataFrame.\n\n    :param df: pandas.DataFrame, the DataFrame to check for missing values.\n    :return: pandas.DataFrame, the DataFrame with information about missing values.\n    \"\"\"\n    return df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:55:51.973931Z","iopub.execute_input":"2023-05-15T16:55:51.974651Z","iopub.status.idle":"2023-05-15T16:55:51.985144Z","shell.execute_reply.started":"2023-05-15T16:55:51.974627Z","shell.execute_reply":"2023-05-15T16:55:51.984117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reading Data from tdcsfog_metadata.csv\n\ntdcsfog = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv')\n\n# Reading Data from defog_metadata.csv\n\ndefog = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv')\n\n# Reading Data from events.csv\n\nevents = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/events.csv')\n\n# Reading Data from train & tdcsfog & daily_metadata.csv\n\nacc = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/08fbe142f9.csv')\n\n# Reading Data from subjects.csv\n\nsubject = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/subjects.csv')","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:55:51.986184Z","iopub.execute_input":"2023-05-15T16:55:51.986505Z","iopub.status.idle":"2023-05-15T16:55:52.053001Z","shell.execute_reply.started":"2023-05-15T16:55:51.986481Z","shell.execute_reply":"2023-05-15T16:55:52.052069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explore_dataframe(tdcsfog)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:55:52.054171Z","iopub.execute_input":"2023-05-15T16:55:52.054991Z","iopub.status.idle":"2023-05-15T16:55:52.117205Z","shell.execute_reply.started":"2023-05-15T16:55:52.054966Z","shell.execute_reply":"2023-05-15T16:55:52.116182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The dataset appears to be complete and clean, with no missing values or duplicates.\n\nThe dataset consists of 833 records and 5 columns, which describe the patient identifier, visit number, test number, medication treatment, and test results. The \"Test\" column has only three possible values, indicating that each test measured only one parameter. These attributes provide valuable information for analysis and model building.","metadata":{}},{"cell_type":"code","source":"explore_dataframe(defog)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:55:52.118767Z","iopub.execute_input":"2023-05-15T16:55:52.119359Z","iopub.status.idle":"2023-05-15T16:55:52.146473Z","shell.execute_reply.started":"2023-05-15T16:55:52.119320Z","shell.execute_reply":"2023-05-15T16:55:52.145526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This dataset contains information about 137 patients who underwent treatment with various medications to improve brain function.\n\nEach row provides information about the patient's identifier, their FOG identifier, and whether or not they took medication.\n\nThe dataset does not contain any missing values or duplicates.\n\nAmong the patients, there are only two values for the visit number: 1 and 2. The majority of patients (75%) underwent treatment during the first visit.","metadata":{}},{"cell_type":"code","source":"explore_dataframe(events)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:55:52.150994Z","iopub.execute_input":"2023-05-15T16:55:52.151672Z","iopub.status.idle":"2023-05-15T16:55:52.183075Z","shell.execute_reply.started":"2023-05-15T16:55:52.151634Z","shell.execute_reply":"2023-05-15T16:55:52.181949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The dataset consists of 3544 events, each characterized by five columns: \"Id\", \"Init\", \"Completion\", \"Type\", and \"Kinetic\". \"Id\" represents a unique identifier for each event. \"Init\" and \"Completion\" describe the start and end time of the event, respectively. \"Type\" and \"Kinetic\" are categorical and numerical variables, respectively, that describe the type and kinetic activity of the event.\n\nThe number of non-null values for all columns is 3544, except for the \"Type\" and \"Kinetic\" columns, which have 1042 missing values. There are no duplicate rows in the dataset.\n\nThe mean values for \"Init\", \"Completion\", and \"Kinetic\" are 956.30, 964.49, and 0.82, respectively. The standard deviations are 946.36, 943.97, and 0.39, respectively. The minimum and maximum values for \"Init\" are -30.67 and 4381.22, respectively, while for \"Completion\", the minimum and maximum values are -29.72 and 4392.75, respectively.","metadata":{}},{"cell_type":"code","source":"explore_dataframe(acc)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:55:52.184328Z","iopub.execute_input":"2023-05-15T16:55:52.185349Z","iopub.status.idle":"2023-05-15T16:55:52.222679Z","shell.execute_reply.started":"2023-05-15T16:55:52.185320Z","shell.execute_reply":"2023-05-15T16:55:52.221971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\nThe dataset contains 4474 rows and 7 columns: Time, AccV, AccML, AccAP, StartHesitation, Turn, and Walking.\n\nThe Time column represents the timestamp of the accelerometer data. AccV, AccML, and AccAP represent the acceleration values in the vertical, medio-lateral, and antero-posterior directions, respectively. StartHesitation, Turn, and Walking are binary variables indicating whether the subject is starting to walk, turning, or already walking at a given timestamp.\n\nThe summary statistics show that the mean value of vertical acceleration is -9.61 m/s², the mean value of medio-lateral acceleration is -0.14 m/s², and the mean value of antero-posterior acceleration is 1.14 m/s². There are no missing values or duplicate rows in the dataset.","metadata":{}},{"cell_type":"code","source":"explore_dataframe(subject)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:55:52.223790Z","iopub.execute_input":"2023-05-15T16:55:52.225076Z","iopub.status.idle":"2023-05-15T16:55:52.254979Z","shell.execute_reply.started":"2023-05-15T16:55:52.225027Z","shell.execute_reply":"2023-05-15T16:55:52.254320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\nThis dataset contains data on 173 patients with Parkinson's disease. It has 8 columns, including patient ID, age, gender, time since diagnosis, UPDRSIII_On and UPDRSIII_Off test results, as well as NFOGQ questionnaire results. Some patients were tested more than once, which is reflected in the Visit column.\n\nThe dataset contains some missing values in the Visit, UPDRSIII_On, and UPDRSIII_Off columns, but only the Visit and UPDRSIII_Off columns have more than 20% missing values. The majority of patients are male (M), and the average age is 67.76 years. Regarding the NFOGQ column, the mean value is 17.12, and the minimum value is 0, which may indicate that some patients do not experience freezing of gait (FOG).","metadata":{}},{"cell_type":"markdown","source":"For our task, we will need the following columns from the remaining datasets:\n\nIn tdcsfog and defog: Medication - to analyze the impact of medications on disease symptoms.\nIn events: Init, Completion, Type - to analyze the timing of various tasks and their types.\nIn acc: AccV, AccML, AccAP - to analyze patient activity and assess their motor abilities.","metadata":{}},{"cell_type":"code","source":"#To prepare the datasets for merging, we will first remove the column \"Kinetic\" as it is not informative for our analysis\n\nevents = events.drop('Kinetic', axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:55:52.255937Z","iopub.execute_input":"2023-05-15T16:55:52.256728Z","iopub.status.idle":"2023-05-15T16:55:52.261992Z","shell.execute_reply.started":"2023-05-15T16:55:52.256704Z","shell.execute_reply":"2023-05-15T16:55:52.261098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# To create three new columns, \"StartHesitation,\" \"Turn,\" and \"Walking,\" in the \"events\" dataset and assign them values from the \"Type\" column\n\nevents['StartHesitation'] = events['Type'].apply(lambda x: x == 'Start hesitation')\nevents['Turn'] = events['Type'].apply(lambda x: x == 'Turn')\nevents['Walking'] = events['Type'].apply(lambda x: x == 'Walking')\n\n# To remove the \"Type\" column from the \"events\" dataset\n\nevents = events.drop('Type', axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:55:52.263174Z","iopub.execute_input":"2023-05-15T16:55:52.264031Z","iopub.status.idle":"2023-05-15T16:55:52.282518Z","shell.execute_reply.started":"2023-05-15T16:55:52.263994Z","shell.execute_reply":"2023-05-15T16:55:52.281335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# To convert the values in the \"Turn,\" \"Walking,\" and \"StartHesitation\" columns to integers\n\nevents['StartHesitation'] = events['StartHesitation'].astype(int)\nevents['Turn'] = events['Turn'].astype(int)\nevents['Walking'] = events['Walking'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:55:52.283853Z","iopub.execute_input":"2023-05-15T16:55:52.284168Z","iopub.status.idle":"2023-05-15T16:55:52.300963Z","shell.execute_reply.started":"2023-05-15T16:55:52.284143Z","shell.execute_reply":"2023-05-15T16:55:52.300168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged_df = pd.merge(tdcsfog, defog, on=['Id', 'Subject', 'Visit', 'Medication'], how='outer')\nevents = events[['Id','Init','Completion','Turn', 'Walking', 'StartHesitation']]\nmerged_df = pd.merge(merged_df, events, on='Id', how='outer')\nmerged_df = pd.merge(merged_df, acc, on='StartHesitation', how='left')","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:55:52.302103Z","iopub.execute_input":"2023-05-15T16:55:52.302908Z","iopub.status.idle":"2023-05-15T16:55:55.047552Z","shell.execute_reply.started":"2023-05-15T16:55:52.302878Z","shell.execute_reply":"2023-05-15T16:55:55.046578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# To merge the columns \"Turn\" and \"Walking\" into a single column\n\nmerged_df['Turn'] = merged_df['Turn_x'].fillna(merged_df['Turn_y'])\nmerged_df['Walking'] = merged_df['Walking_x'].fillna(merged_df['Walking_y'])","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:55:55.048618Z","iopub.execute_input":"2023-05-15T16:55:55.048961Z","iopub.status.idle":"2023-05-15T16:55:55.182408Z","shell.execute_reply.started":"2023-05-15T16:55:55.048929Z","shell.execute_reply":"2023-05-15T16:55:55.181572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# To remove the unwanted columns\n\nmerged_df = merged_df.drop(['Turn_x', 'Turn_y', 'Walking_x', 'Walking_y'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:55:55.183855Z","iopub.execute_input":"2023-05-15T16:55:55.184144Z","iopub.status.idle":"2023-05-15T16:55:56.270598Z","shell.execute_reply.started":"2023-05-15T16:55:55.184119Z","shell.execute_reply":"2023-05-15T16:55:56.269543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(check_missing_values(merged_df))","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:55:56.272656Z","iopub.execute_input":"2023-05-15T16:55:56.273386Z","iopub.status.idle":"2023-05-15T16:56:02.543491Z","shell.execute_reply.started":"2023-05-15T16:55:56.273355Z","shell.execute_reply":"2023-05-15T16:56:02.542419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are missing values, but considering that the dataset is comprehensive, they are not significant.  \nThe \"Test\" column, which indicates the type of test performed, stands out. To preserve the integrity of the model's statistics, I will remove it at this stage but examine it together with the data from the \"Subject\" dataset.  \nFor the missing values in the accelerometer measurements, I will replace them with the median values, which should not distort the results given the small number of missing values compared to the total number of rows.","metadata":{}},{"cell_type":"code","source":"# Remove the \"Test\" column\n\nmerged_df = merged_df.drop('Test', axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:56:02.544726Z","iopub.execute_input":"2023-05-15T16:56:02.545023Z","iopub.status.idle":"2023-05-15T16:56:03.614694Z","shell.execute_reply.started":"2023-05-15T16:56:02.544995Z","shell.execute_reply":"2023-05-15T16:56:03.613370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged_df = fill_missing_values(merged_df)\n\n# Checking the result\n\ncheck_missing_values(merged_df)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:56:03.617077Z","iopub.execute_input":"2023-05-15T16:56:03.617825Z","iopub.status.idle":"2023-05-15T16:56:12.050890Z","shell.execute_reply.started":"2023-05-15T16:56:03.617775Z","shell.execute_reply":"2023-05-15T16:56:12.049685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We have obtained a dataset based on which we will further investigate the task.","metadata":{}},{"cell_type":"markdown","source":"## Descriptive statistics and analysis of value distribution","metadata":{}},{"cell_type":"markdown","source":"I don't remove duplicates as I believe they represent different measurements, for example, of the same patient.","metadata":{}},{"cell_type":"code","source":"merged_df = merged_df.reindex(columns=['Id', 'Subject', 'Visit', 'Medication', 'Time','Init','Completion', 'AccV', 'AccML', 'AccAP', 'StartHesitation', 'Turn', 'Walking'])\n","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:56:12.053011Z","iopub.execute_input":"2023-05-15T16:56:12.053334Z","iopub.status.idle":"2023-05-15T16:56:14.040404Z","shell.execute_reply.started":"2023-05-15T16:56:12.053308Z","shell.execute_reply":"2023-05-15T16:56:14.039324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explore_dataframe(merged_df)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:56:14.041529Z","iopub.execute_input":"2023-05-15T16:56:14.041784Z","iopub.status.idle":"2023-05-15T16:56:35.226157Z","shell.execute_reply.started":"2023-05-15T16:56:14.041762Z","shell.execute_reply":"2023-05-15T16:56:35.224767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(5, 2, figsize=(15, 8))\naxs = axs.flatten()\n\nfor i, col in enumerate(['Visit', 'Time','Init','Completion','AccV', 'AccML', 'AccAP', 'StartHesitation', 'Turn', 'Walking']):\n    axs[i].hist(merged_df[col], bins=20)\n    axs[i].set_title(col)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:56:35.227684Z","iopub.execute_input":"2023-05-15T16:56:35.228326Z","iopub.status.idle":"2023-05-15T16:56:38.419214Z","shell.execute_reply.started":"2023-05-15T16:56:35.228296Z","shell.execute_reply":"2023-05-15T16:56:38.418327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(5,2, figsize=(15, 10))\naxs = axs.flatten()\n\nfor i, col in enumerate(['Visit', 'Time','Init','Completion','AccV', 'AccML', 'AccAP', 'StartHesitation', 'Turn', 'Walking']):\n    axs[i].boxplot(merged_df[col])\n    axs[i].set_title(col)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:56:38.420630Z","iopub.execute_input":"2023-05-15T16:56:38.421146Z","iopub.status.idle":"2023-05-15T16:56:57.229748Z","shell.execute_reply.started":"2023-05-15T16:56:38.421113Z","shell.execute_reply":"2023-05-15T16:56:57.228204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Accelerometer Time Series Analysis for Fade Detection\n","metadata":{}},{"cell_type":"markdown","source":"Accelerometers in modern devices are used to measure acceleration and change in speed. In the context of this task, accelerations in the x, y, and z directions can provide some information about human behavior, including the presence of episodes of freezing, hesitation in turning or moving, delays in movement, etc.\n\nThe direction of acceleration along the x-axis (AccML) can reflect a change in the direction of movement of a person in the ground plane (for example, when turning), the direction of acceleration along the y-axis (AccAP) can reflect vertical movement (for example, going up or down stairs), and the direction of acceleration along z-axis (AccV) can reflect a change in forward or reverse speed.\n\nIn scoring validation, scoring values ​​can help identify episodes of stalling and hesitation in turning or scoring, for example, by analyzing the length of low score periods (or observed scoring variation) and scoring them against the overall length of the test run. Also, measurement values ​​can help determine the change in motion, for example, by measuring the mean or standard deviation of measurements along each of the axes.\n\nThe obtained features were used to build a model that can classify different motion episodes with high probability.","metadata":{}},{"cell_type":"code","source":"# Select the desired columns\n\ncolumns = ['Visit','Time','Init','Completion','AccV', 'AccML', 'AccAP','StartHesitation','Turn', 'Walking']\n\n# Create a new DataFrame containing only the selected columns\n\ndf_selected = merged_df[columns]\n\n# Calculate the correlation matrix\n\ncorr_matrix = df_selected.corr()\n\n# Visualize the correlation matrix with heatmap\n\nfig, ax = plt.subplots(figsize=(10, 10), dpi=100)\nsns.heatmap(corr_matrix, annot=True, cmap='Blues', fmt='.2f')","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:56:57.231638Z","iopub.execute_input":"2023-05-15T16:56:57.231940Z","iopub.status.idle":"2023-05-15T16:57:01.336867Z","shell.execute_reply.started":"2023-05-15T16:56:57.231915Z","shell.execute_reply":"2023-05-15T16:57:01.336116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For plotting time series graphs of AccV, AccML and AccAP accelerometer values, I used the Time column because it is a time scale containing information about the time for which the measurements were taken.\n\nThe Init and Completion columns indicate the start and end of a fade (FOG) episode, respectively. If we use these columns to plot the time series, then we will only see the time intervals when the fading episodes occurred, not the full time series.\n\nTherefore, to plot time series graphs based on accelerometer values ​​AccV, AccML and AccAP, it was logical to use the Time column, since it contains information about the time interval in which the accelerometer measurements were taken.","metadata":{}},{"cell_type":"code","source":"plt.rcParams['agg.path.chunksize'] = 200\n\nplt.figure(figsize=(15, 5))\nplt.subplot(131)\nplt.plot(df_selected['Time'], df_selected['AccV'], color='lightblue')\nplt.xlabel('Time')\nplt.ylabel('AccV')\n\nplt.subplot(132)\nplt.plot(df_selected['Time'], df_selected['AccML'],color='pink')\nplt.xlabel('Time')\nplt.ylabel('AccML')\n\nplt.subplot(133)\nplt.plot(df_selected['Time'], df_selected['AccAP'],color='lightgreen')\nplt.xlabel('Time')\nplt.ylabel('AccAP')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:57:01.338001Z","iopub.execute_input":"2023-05-15T16:57:01.338509Z","iopub.status.idle":"2023-05-15T16:57:38.124184Z","shell.execute_reply.started":"2023-05-15T16:57:01.338480Z","shell.execute_reply":"2023-05-15T16:57:38.123326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Sudden up or down jerks may indicate sudden changes in acceleration in the direction corresponding to the corresponding accelerometer column (eg AccV, AccML, AccAP). For example, a sharp drop down in the AccV chart could mean a sharp slowdown in the vertical direction.\n\n* Sharp horizontal tapers can indicate the start or end of fade episodes or other motion uncertainty events such as starting uncertainty or movement delays. For example, a sharp branch in the AccV graph may indicate the onset of an episode of starting uncertainty when the person is unable to start walking.\n\n\n* Other factors may also affect the shape and values ​​of the accelerometer time series, such as vibration, noise, and measurement errors. Therefore, before interpreting the graphs, it is necessary to make sure that the measurements are correct and accurate, and also take into account the context and features of a particular data set.","metadata":{}},{"cell_type":"markdown","source":"## Parse duration of FOG event types","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 3, figsize=(15, 5))\n\nax[0].hist(df_selected['StartHesitation'], color='lightblue')\nax[0].set_title('StartHesitation')\n\nax[1].hist(df_selected['Turn'], color='pink')\nax[1].set_title('Turn')\n\nax[2].hist(df_selected['Walking'], color='lightgreen')\nax[2].set_title('Walking')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:57:38.129369Z","iopub.execute_input":"2023-05-15T16:57:38.130128Z","iopub.status.idle":"2023-05-15T16:57:39.094519Z","shell.execute_reply.started":"2023-05-15T16:57:38.130098Z","shell.execute_reply":"2023-05-15T16:57:39.093319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select only the Init and Completion columns for each type of event\n\nstart_hesitation_df = merged_df[['Init', 'Completion', 'StartHesitation']]\nturn_df = merged_df[['Init', 'Completion', 'Turn']]\nwalking_df = merged_df[['Init', 'Completion', 'Walking']]","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:57:39.095928Z","iopub.execute_input":"2023-05-15T16:57:39.096211Z","iopub.status.idle":"2023-05-15T16:57:39.302220Z","shell.execute_reply.started":"2023-05-15T16:57:39.096184Z","shell.execute_reply":"2023-05-15T16:57:39.300532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the duration of each event\n\nstart_hesitation_df = start_hesitation_df.assign(Duration=start_hesitation_df['Completion'] - start_hesitation_df['Init'])\nturn_df = turn_df.assign(Duration=turn_df['Completion'] - turn_df['Init'])\nwalking_df = walking_df.assign(Duration=walking_df['Completion'] - walking_df['Init'])","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:57:39.303671Z","iopub.execute_input":"2023-05-15T16:57:39.304059Z","iopub.status.idle":"2023-05-15T16:57:39.598815Z","shell.execute_reply.started":"2023-05-15T16:57:39.304029Z","shell.execute_reply":"2023-05-15T16:57:39.597928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Start Hesitation Event Duration:\")\nprint(start_hesitation_df['Duration'].describe())\n\nprint(\"\\nTurn Event Duration:\")\nprint(turn_df['Duration'].describe())\n\nprint(\"\\nWalking Event Duration:\")\nprint(walking_df['Duration'].describe())","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:57:39.599988Z","iopub.execute_input":"2023-05-15T16:57:39.600211Z","iopub.status.idle":"2023-05-15T16:57:40.484674Z","shell.execute_reply.started":"2023-05-15T16:57:39.600191Z","shell.execute_reply":"2023-05-15T16:57:40.483837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build duration histograms for each type of event\n\nfig, axes = plt.subplots(nrows=1, ncols=3, figsize=(12,5))\nstart_hesitation_df.hist(column='Duration', ax=axes[0], color='lightblue')\nturn_df.hist(column='Duration', ax=axes[1], color='pink')\nwalking_df.hist(column='Duration', ax=axes[2], color='lightgreen')\n\n# Add titles and axis labels for each histogram\n\naxes[0].set_title('Start Hesitation Event Duration')\naxes[0].set_xlabel('Duration (seconds)')\naxes[0].set_ylabel('Frequency')\naxes[1].set_title('Turn Event Duration')\naxes[1].set_xlabel('Duration (seconds)')\naxes[1].set_ylabel('Frequency')\naxes[2].set_title('Walking Event Duration')\naxes[2].set_xlabel('Duration (seconds)')\naxes[2].set_ylabel('Frequency')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T16:57:40.485976Z","iopub.execute_input":"2023-05-15T16:57:40.486236Z","iopub.status.idle":"2023-05-15T16:57:42.299965Z","shell.execute_reply.started":"2023-05-15T16:57:40.486214Z","shell.execute_reply":"2023-05-15T16:57:42.299306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Based on the descriptive statistics provided, it can be concluded that the average duration and standard deviation for each type of event (Start Hesitation, Turn, Walking) are the same. In addition, the minimum and maximum values ​​are also the same for all three types of events.\n\nFrom these statistics, no conclusion can be drawn about the differences in duration between types of events. Perhaps the problem is in the distribution of the data itself or in the method of sampling.","metadata":{}},{"cell_type":"code","source":"merged_df.to_csv('EDA_Parkinson.csv', index=False)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-05-15T16:57:42.301083Z","iopub.execute_input":"2023-05-15T16:57:42.301534Z","iopub.status.idle":"2023-05-15T16:59:27.279228Z","shell.execute_reply.started":"2023-05-15T16:57:42.301501Z","shell.execute_reply":"2023-05-15T16:59:27.277750Z"},"trusted":true},"execution_count":null,"outputs":[]}]}