{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### **Intro**","metadata":{}},{"cell_type":"markdown","source":"This notebook will perform some basic EDA to get more insights about datasets. So, we will be using a combination of statistical measures and visualization techniques to gain a deeper understanding of the data and potentially uncover relationships and patterns that may be useful in developing a machine learning model for event detection.","metadata":{}},{"cell_type":"code","source":"!pip install ydata_profiling","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-05-01T22:09:26.309024Z","iopub.execute_input":"2023-05-01T22:09:26.310594Z","iopub.status.idle":"2023-05-01T22:09:38.906693Z","shell.execute_reply.started":"2023-05-01T22:09:26.310544Z","shell.execute_reply":"2023-05-01T22:09:38.904710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport ydata_profiling\nfrom ydata_profiling import ProfileReport\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom phik.phik import phik_matrix\nfrom phik.report import plot_correlation_matrix","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:09:38.916560Z","iopub.execute_input":"2023-05-01T22:09:38.917737Z","iopub.status.idle":"2023-05-01T22:09:40.356980Z","shell.execute_reply.started":"2023-05-01T22:09:38.917639Z","shell.execute_reply":"2023-05-01T22:09:40.355517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Loading the data","metadata":{}},{"cell_type":"code","source":"data_dir = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/'\n# \ndaily_metadata_file = f'{data_dir}daily_metadata.csv'\ndefog_metadata_file = f'{data_dir}defog_metadata.csv'\ntdcsfog_metadata_file = f'{data_dir}tdcsfog_metadata.csv'\nevents_data_file = f'{data_dir}events.csv'\nsubjects_data_file = f'{data_dir}subjects.csv'\ntasks_data_file = f'{data_dir}tasks.csv'\n\n# Read the meta data\ndaily_metadata = pd.read_csv(daily_metadata_file)\ndefog_metadata = pd.read_csv(defog_metadata_file)\ntdcsfog_metadata = pd.read_csv(tdcsfog_metadata_file)\n\nevents_data = pd.read_csv(events_data_file)\nsubjects_data = pd.read_csv(subjects_data_file)\ntasks_data = pd.read_csv(tasks_data_file)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:09:40.358761Z","iopub.execute_input":"2023-05-01T22:09:40.359480Z","iopub.status.idle":"2023-05-01T22:09:40.395740Z","shell.execute_reply.started":"2023-05-01T22:09:40.359439Z","shell.execute_reply":"2023-05-01T22:09:40.394404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## The basic `events` and `subjects` data exploring","metadata":{}},{"cell_type":"markdown","source":"### **`Events`** dataset","metadata":{}},{"cell_type":"code","source":"events_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:09:40.399295Z","iopub.execute_input":"2023-05-01T22:09:40.399705Z","iopub.status.idle":"2023-05-01T22:09:40.422685Z","shell.execute_reply.started":"2023-05-01T22:09:40.399667Z","shell.execute_reply":"2023-05-01T22:09:40.421027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create a barplot of the counts of events of each type \nsns.barplot(x=events_data.Type.value_counts().index,\n            y=events_data.Type.value_counts())\nplt.title('The barplot of the counts of events of each type')\nplt.ylabel('The number of FoG events')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:21:44.588914Z","iopub.execute_input":"2023-05-01T22:21:44.589549Z","iopub.status.idle":"2023-05-01T22:21:44.743577Z","shell.execute_reply.started":"2023-05-01T22:21:44.589485Z","shell.execute_reply":"2023-05-01T22:21:44.741593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The number of FoG events of `Turn` type is higher than other types. ","metadata":{}},{"cell_type":"code","source":"# create a barplot of the counts of kinetic/akinetic events\nfig, ax = plt.subplots()\nsns.barplot(x=events_data.Kinetic.value_counts().index,\n            y=events_data.Kinetic.value_counts())\nplt.title('The barplot of the counts of kinetic/akinetic events')\nplt.ylabel('The number of FoG events')\nax.set_xticklabels(['akinetic', 'kinetic'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:21:47.752209Z","iopub.execute_input":"2023-05-01T22:21:47.752651Z","iopub.status.idle":"2023-05-01T22:21:47.949962Z","shell.execute_reply.started":"2023-05-01T22:21:47.752603Z","shell.execute_reply":"2023-05-01T22:21:47.948739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most of the events in the dataset are `kinetic`.","metadata":{}},{"cell_type":"markdown","source":"Let's calculate the duration of each FoG event by subtracting the `Init` column (Time (s) the event began) from the `Completion` (Time (s) the event ended).","metadata":{}},{"cell_type":"code","source":"events_data['event_duration'] = events_data['Completion'] - events_data['Init']","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:09:40.863244Z","iopub.execute_input":"2023-05-01T22:09:40.863632Z","iopub.status.idle":"2023-05-01T22:09:40.872027Z","shell.execute_reply.started":"2023-05-01T22:09:40.863595Z","shell.execute_reply":"2023-05-01T22:09:40.870416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Distribution of FOG event duration for each type of event","metadata":{}},{"cell_type":"markdown","source":"As we've seen, there are three types of FoG events: StartHesitation, Turn, and Walking. Let's plot the distribution of the duration of FoG events for each type of FoG event and calculate the mean and standard deviation statistics. ","metadata":{}},{"cell_type":"code","source":"## Plot the distributions of FoG event durations for each Type of event\nplt.figure(figsize=(14, 12))\nfor i, column in enumerate(events_data.dropna().Type.unique(), 1):\n    plt.subplot(2, 2, i)\n    sns.histplot(events_data[events_data.Type == column].event_duration)\n    plt.xlabel(f\"Duration (s)\")\n    plt.ylabel('The number of FoG events')\n    plt.axvline(\n        events_data[events_data.Type == column].event_duration.mean(), linestyle='dashed',\n        label=f'mean = {events_data[events_data.Type == column].event_duration.mean():.2f}',\n        color='black')\n    plt.legend(loc='best')\n    plt.title(f\"The distribution of FoG duration for events of type '{column}' \\n \\\n(standard deviation = {events_data[events_data.Type == column].event_duration.std():.2f})\")\nplt.subplots_adjust(wspace=0.3, hspace=0.2)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:09:40.874025Z","iopub.execute_input":"2023-05-01T22:09:40.874630Z","iopub.status.idle":"2023-05-01T22:09:43.676273Z","shell.execute_reply.started":"2023-05-01T22:09:40.874569Z","shell.execute_reply":"2023-05-01T22:09:43.675070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the boxplots of FoG event durations for each event Type\ndata_1 = events_data[events_data.Type == 'Turn'].event_duration\ndata_2 = events_data[events_data.Type == 'Walking'].event_duration\ndata_3 = events_data[events_data.Type == 'StartHesitation'].event_duration\n\nfig, ax = plt.subplots()\nsns.boxplot([data_1, data_2, data_3], orient='h', palette='Set3')\nplt.xlabel(f\"Duration (s)\")\nplt.ylabel('Event Type')\nax.set_yticklabels(['Turn', 'Walking', 'StartHesitation'])\nplt.title(f\"The boxplots of FoG event durations for each event Type\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:09:43.677806Z","iopub.execute_input":"2023-05-01T22:09:43.678424Z","iopub.status.idle":"2023-05-01T22:09:43.943922Z","shell.execute_reply.started":"2023-05-01T22:09:43.678385Z","shell.execute_reply":"2023-05-01T22:09:43.941563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see, the distributions of FoG event durations for all event Types are highly right-skewed with a large number of outliers, and most FoG episodes lasting less than 30 seconds.","metadata":{}},{"cell_type":"markdown","source":"### Statistics of event durations for each Type","metadata":{}},{"cell_type":"code","source":"for type_name in events_data.dropna().Type.unique():\n    print('Type: ', type_name)\n    print(events_data[events_data.Type == type_name].event_duration.agg(\n                                    ['min', 'max', 'median', 'mean', 'std']))\n    print(f'The value of 95th percentile: \\\n{events_data[events_data.Type == type_name].event_duration.quantile(0.95):.2f}')\n    print('-'* 10 + '\\n')","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:09:43.946483Z","iopub.execute_input":"2023-05-01T22:09:43.947146Z","iopub.status.idle":"2023-05-01T22:09:43.991760Z","shell.execute_reply.started":"2023-05-01T22:09:43.947084Z","shell.execute_reply":"2023-05-01T22:09:43.989929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The duration of FoG events of type **`Turn`** ranged from 0.0 to 581.98 seconds (mean = 8.83 s, standard deviation = 28.09 s). The 50% of this FoG episodes lasted less than 2.68 seconds, whereas the majority (95%) lasted less than 26.26 seconds.\n\nThe duration of FoG events of type **`Walking`** ranged from 0.003 to 164.20 seconds (mean = 6.29 s, standard deviation = 17.32 s). The 50% of this FoG episodes lasted less than 2.01 seconds, whereas the majority (95%) lasted less than 16.80 seconds.\n\nThe duration of FoG events of type **`StartHesitation`** ranged from 0.24 to 151.24 seconds (mean = 22.34 s, standard deviation = 37.79 s). The 50% of this FoG episodes lasted less than 3.32 seconds, whereas the majority (95%) lasted less than 120.58 seconds. ","metadata":{}},{"cell_type":"markdown","source":"### **`Subjects`** dataset","metadata":{}},{"cell_type":"code","source":"ProfileReport(subjects_data, title='Subjects Dataset Report',\n              progress_bar=False,\n              interactions=None,\n              explorative=True, dark_mode=True,\n              notebook={'iframe':{'height': '600px'}},\n              missing_diagrams={'heatmap': False, 'dendrogram': False}).to_notebook_iframe()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:09:43.994571Z","iopub.execute_input":"2023-05-01T22:09:43.996547Z","iopub.status.idle":"2023-05-01T22:09:49.781383Z","shell.execute_reply.started":"2023-05-01T22:09:43.996456Z","shell.execute_reply":"2023-05-01T22:09:49.779881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As can be seen from the profiling report, the number of male subjects is greater than female subjects. The age column seems to have a normal distribution based on the histogram, skewness and kurtosis values. The majority of patients in our dataset are aged between 56 and 79 years old (the average age is around 68).\nColumn `UPDRSIII_On` is highly correlated with `UPDRSIII_Off` column. Columns `Visit` and `UPDRSIII_Off` have missing values. `NFOGQ` column has 12.1% zeros.","metadata":{}},{"cell_type":"markdown","source":"Let's concatenate the`tdcsfog` metadata with the `defog` metadata, and then merge the result with the `subjects` dataset for further analysis.","metadata":{}},{"cell_type":"code","source":"tdcsfog_metadata['dataset'] = 'tdcsfog'\ndefog_metadata['dataset'] = 'defog'\n\nfull_metadata = pd.concat([tdcsfog_metadata, defog_metadata])","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:09:49.783722Z","iopub.execute_input":"2023-05-01T22:09:49.784360Z","iopub.status.idle":"2023-05-01T22:09:49.797700Z","shell.execute_reply.started":"2023-05-01T22:09:49.784278Z","shell.execute_reply":"2023-05-01T22:09:49.796455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_metadata_df = events_data[['Id', 'Type', 'Kinetic', 'event_duration']] \\\n                            .merge(full_metadata, on='Id') \\\n                            .merge(subjects_data.drop('Visit', axis=1), on='Subject')\nall_metadata_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:09:49.805510Z","iopub.execute_input":"2023-05-01T22:09:49.805949Z","iopub.status.idle":"2023-05-01T22:09:49.846709Z","shell.execute_reply.started":"2023-05-01T22:09:49.805909Z","shell.execute_reply":"2023-05-01T22:09:49.845339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_metadata_df.Sex = np.where(all_metadata_df.Sex == 'F', 0, 1)\nall_metadata_df.Medication = np.where(all_metadata_df.Medication == 'off', 0, 1)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:09:49.848440Z","iopub.execute_input":"2023-05-01T22:09:49.848823Z","iopub.status.idle":"2023-05-01T22:09:49.863293Z","shell.execute_reply.started":"2023-05-01T22:09:49.848787Z","shell.execute_reply":"2023-05-01T22:09:49.861519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nmask = np.triu(np.ones_like(all_metadata_df.corr(), dtype=bool))\nsns.heatmap(all_metadata_df.corr(), mask=mask, annot=True, cmap='coolwarm',\n                linewidths=0.2, annot_kws={\"size\": 7}, rasterized=True)\nplt.xticks(rotation=55)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:09:49.865708Z","iopub.execute_input":"2023-05-01T22:09:49.866418Z","iopub.status.idle":"2023-05-01T22:09:50.458668Z","shell.execute_reply.started":"2023-05-01T22:09:49.866344Z","shell.execute_reply":"2023-05-01T22:09:50.457189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As our dataset contain a mix of different variable types and Pearson's correlation coefficients are not appropriate for mixed data types, we will use PhiK correlation that can provide a more accurate picture of the relationship between our variables.","metadata":{}},{"cell_type":"code","source":"columns_phik = ['Type', 'event_duration', 'Age', 'UPDRSIII_On', 'YearsSinceDx',\n                'UPDRSIII_Off', 'NFOGQ', 'Sex', 'Medication', 'Kinetic', 'Visit']\nphik_overview = all_metadata_df[columns_phik].phik_matrix( \\\n                                    interval_cols=['event_duration', 'Age',\n                                                'UPDRSIII_On', 'YearsSinceDx',\n                                                'UPDRSIII_Off','NFOGQ'])\n\nplot_correlation_matrix(phik_overview.values, x_labels=phik_overview.columns,\n                        y_labels=phik_overview.index, vmin=0, vmax=1,\n                        color_map='Blues', title=r'correlation matrix $\\phi_K$',\n                        figsize=(9, 6))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:09:50.460127Z","iopub.execute_input":"2023-05-01T22:09:50.460546Z","iopub.status.idle":"2023-05-01T22:09:54.203484Z","shell.execute_reply.started":"2023-05-01T22:09:50.460506Z","shell.execute_reply":"2023-05-01T22:09:54.202390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's visualize the counts of FoG events for different age ranges in the dataset. This will help us to understand if there is any correlation between age and the frequency of FoG events. We can use the `pd.qcut()` function to bin the age column into different age ranges, then count the number of FoG events in each range and plot it.","metadata":{}},{"cell_type":"code","source":"# bin the Age column based on quantiles\nall_metadata_df['age_bin'] = pd.qcut(all_metadata_df['Age'], q=5)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:09:54.204830Z","iopub.execute_input":"2023-05-01T22:09:54.206128Z","iopub.status.idle":"2023-05-01T22:09:54.218287Z","shell.execute_reply.started":"2023-05-01T22:09:54.206072Z","shell.execute_reply":"2023-05-01T22:09:54.216863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the frequency of FoG events of each Type for different age ranges\nplt.figure(figsize=(15, 14))\nfor i, age_bin in enumerate(all_metadata_df.age_bin.unique(), 1):\n    plt.subplot(3, 3, i)\n    sns.barplot(\n        x=all_metadata_df[all_metadata_df.age_bin == age_bin].Type\n                                        .value_counts(normalize=True).index,\n        y=all_metadata_df[all_metadata_df.age_bin == age_bin].Type\n                                        .value_counts(normalize=True))\n    plt.ylabel('The frequency of FoG events')\n    plt.title(f\"Age in range {age_bin}\")\n    \n    # Add frequency values on top of the bars\n    for index, value in enumerate(\n                all_metadata_df[all_metadata_df.age_bin == age_bin].Type\n                                            .value_counts(normalize=True)):\n        plt.annotate(f\"{value * 100:.1f}%\", xy=(index, value),\n                                    ha='center', va='bottom')\nplt.subplots_adjust(wspace=0.3, hspace=0.2)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:09:54.219774Z","iopub.execute_input":"2023-05-01T22:09:54.220531Z","iopub.status.idle":"2023-05-01T22:09:55.011081Z","shell.execute_reply.started":"2023-05-01T22:09:54.220480Z","shell.execute_reply":"2023-05-01T22:09:55.009332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next we will plot the frequency of FoG events of each Type for male and female patients in the dataset.","metadata":{}},{"cell_type":"code","source":"# Plot the frequency of FoG events of each Type for male and female patients\nplt.figure(figsize=(12, 5))\ngender_dict = {0: 'Female', 1: 'Male'}\nfor i, gender in enumerate(all_metadata_df.Sex.unique(), 1):\n    plt.subplot(1, 2, i)\n    sns.barplot(\n        x=all_metadata_df[all_metadata_df.Sex == gender].Type\n                                .value_counts(normalize=True).index,\n        y=all_metadata_df[all_metadata_df.Sex == gender].Type\n                                .value_counts(normalize=True))\n    plt.ylabel('The frequency of FoG events')\n    plt.title(f\"Gender = {gender_dict[gender]}\")\n    \n    # Add frequency values on top of the bars\n    for index, value in enumerate(\n            all_metadata_df[all_metadata_df.Sex == gender].Type\n                                    .value_counts(normalize=True)):\n        plt.annotate(f\"{value * 100:.1f}%\", xy=(index, value),\n                                    ha='center', va='bottom')\nplt.subplots_adjust(wspace=0.4)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:09:55.012913Z","iopub.execute_input":"2023-05-01T22:09:55.014127Z","iopub.status.idle":"2023-05-01T22:09:55.363269Z","shell.execute_reply.started":"2023-05-01T22:09:55.014058Z","shell.execute_reply":"2023-05-01T22:09:55.361720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the counts of FoG events for different age ranges\nplt.figure(figsize=(12, 5))\nmed_dict = {0: 'Medication status: off', 1: 'Medication status: on'}\nfor i, med_status in enumerate(all_metadata_df.Medication.unique(), 1):\n    plt.subplot(1, 2, i)\n    sns.barplot(\n        x=all_metadata_df[all_metadata_df.Medication == med_status].Type\n                                        .value_counts(normalize=True).index,\n        y=all_metadata_df[all_metadata_df.Medication == med_status].Type\n                                        .value_counts(normalize=True))\n    plt.ylabel('The frequency of FoG events')\n    plt.title(f\"{med_dict[med_status]}\")\n    \n    # Add frequency values on top of the bars\n    for index, value in enumerate(\n            all_metadata_df[all_metadata_df.Medication == med_status].Type\n                                            .value_counts(normalize=True)):\n        plt.annotate(f\"{value * 100:.1f}%\", xy=(index, value),\n                                    ha='center', va='bottom')\nplt.subplots_adjust(wspace=0.4)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:09:55.365859Z","iopub.execute_input":"2023-05-01T22:09:55.366297Z","iopub.status.idle":"2023-05-01T22:09:55.764407Z","shell.execute_reply.started":"2023-05-01T22:09:55.366255Z","shell.execute_reply":"2023-05-01T22:09:55.762711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## The train Datasets exploring","metadata":{}},{"cell_type":"markdown","source":"### The lab data (`tdcsfog` Dataset)","metadata":{}},{"cell_type":"markdown","source":"The tDCS FOG (tdcsfog) dataset comprises data series collected in the lab, as subjects completed a FOG-provoking protocol.\n\nLet's select only 500 files from the `tdcsfog` folder and take a look at the dataset profile report and correlation matrix.","metadata":{}},{"cell_type":"code","source":"# Read 500 files of the tdcsfog metadata (the total number of files is 833)\ntdcsfog_files = [f'{data_dir}train/tdcsfog/{id}.csv' for id in \\\n                                         tdcsfog_metadata.Id.to_list()[:501]]\n\ntdcsfog_data = pd.concat([pd.read_csv(file) for file in tdcsfog_files])","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:09:55.765616Z","iopub.execute_input":"2023-05-01T22:09:55.765980Z","iopub.status.idle":"2023-05-01T22:10:02.213434Z","shell.execute_reply.started":"2023-05-01T22:09:55.765944Z","shell.execute_reply":"2023-05-01T22:10:02.211955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ProfileReport(tdcsfog_data, title='Tdcsfog Dataset Report',\n              progress_bar=False,\n              interactions=None,\n              explorative=True, dark_mode=True,\n              notebook={'iframe':{'height': '600px'}},\n              missing_diagrams={'heatmap': False, 'dendrogram': False}).to_notebook_iframe()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:10:02.215679Z","iopub.execute_input":"2023-05-01T22:10:02.216097Z","iopub.status.idle":"2023-05-01T22:11:51.525846Z","shell.execute_reply.started":"2023-05-01T22:10:02.216060Z","shell.execute_reply":"2023-05-01T22:11:51.524310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask = np.triu(np.ones_like(tdcsfog_data.corr(), dtype=bool))\nfig, ax = plt.subplots(figsize=(7, 5))\nsns.heatmap(tdcsfog_data.corr(), mask=mask, annot=True, cmap='coolwarm',\n                linewidths=0.2, annot_kws={\"size\": 7}, rasterized=True)\nplt.xticks(rotation=35)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:11:51.527658Z","iopub.execute_input":"2023-05-01T22:11:51.528047Z","iopub.status.idle":"2023-05-01T22:11:53.398555Z","shell.execute_reply.started":"2023-05-01T22:11:51.528007Z","shell.execute_reply":"2023-05-01T22:11:53.397167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"phik_overview = tdcsfog_data.phik_matrix(\n                            interval_cols=['AccV', 'AccML', 'AccAP','Time'])\n\nplot_correlation_matrix(phik_overview.values, x_labels=phik_overview.columns,\n                        y_labels=phik_overview.index, vmin=0, vmax=1, \n                        color_map='Blues', title=r'correlation matrix $\\phi_K$',\n                        figsize=(7, 5))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:11:53.400337Z","iopub.execute_input":"2023-05-01T22:11:53.400832Z","iopub.status.idle":"2023-05-01T22:12:15.308405Z","shell.execute_reply.started":"2023-05-01T22:11:53.400780Z","shell.execute_reply":"2023-05-01T22:12:15.306914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From dataset report and correlation matrix, it can be concluded that:\n\n  - There are no missing values and duplicate rows detected in the dataset.\n  - As can be seen from the profiling report, all target variables are highly imbalanced, especially `StartHesitation` (78.5%) and `Walking` (80.1%).\n  - Based on histograms and skewness values the distributions of the `AccAP` and `AccV` columns are moderately left-skewed, the `AccML` column seems to have a close to normal distribution.\n  - The `AccAP` column have a kurtosis value of less than 3, which indicates that the column is platikurtic. Meanwhile, the `AccV` column has a kurtosis value of more than 3, which indicates that the column is leptokurtic. And  a kurtosis value of the`AccML` column is close to 3, which is recognized as mesokurtic column.\n  - As can be seen from Pearson's and Phik correlation matrices `Time` column have a moderate positive correlation with two target variables `Turn`, `Walking` and variable of anteroposterior acceleration measurements. However, between the target variable `StartHesitation` and other variables there is no strong or moderate correlation.","metadata":{}},{"cell_type":"markdown","source":"#### Plotting the acceleration measurements for events of each Type","metadata":{}},{"cell_type":"code","source":"# Read one file of the tdcsfog train data with event of Turn type\ntdcsfog_files_turn = [f'{data_dir}train/tdcsfog/{id}.csv' for id in \\\n                              tdcsfog_metadata.Id.to_list()[16:17]]\n\ntdcsfog_data_turn = pd.concat([pd.read_csv(file) for file in tdcsfog_files_turn])\n\n# Read one file of the tdcsfog train data with event of Walking type\ntdcsfog_files_walk = [f'{data_dir}train/tdcsfog/{id}.csv' for id in \\\n                              tdcsfog_metadata.Id.to_list()[19:20]]\n\ntdcsfog_data_walk = pd.concat([pd.read_csv(file) for file in tdcsfog_files_walk])\n\n# Read one file of the tdcsfog train data with event of StartHesitation type\ntdcsfog_files_hes = [f'{data_dir}train/tdcsfog/{id}.csv' for id in \\\n                             tdcsfog_metadata.Id.to_list()[77:78]]\n\ntdcsfog_data_hes = pd.concat([pd.read_csv(file) for file in tdcsfog_files_hes])","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:12:15.310325Z","iopub.execute_input":"2023-05-01T22:12:15.311619Z","iopub.status.idle":"2023-05-01T22:12:15.408092Z","shell.execute_reply.started":"2023-05-01T22:12:15.311556Z","shell.execute_reply":"2023-05-01T22:12:15.406563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since the series in the `tdcsfog` dataset are recorded at a frequency of 128Hz (128 time steps per second), we can plot the acceleration measurements for each axis (V - vertical, ML - mediolateral, AP - anteroposterior) using a rolling mean method with a window size of 128. This helps to smooth out fluctuations in the data and provides a better representation of the overall trend of the acceleration measurements over time.","metadata":{}},{"cell_type":"markdown","source":"Let's plot the overall trends of acceleration measurements for events of each Type.","metadata":{}},{"cell_type":"code","source":"# Plot the overall trends of acceleration measurements for events of each Type\nplt.figure(figsize=(14, 12))\n\ntypes_dict = {1: 'Turn', 2: 'Walking', 3: 'StartHesitation'}\n\nfor i, data in enumerate(\n            [tdcsfog_data_turn, tdcsfog_data_walk, tdcsfog_data_hes], 1):\n    plt.subplot(2, 2, i)\n    window_size = 128\n    accv_rolling_mean = data[\"AccV\"].rolling(window_size).mean()\n    accap_rolling_mean = data[\"AccAP\"].rolling(window_size).mean()\n    accml_rolling_mean = data[\"AccML\"].rolling(window_size).mean()\n\n    sns.lineplot(x=data.Time, y=accv_rolling_mean)\n    sns.lineplot(x=data.Time, y=accap_rolling_mean)\n    sns.lineplot(x=data.Time, y=accml_rolling_mean)\n    plt.title(f'The trend of acceleration measurements (\"{types_dict[i]}\" event)')\n    plt.xlabel('Time')\n    plt.ylabel('Acceleration value (m/s^2)')\n\nplt.legend(['Anteroposterior', 'Mediolateral', 'Vertical'], loc='upper left',\n               bbox_to_anchor=(2.3, 2.1), handlelength=0,\n               shadow=True, labelcolor=['orange', 'green', 'blue'])\nplt.subplots_adjust(wspace=0.3, hspace=0.2)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:12:15.411720Z","iopub.execute_input":"2023-05-01T22:12:15.412655Z","iopub.status.idle":"2023-05-01T22:12:16.571560Z","shell.execute_reply.started":"2023-05-01T22:12:15.412593Z","shell.execute_reply":"2023-05-01T22:12:16.570117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### The home data (`DeFOG` Dataset)","metadata":{}},{"cell_type":"markdown","source":"The DeFOG (defog) dataset comprises data series collected in the subject's home, as subjects completed a FOG-provoking protocol.\n\nAs we know from the data descriptions, the series in the `notype` folder are from the DeFOG dataset but lack event-type annotations. Let's create a list of indices that correspond to the notype series IDs. Using this list, we can separate the `defog_metadata` set into two metadata sets: one for notype and one for defog series.","metadata":{}},{"cell_type":"code","source":"# Creating a list of indices that correspond to the notype series IDs in the defog_metadata\nnotype_idx = [0, 4, 13, 15, 17, 19, 20, 22, 25, 27, 30, 31, 32, 34, 36, 40, 42,\n              53, 54, 55, 58, 60, 62, 66, 67, 73, 74, 76, 79, 83, 84, 88, 89, 94,\n              96, 97, 98, 101, 102, 105, 113, 114, 115, 121, 124, 126]\n\n# Separating the defog_metadata\ndefog_metadata_only = defog_metadata[~defog_metadata.index.isin(notype_idx)].copy()\nnotype_metadata = defog_metadata[defog_metadata.index.isin(notype_idx)].copy()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:12:16.573484Z","iopub.execute_input":"2023-05-01T22:12:16.576614Z","iopub.status.idle":"2023-05-01T22:12:16.585805Z","shell.execute_reply.started":"2023-05-01T22:12:16.576552Z","shell.execute_reply":"2023-05-01T22:12:16.584397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read one file of the defog metadata (the total number of files is 91)\ndefog_files = [f'{data_dir}train/defog/{id}.csv' for id in \\\n                               defog_metadata_only.Id.to_list()[:1]]\n\ndefog_data = pd.concat([pd.read_csv(file) for file in defog_files])","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:12:16.587472Z","iopub.execute_input":"2023-05-01T22:12:16.587873Z","iopub.status.idle":"2023-05-01T22:12:16.726798Z","shell.execute_reply.started":"2023-05-01T22:12:16.587833Z","shell.execute_reply":"2023-05-01T22:12:16.725237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.concat([defog_metadata[defog_metadata.Id == '02ea782681'],\n           events_data[events_data.Id == '02ea782681']], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:12:16.728715Z","iopub.execute_input":"2023-05-01T22:12:16.729096Z","iopub.status.idle":"2023-05-01T22:12:16.754002Z","shell.execute_reply.started":"2023-05-01T22:12:16.729055Z","shell.execute_reply":"2023-05-01T22:12:16.752817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_data_valid = defog_data.loc[ \\\n                        (defog_data.Valid == True) & (defog_data.Task == True)]","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:12:16.755391Z","iopub.execute_input":"2023-05-01T22:12:16.755773Z","iopub.status.idle":"2023-05-01T22:12:16.772976Z","shell.execute_reply.started":"2023-05-01T22:12:16.755735Z","shell.execute_reply":"2023-05-01T22:12:16.771430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_data_valid.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:12:16.774787Z","iopub.execute_input":"2023-05-01T22:12:16.775534Z","iopub.status.idle":"2023-05-01T22:12:16.784804Z","shell.execute_reply.started":"2023-05-01T22:12:16.775488Z","shell.execute_reply":"2023-05-01T22:12:16.783777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Ploting the acceleration measurements for each axis","metadata":{}},{"cell_type":"markdown","source":"Since the series in the `defog` dataset are recorded at a frequency of 100Hz (100 time steps per second), we can plot the smoothed acceleration measurements for each axis (V - vertical, ML - mediolateral, AP - anteroposterior) using a rolling window technique with a window size of 100 and see the overall trend of acceleration measurements for each axis.","metadata":{}},{"cell_type":"code","source":"# Plot the overall trend of acceleration measurements for each axis\nplt.figure(figsize=(14, 5))\n\ndefogdata_dict = {1: 'All defogdata', 2: 'Valid defog data'}\n\nfor i, data in enumerate([defog_data, defog_data_valid], 1):\n    plt.subplot(1, 2, i)\n    window_size = 100\n    accv_rolling_mean = data[\"AccV\"].rolling(window_size).mean()\n    accap_rolling_mean = data[\"AccAP\"].rolling(window_size).mean()\n    accml_rolling_mean = data[\"AccML\"].rolling(window_size).mean()\n\n    sns.lineplot(x=data.Time, y=accv_rolling_mean)\n    sns.lineplot(x=data.Time, y=accap_rolling_mean)\n    sns.lineplot(x=data.Time, y=accml_rolling_mean)\n    plt.title(f'The trend of acceleration measurements ({defogdata_dict[i]})')\n    plt.xlabel('Time')\n    plt.ylabel('Acceleration value (g)')\n\nplt.legend(['Mediolateral', 'Anteroposterior', 'Vertical'], loc='upper left',\n               bbox_to_anchor=(1.05, 1), shadow=True, handlelength=0,\n               labelcolor=['green', 'orange', 'blue'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:12:16.786207Z","iopub.execute_input":"2023-05-01T22:12:16.786800Z","iopub.status.idle":"2023-05-01T22:12:18.388782Z","shell.execute_reply.started":"2023-05-01T22:12:16.786764Z","shell.execute_reply":"2023-05-01T22:12:18.387438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### The daily data (`Daily Living` dataset)","metadata":{}},{"cell_type":"markdown","source":"The Daily Living (daily) dataset comprises one week of continuous 24/7 recordings from sixty-five subjects. Forty-five subjects exhibit FOG symptoms and also have series in the defog dataset, while the other twenty subjects do not exhibit FOG symptoms and do not have series elsewhere in the data.\n","metadata":{}},{"cell_type":"code","source":"# Read one file of the daily metadata (the total number of files is 65)\ndaily_files = [f'{data_dir}unlabeled/{id}.parquet' for id in \\\n                                       daily_metadata.Id.to_list()[:1]]\n\ndaily_data = pd.concat([pd.read_parquet(file) for file in daily_files])","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:12:18.390328Z","iopub.execute_input":"2023-05-01T22:12:18.390719Z","iopub.status.idle":"2023-05-01T22:12:24.900939Z","shell.execute_reply.started":"2023-05-01T22:12:18.390681Z","shell.execute_reply":"2023-05-01T22:12:24.899898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_data.describe(include='float64').T","metadata":{"execution":{"iopub.status.busy":"2023-05-01T22:12:24.902150Z","iopub.execute_input":"2023-05-01T22:12:24.902996Z","iopub.status.idle":"2023-05-01T22:12:24.946442Z","shell.execute_reply.started":"2023-05-01T22:12:24.902956Z","shell.execute_reply":"2023-05-01T22:12:24.945164Z"},"trusted":true},"execution_count":null,"outputs":[]}]}