{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install ydata_profiling","metadata":{"execution":{"iopub.status.busy":"2023-05-07T12:22:50.291297Z","iopub.execute_input":"2023-05-07T12:22:50.291632Z","iopub.status.idle":"2023-05-07T12:23:03.511517Z","shell.execute_reply.started":"2023-05-07T12:22:50.291605Z","shell.execute_reply":"2023-05-07T12:23:03.510345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport ydata_profiling\nfrom ydata_profiling import ProfileReport\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom phik.phik import phik_matrix\nfrom phik.report import plot_correlation_matrix","metadata":{"execution":{"iopub.status.busy":"2023-05-07T12:23:19.033814Z","iopub.execute_input":"2023-05-07T12:23:19.034216Z","iopub.status.idle":"2023-05-07T12:23:22.047454Z","shell.execute_reply.started":"2023-05-07T12:23:19.034174Z","shell.execute_reply":"2023-05-07T12:23:22.046581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/'\n# \ndaily_metadata_file = f'{data_dir}daily_metadata.csv'\ndefog_metadata_file = f'{data_dir}defog_metadata.csv'\ntdcsfog_metadata_file = f'{data_dir}tdcsfog_metadata.csv'\nevents_data_file = f'{data_dir}events.csv'\nsubjects_data_file = f'{data_dir}subjects.csv'\ntasks_data_file = f'{data_dir}tasks.csv'\n\n# Read the meta data\ndaily_metadata = pd.read_csv(daily_metadata_file)\ndefog_metadata = pd.read_csv(defog_metadata_file)\ntdcsfog_metadata = pd.read_csv(tdcsfog_metadata_file)\n\nevents_data = pd.read_csv(events_data_file)\nsubjects_data = pd.read_csv(subjects_data_file)\ntasks_data = pd.read_csv(tasks_data_file)","metadata":{"execution":{"iopub.status.busy":"2023-05-07T12:28:13.739346Z","iopub.execute_input":"2023-05-07T12:28:13.739755Z","iopub.status.idle":"2023-05-07T12:28:13.808853Z","shell.execute_reply.started":"2023-05-07T12:28:13.739723Z","shell.execute_reply":"2023-05-07T12:28:13.808022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## The basic \"events\" and \"subjects\" data exploring!","metadata":{}},{"cell_type":"markdown","source":"### Events dataset","metadata":{}},{"cell_type":"code","source":"events_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T12:30:57.283754Z","iopub.execute_input":"2023-05-07T12:30:57.284554Z","iopub.status.idle":"2023-05-07T12:30:57.320815Z","shell.execute_reply.started":"2023-05-07T12:30:57.284508Z","shell.execute_reply":"2023-05-07T12:30:57.319706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x = events_data.Type.value_counts().index,\n           y = events_data.Type.value_counts())\nplt.title(\"The barplot of the counts of events of each type!\")\nplt.ylabel(\"The number of FoG events!\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T12:37:07.002352Z","iopub.execute_input":"2023-05-07T12:37:07.002728Z","iopub.status.idle":"2023-05-07T12:37:07.266750Z","shell.execute_reply.started":"2023-05-07T12:37:07.002698Z","shell.execute_reply":"2023-05-07T12:37:07.265687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### The number of FoG events of \"Turn\" type is higher than other types.","metadata":{}},{"cell_type":"code","source":"# create a barplot of the counts of kinetic/akinetic events\nfig, ax = plt.subplots()\nsns.barplot(x=events_data.Kinetic.value_counts().index,\n            y=events_data.Kinetic.value_counts())\nplt.title('The barplot of the counts of kinetic/akinetic events')\nplt.ylabel('The number of FoG events')\nax.set_xticklabels(['akinetic', 'kinetic'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T12:39:47.575979Z","iopub.execute_input":"2023-05-07T12:39:47.576362Z","iopub.status.idle":"2023-05-07T12:39:47.810979Z","shell.execute_reply.started":"2023-05-07T12:39:47.576333Z","shell.execute_reply":"2023-05-07T12:39:47.809951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"events_data['event_duration'] = events_data['Completion'] - events_data['Init']","metadata":{"execution":{"iopub.status.busy":"2023-05-07T12:41:30.274861Z","iopub.execute_input":"2023-05-07T12:41:30.275251Z","iopub.status.idle":"2023-05-07T12:41:30.281442Z","shell.execute_reply.started":"2023-05-07T12:41:30.275222Z","shell.execute_reply":"2023-05-07T12:41:30.280458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Plot the distributions of FoG event durations for each Type of event\nplt.figure(figsize=(14, 12))\nfor i, column in enumerate(events_data.dropna().Type.unique(), 1):\n    plt.subplot(2, 2, i)\n    sns.histplot(events_data[events_data.Type == column].event_duration)\n    plt.xlabel(f\"Duration (s)\")\n    plt.ylabel('The number of FoG events')\n    plt.axvline(\n        events_data[events_data.Type == column].event_duration.mean(), linestyle='dashed',\n        label=f'mean = {events_data[events_data.Type == column].event_duration.mean():.2f}',\n        color='black')\n    plt.legend(loc='best')\n    plt.title(f\"The distribution of FoG duration for events of type '{column}' \\n \\\n(standard deviation = {events_data[events_data.Type == column].event_duration.std():.2f})\")\nplt.subplots_adjust(wspace=0.3, hspace=0.2)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T12:46:48.754171Z","iopub.execute_input":"2023-05-07T12:46:48.754605Z","iopub.status.idle":"2023-05-07T12:46:51.481077Z","shell.execute_reply.started":"2023-05-07T12:46:48.754573Z","shell.execute_reply":"2023-05-07T12:46:51.479945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the boxplots of FoG event durations for each event Type\ndata_1 = events_data[events_data.Type == 'Turn'].event_duration\ndata_2 = events_data[events_data.Type == 'Walking'].event_duration\ndata_3 = events_data[events_data.Type == 'StartHesitation'].event_duration\n\nfig, ax = plt.subplots()\nsns.boxplot([data_1, data_2, data_3], orient='h', palette='Set3')\nplt.xlabel(f\"Duration (s)\")\nplt.ylabel('Event Type')\nax.set_yticklabels(['Turn', 'Walking', 'StartHesitation'])\nplt.title(f\"The boxplots of FoG event durations for each event Type\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T12:50:36.597601Z","iopub.execute_input":"2023-05-07T12:50:36.598010Z","iopub.status.idle":"2023-05-07T12:50:36.781171Z","shell.execute_reply.started":"2023-05-07T12:50:36.597978Z","shell.execute_reply":"2023-05-07T12:50:36.780107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for type_name in events_data.dropna().Type.unique():\n    print('Type: ', type_name)\n    print(events_data[events_data.Type == type_name].event_duration.agg(\n                                    ['min', 'max', 'median', 'mean', 'std']))\n    print(f'The value of 95th percentile: \\\n{events_data[events_data.Type == type_name].event_duration.quantile(0.95):.2f}')\n    print('-'* 10 + '\\n')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T12:53:38.176170Z","iopub.execute_input":"2023-05-07T12:53:38.176587Z","iopub.status.idle":"2023-05-07T12:53:38.202147Z","shell.execute_reply.started":"2023-05-07T12:53:38.176556Z","shell.execute_reply":"2023-05-07T12:53:38.201032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The duration of FoG events of type **`Turn`** ranged from 0.0 to 581.98 seconds (mean = 8.83 s, standard deviation = 28.09 s). The 50% of this FoG episodes lasted less than 2.68 seconds, whereas the majority (95%) lasted less than 26.26 seconds.\n\nThe duration of FoG events of type **`Walking`** ranged from 0.003 to 164.20 seconds (mean = 6.29 s, standard deviation = 17.32 s). The 50% of this FoG episodes lasted less than 2.01 seconds, whereas the majority (95%) lasted less than 16.80 seconds.\n\nThe duration of FoG events of type **`StartHesitation`** ranged from 0.24 to 151.24 seconds (mean = 22.34 s, standard deviation = 37.79 s). The 50% of this FoG episodes lasted less than 3.32 seconds, whereas the majority (95%) lasted less than 120.58 seconds. ","metadata":{}},{"cell_type":"code","source":"ProfileReport(subjects_data, title='Subjects Dataset Report',\n              progress_bar=False,\n              interactions=None,\n              explorative=True, dark_mode=True,\n              notebook={'iframe':{'height': '600px'}},\n              missing_diagrams={'heatmap': False, 'dendrogram': False}).to_notebook_iframe()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T12:55:43.466669Z","iopub.execute_input":"2023-05-07T12:55:43.467055Z","iopub.status.idle":"2023-05-07T12:55:57.004793Z","shell.execute_reply.started":"2023-05-07T12:55:43.467027Z","shell.execute_reply":"2023-05-07T12:55:57.003745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_metadata['dataset'] = 'tdcsfog'\ndefog_metadata['dataset'] = 'defog'\n\nfull_metadata = pd.concat([tdcsfog_metadata, defog_metadata])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T12:56:49.237474Z","iopub.execute_input":"2023-05-07T12:56:49.238175Z","iopub.status.idle":"2023-05-07T12:56:49.246713Z","shell.execute_reply.started":"2023-05-07T12:56:49.238140Z","shell.execute_reply":"2023-05-07T12:56:49.245398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_metadata_df = events_data[['Id', 'Type', 'Kinetic', 'event_duration']] \\\n                            .merge(full_metadata, on='Id') \\\n                            .merge(subjects_data.drop('Visit', axis=1), on='Subject')\nall_metadata_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T12:57:56.939084Z","iopub.execute_input":"2023-05-07T12:57:56.939480Z","iopub.status.idle":"2023-05-07T12:57:56.973362Z","shell.execute_reply.started":"2023-05-07T12:57:56.939450Z","shell.execute_reply":"2023-05-07T12:57:56.972399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_metadata_df.Sex = np.where(all_metadata_df.Sex == \"F\", 0, 1)\n\nall_metadata_df.Medication = np.where(all_metadata_df.Medication == \"of\", 0, 1)","metadata":{"execution":{"iopub.status.busy":"2023-05-07T12:59:09.140202Z","iopub.execute_input":"2023-05-07T12:59:09.140712Z","iopub.status.idle":"2023-05-07T12:59:09.150309Z","shell.execute_reply.started":"2023-05-07T12:59:09.140667Z","shell.execute_reply":"2023-05-07T12:59:09.149340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\n\nmask = np.triu(np.ones_like(all_metadata_df.corr(), dtype=bool))\n\nsns.heatmap(all_metadata_df.corr(), mask=mask, annot=True, cmap='coolwarm',\n                linewidths=0.2, annot_kws={\"size\": 7}, rasterized=True)\n\nplt.xticks(rotation=55)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T13:00:41.285480Z","iopub.execute_input":"2023-05-07T13:00:41.285900Z","iopub.status.idle":"2023-05-07T13:00:41.706537Z","shell.execute_reply.started":"2023-05-07T13:00:41.285865Z","shell.execute_reply":"2023-05-07T13:00:41.705498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns_phik = ['Type', 'event_duration', 'Age', 'UPDRSIII_On', 'YearsSinceDx',\n                'UPDRSIII_Off', 'NFOGQ', 'Sex', 'Medication', 'Kinetic', 'Visit']\nphik_overview = all_metadata_df[columns_phik].phik_matrix( \\\n                                    interval_cols=['event_duration', 'Age',\n                                                'UPDRSIII_On', 'YearsSinceDx',\n                                                'UPDRSIII_Off','NFOGQ'])\n\nplot_correlation_matrix(phik_overview.values, x_labels=phik_overview.columns,\n                        y_labels=phik_overview.index, vmin=0, vmax=1,\n                        color_map='Blues', title=r'correlation matrix $\\phi_K$',\n                        figsize=(9, 6))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T13:02:19.306325Z","iopub.execute_input":"2023-05-07T13:02:19.306732Z","iopub.status.idle":"2023-05-07T13:02:22.059822Z","shell.execute_reply.started":"2023-05-07T13:02:19.306701Z","shell.execute_reply":"2023-05-07T13:02:22.058424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# bin the Age column based on quantiles\nall_metadata_df['age_bin'] = pd.qcut(all_metadata_df['Age'], q=5)","metadata":{"execution":{"iopub.status.busy":"2023-05-07T13:02:55.411382Z","iopub.execute_input":"2023-05-07T13:02:55.411929Z","iopub.status.idle":"2023-05-07T13:02:55.421239Z","shell.execute_reply.started":"2023-05-07T13:02:55.411890Z","shell.execute_reply":"2023-05-07T13:02:55.420454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the frequency of FoG events of each Type for different age ranges\nplt.figure(figsize=(15, 14))\nfor i, age_bin in enumerate(all_metadata_df.age_bin.unique(), 1):\n    plt.subplot(3, 3, i)\n    sns.barplot(\n        x=all_metadata_df[all_metadata_df.age_bin == age_bin].Type\n                                        .value_counts(normalize=True).index,\n        y=all_metadata_df[all_metadata_df.age_bin == age_bin].Type\n                                        .value_counts(normalize=True))\n    plt.ylabel('The frequency of FoG events')\n    plt.title(f\"Age in range {age_bin}\")\n    \n    # Add frequency values on top of the bars\n    for index, value in enumerate(\n                all_metadata_df[all_metadata_df.age_bin == age_bin].Type\n                                            .value_counts(normalize=True)):\n        plt.annotate(f\"{value * 100:.1f}%\", xy=(index, value),\n                                    ha='center', va='bottom')\nplt.subplots_adjust(wspace=0.3, hspace=0.2)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T13:07:50.691780Z","iopub.execute_input":"2023-05-07T13:07:50.692146Z","iopub.status.idle":"2023-05-07T13:07:51.593996Z","shell.execute_reply.started":"2023-05-07T13:07:50.692116Z","shell.execute_reply":"2023-05-07T13:07:51.592883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the frequency of FoG events of each Type for male and female patients\nplt.figure(figsize=(12, 5))\ngender_dict = {0: 'Female', 1: 'Male'}\nfor i, gender in enumerate(all_metadata_df.Sex.unique(), 1):\n    plt.subplot(1, 2, i)\n    sns.barplot(\n        x=all_metadata_df[all_metadata_df.Sex == gender].Type\n                                .value_counts(normalize=True).index,\n        y=all_metadata_df[all_metadata_df.Sex == gender].Type\n                                .value_counts(normalize=True))\n    plt.ylabel('The frequency of FoG events')\n    plt.title(f\"Gender = {gender_dict[gender]}\")\n    \n    # Add frequency values on top of the bars\n    for index, value in enumerate(\n            all_metadata_df[all_metadata_df.Sex == gender].Type\n                                    .value_counts(normalize=True)):\n        plt.annotate(f\"{value * 100:.1f}%\", xy=(index, value),\n                                    ha='center', va='bottom')\nplt.subplots_adjust(wspace=0.4)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T13:12:35.177565Z","iopub.execute_input":"2023-05-07T13:12:35.177968Z","iopub.status.idle":"2023-05-07T13:12:35.593981Z","shell.execute_reply.started":"2023-05-07T13:12:35.177936Z","shell.execute_reply":"2023-05-07T13:12:35.592661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the counts of FoG events for different age ranges\n\nplt.figure(figsize=(12, 5))\nmed_dict = {0: 'Medication status: off', 1: 'Medication status: on'}\nfor i, med_status in enumerate(all_metadata_df.Medication.unique(), 1):\n    plt.subplot(1, 2, i)\n    sns.barplot(\n        x=all_metadata_df[all_metadata_df.Medication == med_status].Type\n                                        .value_counts(normalize=True).index,\n        y=all_metadata_df[all_metadata_df.Medication == med_status].Type\n                                        .value_counts(normalize=True))\n    plt.ylabel('The frequency of FoG events')\n    plt.title(f\"{med_dict[med_status]}\")\n    \n    # Add frequency values on top of the bars\n    for index, value in enumerate(\n            all_metadata_df[all_metadata_df.Medication == med_status].Type\n                                            .value_counts(normalize=True)):\n        plt.annotate(f\"{value * 100:.1f}%\", xy=(index, value),\n                                    ha='center', va='bottom')\nplt.subplots_adjust(wspace=0.4)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T13:17:13.250659Z","iopub.execute_input":"2023-05-07T13:17:13.251028Z","iopub.status.idle":"2023-05-07T13:17:13.501122Z","shell.execute_reply.started":"2023-05-07T13:17:13.250999Z","shell.execute_reply":"2023-05-07T13:17:13.500012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read 500 files of the tdcsfog metadata (the total number of files is 833)\n\ntdcsfog_files = [f'{data_dir}train/tdcsfog/{id}.csv' for id in \\\n                                         tdcsfog_metadata.Id.to_list()[:501]]\n\ntdcsfog_data = pd.concat([pd.read_csv(file) for file in tdcsfog_files])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T13:18:54.222438Z","iopub.execute_input":"2023-05-07T13:18:54.222833Z","iopub.status.idle":"2023-05-07T13:19:05.298141Z","shell.execute_reply.started":"2023-05-07T13:18:54.222802Z","shell.execute_reply":"2023-05-07T13:19:05.296538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ProfileReport(tdcsfog_data, title='Tdcsfog Dataset Report',\n              progress_bar=False,\n              interactions=None,\n              explorative=True, dark_mode=True,\n              notebook={'iframe':{'height': '600px'}},\n              missing_diagrams={'heatmap': False, 'dendrogram': False}).to_notebook_iframe()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T13:20:11.259855Z","iopub.execute_input":"2023-05-07T13:20:11.260250Z","iopub.status.idle":"2023-05-07T13:21:34.656174Z","shell.execute_reply.started":"2023-05-07T13:20:11.260218Z","shell.execute_reply":"2023-05-07T13:21:34.655077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask = np.triu(np.ones_like(tdcsfog_data.corr(), dtype=bool))\nfig, ax = plt.subplots(figsize=(7, 5))\nsns.heatmap(tdcsfog_data.corr(), mask=mask, annot=True, cmap='coolwarm',\n                linewidths=0.2, annot_kws={\"size\": 7}, rasterized=True)\nplt.xticks(rotation=35)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T13:22:24.406489Z","iopub.execute_input":"2023-05-07T13:22:24.406841Z","iopub.status.idle":"2023-05-07T13:22:26.212017Z","shell.execute_reply.started":"2023-05-07T13:22:24.406816Z","shell.execute_reply":"2023-05-07T13:22:26.210991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"phik_overview = tdcsfog_data.phik_matrix(\n                            interval_cols=['AccV', 'AccML', 'AccAP','Time'])\n\nplot_correlation_matrix(phik_overview.values, x_labels=phik_overview.columns,\n                        y_labels=phik_overview.index, vmin=0, vmax=1, \n                        color_map='Blues', title=r'correlation matrix $\\phi_K$',\n                        figsize=(7, 5))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T13:24:46.245850Z","iopub.execute_input":"2023-05-07T13:24:46.246207Z","iopub.status.idle":"2023-05-07T13:25:07.597762Z","shell.execute_reply.started":"2023-05-07T13:24:46.246181Z","shell.execute_reply":"2023-05-07T13:25:07.596489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read one file of the tdcsfog train data with event of Turn type\ntdcsfog_files_turn = [f'{data_dir}train/tdcsfog/{id}.csv' for id in \\\n                              tdcsfog_metadata.Id.to_list()[16:17]]\n\ntdcsfog_data_turn = pd.concat([pd.read_csv(file) for file in tdcsfog_files_turn])\n\n# Read one file of the tdcsfog train data with event of Walking type\ntdcsfog_files_walk = [f'{data_dir}train/tdcsfog/{id}.csv' for id in \\\n                              tdcsfog_metadata.Id.to_list()[19:20]]\n\ntdcsfog_data_walk = pd.concat([pd.read_csv(file) for file in tdcsfog_files_walk])\n\n# Read one file of the tdcsfog train data with event of StartHesitation type\n\ntdcsfog_files_hes = [f'{data_dir}train/tdcsfog/{id}.csv' for id in \\\n                             tdcsfog_metadata.Id.to_list()[77:78]]\n\ntdcsfog_data_hes = pd.concat([pd.read_csv(file) for file in tdcsfog_files_hes])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T13:31:44.296970Z","iopub.execute_input":"2023-05-07T13:31:44.297392Z","iopub.status.idle":"2023-05-07T13:31:44.391312Z","shell.execute_reply.started":"2023-05-07T13:31:44.297346Z","shell.execute_reply":"2023-05-07T13:31:44.390191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the overall trends of acceleration measurements for events of each Type\nplt.figure(figsize=(14, 12))\n\ntypes_dict = {1: 'Turn', 2: 'Walking', 3: 'StartHesitation'}\n\nfor i, data in enumerate(\n            [tdcsfog_data_turn, tdcsfog_data_walk, tdcsfog_data_hes], 1):\n    plt.subplot(2, 2, i)\n    window_size = 128\n    accv_rolling_mean = data[\"AccV\"].rolling(window_size).mean()\n    accap_rolling_mean = data[\"AccAP\"].rolling(window_size).mean()\n    accml_rolling_mean = data[\"AccML\"].rolling(window_size).mean()\n\n    sns.lineplot(x=data.Time, y=accv_rolling_mean)\n    sns.lineplot(x=data.Time, y=accap_rolling_mean)\n    sns.lineplot(x=data.Time, y=accml_rolling_mean)\n    plt.title(f'The trend of acceleration measurements (\"{types_dict[i]}\" event)')\n    plt.xlabel('Time')\n    plt.ylabel('Acceleration value (m/s^2)')\n\nplt.legend(['Anteroposterior', 'Mediolateral', 'Vertical'], loc='upper left',\n               bbox_to_anchor=(2.3, 2.1), handlelength=0,\n               shadow=True, labelcolor=['orange', 'green', 'blue'])\nplt.subplots_adjust(wspace=0.3, hspace=0.2)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T13:37:13.693006Z","iopub.execute_input":"2023-05-07T13:37:13.693481Z","iopub.status.idle":"2023-05-07T13:37:15.696244Z","shell.execute_reply.started":"2023-05-07T13:37:13.693422Z","shell.execute_reply":"2023-05-07T13:37:15.694938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating a list of indices that correspond to the notype series IDs in the defog_metadata\nnotype_idx = [0, 4, 13, 15, 17, 19, 20, 22, 25, 27, 30, 31, 32, 34, 36, 40, 42,\n              53, 54, 55, 58, 60, 62, 66, 67, 73, 74, 76, 79, 83, 84, 88, 89, 94,\n              96, 97, 98, 101, 102, 105, 113, 114, 115, 121, 124, 126]\n\n# Separating the defog_metadata\ndefog_metadata_only = defog_metadata[~defog_metadata.index.isin(notype_idx)].copy()\nnotype_metadata = defog_metadata[defog_metadata.index.isin(notype_idx)].copy()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T13:39:47.338388Z","iopub.execute_input":"2023-05-07T13:39:47.339770Z","iopub.status.idle":"2023-05-07T13:39:47.351762Z","shell.execute_reply.started":"2023-05-07T13:39:47.339713Z","shell.execute_reply":"2023-05-07T13:39:47.350473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read one file of the defog metadata (the total number of files is 91)\n\ndefog_files = [f\"{data_dir}train/defog/{id}.csv\" for id in \\\n                              defog_metadata_only.Id.to_list()[:1]]\ndefog_data = pd.concat([pd.read_csv(file) for file in defog_files])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T13:41:12.344227Z","iopub.execute_input":"2023-05-07T13:41:12.344652Z","iopub.status.idle":"2023-05-07T13:41:12.642086Z","shell.execute_reply.started":"2023-05-07T13:41:12.344621Z","shell.execute_reply":"2023-05-07T13:41:12.640702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.concat([defog_metadata[defog_metadata.Id == '02ea782681'],\n           events_data[events_data.Id == '02ea782681']], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-07T13:43:23.550948Z","iopub.execute_input":"2023-05-07T13:43:23.551548Z","iopub.status.idle":"2023-05-07T13:43:23.582029Z","shell.execute_reply.started":"2023-05-07T13:43:23.551500Z","shell.execute_reply":"2023-05-07T13:43:23.580817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_data_valid = defog_data.loc[ \\\n                        (defog_data.Valid == True) & (defog_data.Task == True)]","metadata":{"execution":{"iopub.status.busy":"2023-05-07T13:44:07.600016Z","iopub.execute_input":"2023-05-07T13:44:07.600434Z","iopub.status.idle":"2023-05-07T13:44:07.609504Z","shell.execute_reply.started":"2023-05-07T13:44:07.600399Z","shell.execute_reply":"2023-05-07T13:44:07.608602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_data_valid.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-07T13:44:16.455420Z","iopub.execute_input":"2023-05-07T13:44:16.456687Z","iopub.status.idle":"2023-05-07T13:44:16.465762Z","shell.execute_reply.started":"2023-05-07T13:44:16.456623Z","shell.execute_reply":"2023-05-07T13:44:16.464302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the overall trend of acceleration measurements for each axis\n\nplt.figure(figsize=(14, 5))\n\ndefogdata_dict = {1: 'All defogdata', 2: 'Valid defog data'}\n\nfor i, data in enumerate([defog_data, defog_data_valid], 1):\n    plt.subplot(1, 2, i)\n    window_size = 100\n    accv_rolling_mean = data[\"AccV\"].rolling(window_size).mean()\n    accap_rolling_mean = data[\"AccAP\"].rolling(window_size).mean()\n    accml_rolling_mean = data[\"AccML\"].rolling(window_size).mean()\n\n    sns.lineplot(x=data.Time, y=accv_rolling_mean)\n    sns.lineplot(x=data.Time, y=accap_rolling_mean)\n    sns.lineplot(x=data.Time, y=accml_rolling_mean)\n    plt.title(f'The trend of acceleration measurements ({defogdata_dict[i]})')\n    plt.xlabel('Time')\n    plt.ylabel('Acceleration value (g)')\n\nplt.legend(['Mediolateral', 'Anteroposterior', 'Vertical'], loc='upper left',\n               bbox_to_anchor=(1.05, 1), shadow=True, handlelength=0,\n               labelcolor=['green', 'orange', 'blue'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T13:49:19.339057Z","iopub.execute_input":"2023-05-07T13:49:19.339800Z","iopub.status.idle":"2023-05-07T13:49:22.351318Z","shell.execute_reply.started":"2023-05-07T13:49:19.339748Z","shell.execute_reply":"2023-05-07T13:49:22.350416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read one file of the daily metadata (the total number of files is 65)\ndaily_files = [f'{data_dir}unlabeled/{id}.parquet' for id in \\\n                                       daily_metadata.Id.to_list()[:1]]\n\ndaily_data = pd.concat([pd.read_parquet(file) for file in daily_files])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T13:50:47.379594Z","iopub.execute_input":"2023-05-07T13:50:47.380045Z","iopub.status.idle":"2023-05-07T13:51:02.391625Z","shell.execute_reply.started":"2023-05-07T13:50:47.380007Z","shell.execute_reply":"2023-05-07T13:51:02.390126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_data.describe(include = \"float64\").T","metadata":{"execution":{"iopub.status.busy":"2023-05-07T13:51:03.406506Z","iopub.execute_input":"2023-05-07T13:51:03.407594Z","iopub.status.idle":"2023-05-07T13:51:03.454302Z","shell.execute_reply.started":"2023-05-07T13:51:03.407546Z","shell.execute_reply":"2023-05-07T13:51:03.453048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}