{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-09T22:49:35.499274Z","iopub.execute_input":"2023-04-09T22:49:35.500241Z","iopub.status.idle":"2023-04-09T22:49:35.834122Z","shell.execute_reply.started":"2023-04-09T22:49:35.500199Z","shell.execute_reply":"2023-04-09T22:49:35.833000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reading in the data\n\nThe following code reads in the data.\nThe first two lines will read in a single file in order to test code.\nOnce the code works, use the remaining lines to get all of the labratory data.","metadata":{}},{"cell_type":"code","source":"df=[]\n#path1 = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/08d6702e8a.csv'\n#df = pd.read_csv(path1)\nfor dirname, _, filenames in os.walk('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog'): \n    for filename in filenames:\n        identifier = filename[:-4]\n        tdcs_inst = pd.read_csv(os.path.join(dirname,filename))\n        tdcs_inst = tdcs_inst.assign(id=identifier)\n        df.append(tdcs_inst)\ndf = pd.concat(df)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:49:35.835561Z","iopub.execute_input":"2023-04-09T22:49:35.835894Z","iopub.status.idle":"2023-04-09T22:49:56.326149Z","shell.execute_reply.started":"2023-04-09T22:49:35.835862Z","shell.execute_reply":"2023-04-09T22:49:56.324712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:49:56.328060Z","iopub.execute_input":"2023-04-09T22:49:56.328846Z","iopub.status.idle":"2023-04-09T22:49:56.349126Z","shell.execute_reply.started":"2023-04-09T22:49:56.328782Z","shell.execute_reply":"2023-04-09T22:49:56.347722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Getting a general view of the data","metadata":{}},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:49:56.352083Z","iopub.execute_input":"2023-04-09T22:49:56.352454Z","iopub.status.idle":"2023-04-09T22:49:58.278065Z","shell.execute_reply.started":"2023-04-09T22:49:56.352418Z","shell.execute_reply":"2023-04-09T22:49:58.276982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:49:58.279333Z","iopub.execute_input":"2023-04-09T22:49:58.279670Z","iopub.status.idle":"2023-04-09T22:49:58.294725Z","shell.execute_reply.started":"2023-04-09T22:49:58.279638Z","shell.execute_reply":"2023-04-09T22:49:58.293385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:49:58.296853Z","iopub.execute_input":"2023-04-09T22:49:58.297351Z","iopub.status.idle":"2023-04-09T22:49:58.319493Z","shell.execute_reply.started":"2023-04-09T22:49:58.297296Z","shell.execute_reply":"2023-04-09T22:49:58.318103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Recode to a categorical variable\n\nNow we need to code the A-Kinetic, Turning, Walking, and StartHesitation into a single variable","metadata":{}},{"cell_type":"code","source":"df['state'] = 'Error'","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:49:58.321250Z","iopub.execute_input":"2023-04-09T22:49:58.322326Z","iopub.status.idle":"2023-04-09T22:49:58.394341Z","shell.execute_reply.started":"2023-04-09T22:49:58.322286Z","shell.execute_reply":"2023-04-09T22:49:58.392850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe(include = 'all')","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:49:58.396461Z","iopub.execute_input":"2023-04-09T22:49:58.397101Z","iopub.status.idle":"2023-04-09T22:50:01.237481Z","shell.execute_reply.started":"2023-04-09T22:49:58.397046Z","shell.execute_reply":"2023-04-09T22:50:01.236126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2 = df.groupby('Turn').describe(include = 'all')\nprint(df2)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:50:01.238936Z","iopub.execute_input":"2023-04-09T22:50:01.239506Z","iopub.status.idle":"2023-04-09T22:50:05.300429Z","shell.execute_reply.started":"2023-04-09T22:50:01.239471Z","shell.execute_reply":"2023-04-09T22:50:05.299066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.loc[df['Turn'] == 1, 'state'] = 'Turn'\ndf.loc[df['StartHesitation'] == 1, 'state'] = 'Hesitate'\ndf.loc[df['Walking'] == 1, 'state'] = 'Walk'\ndf.loc[df['state'] == 'Error', 'state'] = 'AKinetic'","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:50:05.304788Z","iopub.execute_input":"2023-04-09T22:50:05.305222Z","iopub.status.idle":"2023-04-09T22:50:07.097602Z","shell.execute_reply.started":"2023-04-09T22:50:05.305186Z","shell.execute_reply":"2023-04-09T22:50:07.096263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df['state'].unique())","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:50:07.099636Z","iopub.execute_input":"2023-04-09T22:50:07.100177Z","iopub.status.idle":"2023-04-09T22:50:07.606842Z","shell.execute_reply.started":"2023-04-09T22:50:07.100125Z","shell.execute_reply":"2023-04-09T22:50:07.605657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe(include = 'all')\n","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:50:07.608192Z","iopub.execute_input":"2023-04-09T22:50:07.608541Z","iopub.status.idle":"2023-04-09T22:50:10.457955Z","shell.execute_reply.started":"2023-04-09T22:50:07.608494Z","shell.execute_reply":"2023-04-09T22:50:10.456701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analysis!\n\nFinally we need to look at our data.\n\nThe null hypothesis is that the distributions for the acceleration data are the same regardless of the state of the patient.\nIn order to test this, we can look at a number of different atributes.  The most common would be to do a test of \"center.\"\nNormally, this would be ANOVA, but that has some assumptions.  We will consider these first.\n","metadata":{}},{"cell_type":"code","source":"import scipy.stats as stats\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:50:10.459502Z","iopub.execute_input":"2023-04-09T22:50:10.459982Z","iopub.status.idle":"2023-04-09T22:50:10.729300Z","shell.execute_reply.started":"2023-04-09T22:50:10.459935Z","shell.execute_reply":"2023-04-09T22:50:10.728013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  The AccV variable\n\nTesting normality of the AccV variable.  We produce normality plots for the data stratified by the level of state that the patient is in.","metadata":{}},{"cell_type":"code","source":"Unique_States = df['state'].unique()\nfor State in Unique_States:\n    stats.probplot(df[df['state'] == State]['AccV'], dist = 'norm', plot = plt)\n    plt.title(\"Probability Plot - \" + State)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:50:10.730627Z","iopub.execute_input":"2023-04-09T22:50:10.730992Z","iopub.status.idle":"2023-04-09T22:50:31.351479Z","shell.execute_reply.started":"2023-04-09T22:50:10.730960Z","shell.execute_reply":"2023-04-09T22:50:31.350195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"Oh... that's not good.  These are all very fat tailed distributions.  The patterns look similar, so it seems like\nit may be reasonable to test the \"center\" of the data still, but we should look at a univariate histogram or density plot of each to see what we \nare looking at.","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(2,1, figsize = (22,14))\nax = axes.ravel()\n\n#['AKinetic' 'Turn' 'Hesitate' 'Walk']\n\ndf.loc[df.state == 'AKinetic'].AccV.plot(kind = 'hist', ax = ax[0], alpha = 0.5, bins = 50, sharex = True, label = 'A-Kinetic', edgecolor = 'black', linewidth = 3)\ndf.loc[df.state == 'Turn'].AccV.plot(kind = 'hist', ax = ax[0], alpha = 0.5, bins = 50, sharex = True, label = 'Turn', edgecolor = 'blue', linewidth = 3)\ndf.loc[df.state == 'Hesitate'].AccV.plot(kind = 'hist', ax = ax[0], alpha = 0.5, bins = 50, sharex = True, label = 'Hesitate', edgecolor = 'red', linewidth = 3)\ndf.loc[df.state == 'Walk'].AccV.plot(kind = 'hist', ax = ax[0], alpha = 0.5, bins = 50, sharex = True, label = 'Walk', edgecolor = 'green', linewidth = 3)\n\n\n\nfig.suptitle(\"Histogram and Density plot of AccV by state\", fontsize = 20)\n\ndf.loc[df.state == 'AKinetic'].AccV.plot(kind = 'density', ax = ax[1], alpha = 0.5, label = 'A-Kinetic', linewidth = 3)\ndf.loc[df.state == 'Turn'].AccV.plot(kind = 'density', ax = ax[1], alpha = 0.5, label = 'Turn', linewidth = 3)\ndf.loc[df.state == 'Hesitate'].AccV.plot(kind = 'density', ax = ax[1], alpha = 0.5, label = 'Hesitate', linewidth = 3)\ndf.loc[df.state == 'Walk'].AccV.plot(kind = 'density', ax = ax[1], alpha = 0.5, label = 'Walk', linewidth = 3)\n\nax[0].legend(fontsize = 18)\nax[0].set_xlim(-16, 4)\nax[0].set_title('Histogram of AccV based on state')\nax[0].tick_params(axis = 'both', which = 'major', labelsize = 14)\nax[0].tick_params(axis = 'both', which = 'minor', labelsize = 14)\n\nax[1].legend(fontsize = 18)\nax[1].set_xlim(-16,4)\nax[1].set_title('Density of AccV based on state')\nax[1].tick_params(axis = 'both', which = 'major', labelsize = 14)\nax[1].tick_params(axis = 'both', which = 'minor', labelsize = 14)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:50:31.353289Z","iopub.execute_input":"2023-04-09T22:50:31.353829Z","iopub.status.idle":"2023-04-09T22:52:35.924876Z","shell.execute_reply.started":"2023-04-09T22:50:31.353785Z","shell.execute_reply":"2023-04-09T22:52:35.923590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"boxplot = df.boxplot(column=['AccV'], by='state')","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:52:35.926431Z","iopub.execute_input":"2023-04-09T22:52:35.926843Z","iopub.status.idle":"2023-04-09T22:52:47.976197Z","shell.execute_reply.started":"2023-04-09T22:52:35.926805Z","shell.execute_reply":"2023-04-09T22:52:47.974922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Wow,okay, so now what?  The distributions of this data appear to be somewhat unimodal (for the most part) and pretty symmetric in general which means that the mean is probably a representative measure of center.  That means we can probably apply one way ANOVA to see of the centers are all reasonably similar; however, the number of outliers is pretty significant and the spread of the data within group is very high when compared to the between group spread.  This indicates that we probably don't expect to be able to see a difference in means.  \n\nIn general, we can see that there is likely a difference in spread; however, tests of spread are highly sensitive to outliers and won't be a good measure here.  Furthermore, it appears that skewness differs between datasets (skewed left, right, and symmetric.)  I think the best way of testing this is the K-S (Kolmogorov-Smirnov test) which is a non-parametric test of distributional heterogeneity.  Admittedly, this can only test two distributional differences, not 4, so we would have to run pairwise tests.  In that case, 6 tests.  Given that there are 3 univariate continuous variables we will consider, that's 18 hypothesis tests so by taking the bonferroni post-hoc adjustment of 0.05/18 = 0.00277 will be our critical p-value if we run all of these tests. Realistically, there is probably a better approach.\n\n1.)  For each of these distributions, look for those that appear to be the closest on the density plots.\n2.)  Apply the K-S test on these.  If the p-value is very small, we can infer that these are from different distributions.\n3.)  Based on logic, and without running the tests, we can assume that the others will also be different and thus we can start looking for statistics that can characterize these differences effectivelly.\n\n\n","metadata":{}},{"cell_type":"code","source":"AccV_AKin = df.loc[df['state']=='AKinetic','AccV']\n\nAccV_Walk = df.loc[df['state']=='Walk','AccV']\n\nAccV_Turn = df.loc[df['state']=='Turn','AccV']\n\nAccV_Hesitate = df.loc[df['state']=='Hesitate','AccV']","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:52:47.977824Z","iopub.execute_input":"2023-04-09T22:52:47.978557Z","iopub.status.idle":"2023-04-09T22:52:49.770536Z","shell.execute_reply.started":"2023-04-09T22:52:47.978481Z","shell.execute_reply":"2023-04-09T22:52:49.769414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function to calculate Cohen's d for independent samples\ndef cohend(d1, d2):\n # calculate the size of samples\n n1, n2 = len(d1), len(d2)\n # calculate the variance of the samples\n s1, s2 = np.var(d1, ddof=1), np.var(d2, ddof=1)\n # calculate the pooled standard deviation\n s = np.sqrt(((n1 - 1) * s1 + (n2 - 1) * s2) / (n1 + n2 - 2))\n # calculate the means of the samples\n u1, u2 = np.mean(d1), np.mean(d2)\n # calculate the effect size\n return (u1 - u2) / s","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:52:49.772499Z","iopub.execute_input":"2023-04-09T22:52:49.773017Z","iopub.status.idle":"2023-04-09T22:52:49.781993Z","shell.execute_reply.started":"2023-04-09T22:52:49.772967Z","shell.execute_reply":"2023-04-09T22:52:49.780461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(stats.ks_2samp(AccV_AKin, AccV_Walk))\nprint(cohend(AccV_AKin, AccV_Walk))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:52:49.783728Z","iopub.execute_input":"2023-04-09T22:52:49.784104Z","iopub.status.idle":"2023-04-09T22:52:50.912811Z","shell.execute_reply.started":"2023-04-09T22:52:49.784070Z","shell.execute_reply":"2023-04-09T22:52:50.911325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(stats.ks_2samp(AccV_AKin, AccV_Turn))\nprint(cohend(AccV_AKin, AccV_Turn))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:52:50.914288Z","iopub.execute_input":"2023-04-09T22:52:50.914807Z","iopub.status.idle":"2023-04-09T22:52:52.388051Z","shell.execute_reply.started":"2023-04-09T22:52:50.914767Z","shell.execute_reply":"2023-04-09T22:52:52.386915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(stats.ks_2samp(AccV_AKin, AccV_Hesitate))\nprint(cohend(AccV_AKin, AccV_Hesitate))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:52:52.389423Z","iopub.execute_input":"2023-04-09T22:52:52.389856Z","iopub.status.idle":"2023-04-09T22:52:53.543053Z","shell.execute_reply.started":"2023-04-09T22:52:52.389807Z","shell.execute_reply":"2023-04-09T22:52:53.541713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(stats.ks_2samp(AccV_Hesitate, AccV_Walk))\nprint(cohend(AccV_Hesitate, AccV_Walk))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:52:53.545687Z","iopub.execute_input":"2023-04-09T22:52:53.546734Z","iopub.status.idle":"2023-04-09T22:52:53.780977Z","shell.execute_reply.started":"2023-04-09T22:52:53.546692Z","shell.execute_reply":"2023-04-09T22:52:53.779440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(stats.ks_2samp(AccV_Hesitate, AccV_Turn))\nprint(cohend(AccV_Hesitate, AccV_Turn))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:52:53.782678Z","iopub.execute_input":"2023-04-09T22:52:53.783091Z","iopub.status.idle":"2023-04-09T22:52:54.194106Z","shell.execute_reply.started":"2023-04-09T22:52:53.783052Z","shell.execute_reply":"2023-04-09T22:52:54.192951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(stats.ks_2samp(AccV_Walk, AccV_Turn))\nprint(cohend(AccV_Walk, AccV_Turn))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:52:54.195598Z","iopub.execute_input":"2023-04-09T22:52:54.196177Z","iopub.status.idle":"2023-04-09T22:52:54.583379Z","shell.execute_reply.started":"2023-04-09T22:52:54.196120Z","shell.execute_reply":"2023-04-09T22:52:54.582244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  The AccML variable","metadata":{}},{"cell_type":"code","source":"Unique_States = df['state'].unique()\nfor State in Unique_States:\n    stats.probplot(df[df['state'] == State]['AccML'], dist = 'norm', plot = plt)\n    plt.title(\"Probability Plot - \" + State)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:52:54.584669Z","iopub.execute_input":"2023-04-09T22:52:54.585010Z","iopub.status.idle":"2023-04-09T22:53:15.331036Z","shell.execute_reply.started":"2023-04-09T22:52:54.584978Z","shell.execute_reply":"2023-04-09T22:53:15.329571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(2,1, figsize = (22,14))\nax = axes.ravel()\n\n#['AKinetic' 'Turn' 'Hesitate' 'Walk']\n\ndf.loc[df.state == 'AKinetic'].AccML.plot(kind = 'hist', ax = ax[0], alpha = 0.5, bins = 50, sharex = True, label = 'A-Kinetic', edgecolor = 'black', linewidth = 3)\ndf.loc[df.state == 'Turn'].AccML.plot(kind = 'hist', ax = ax[0], alpha = 0.5, bins = 50, sharex = True, label = 'Turn', edgecolor = 'blue', linewidth = 3)\ndf.loc[df.state == 'Hesitate'].AccML.plot(kind = 'hist', ax = ax[0], alpha = 0.5, bins = 50, sharex = True, label = 'Hesitate', edgecolor = 'red', linewidth = 3)\ndf.loc[df.state == 'Walk'].AccML.plot(kind = 'hist', ax = ax[0], alpha = 0.5, bins = 50, sharex = True, label = 'Walk', edgecolor = 'green', linewidth = 3)\n\n\n\nfig.suptitle(\"Histogram and Density plot of AccML by state\", fontsize = 20)\n\ndf.loc[df.state == 'AKinetic'].AccML.plot(kind = 'density', ax = ax[1], alpha = 0.5, label = 'A-Kinetic', linewidth = 3)\ndf.loc[df.state == 'Turn'].AccML.plot(kind = 'density', ax = ax[1], alpha = 0.5, label = 'Turn', linewidth = 3)\ndf.loc[df.state == 'Hesitate'].AccML.plot(kind = 'density', ax = ax[1], alpha = 0.5, label = 'Hesitate', linewidth = 3)\ndf.loc[df.state == 'Walk'].AccML.plot(kind = 'density', ax = ax[1], alpha = 0.5, label = 'Walk', linewidth = 3)\n\nax[0].legend(fontsize = 18)\nax[0].set_xlim(-8, 8)\nax[0].set_title('Histogram of AccML based on state')\nax[0].tick_params(axis = 'both', which = 'major', labelsize = 14)\nax[0].tick_params(axis = 'both', which = 'minor', labelsize = 14)\n\nax[1].legend(fontsize = 18)\nax[1].set_xlim(-8,8)\nax[1].set_title('Density of AccML based on state')\nax[1].tick_params(axis = 'both', which = 'major', labelsize = 14)\nax[1].tick_params(axis = 'both', which = 'minor', labelsize = 14)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:53:15.332706Z","iopub.execute_input":"2023-04-09T22:53:15.333090Z","iopub.status.idle":"2023-04-09T22:55:20.141634Z","shell.execute_reply.started":"2023-04-09T22:53:15.333054Z","shell.execute_reply":"2023-04-09T22:55:20.140198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"boxplot = df.boxplot(column=['AccML'], by='state')","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:55:20.143260Z","iopub.execute_input":"2023-04-09T22:55:20.143958Z","iopub.status.idle":"2023-04-09T22:55:31.677188Z","shell.execute_reply.started":"2023-04-09T22:55:20.143918Z","shell.execute_reply":"2023-04-09T22:55:31.675776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AccML_AKin = df.loc[df['state']=='AKinetic','AccML']\n\nAccML_Walk = df.loc[df['state']=='Walk','AccML']\n\nAccML_Turn = df.loc[df['state']=='Turn','AccML']\n\nAccML_Hesitate = df.loc[df['state']=='Hesitate','AccML']","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:55:31.684414Z","iopub.execute_input":"2023-04-09T22:55:31.685580Z","iopub.status.idle":"2023-04-09T22:55:33.512818Z","shell.execute_reply.started":"2023-04-09T22:55:31.685485Z","shell.execute_reply":"2023-04-09T22:55:33.511585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(stats.ks_2samp(AccML_AKin, AccML_Walk))\nprint(cohend(AccML_AKin, AccML_Walk))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:55:33.514280Z","iopub.execute_input":"2023-04-09T22:55:33.514672Z","iopub.status.idle":"2023-04-09T22:55:34.611664Z","shell.execute_reply.started":"2023-04-09T22:55:33.514623Z","shell.execute_reply":"2023-04-09T22:55:34.609886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(stats.ks_2samp(AccML_AKin, AccML_Turn))\nprint(cohend(AccML_AKin, AccML_Hesitate))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:55:34.613601Z","iopub.execute_input":"2023-04-09T22:55:34.613975Z","iopub.status.idle":"2023-04-09T22:55:36.081891Z","shell.execute_reply.started":"2023-04-09T22:55:34.613940Z","shell.execute_reply":"2023-04-09T22:55:36.080302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(stats.ks_2samp(AccML_AKin, AccML_Hesitate))\nprint(cohend(AccML_AKin, AccML_Hesitate))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:55:36.083449Z","iopub.execute_input":"2023-04-09T22:55:36.084615Z","iopub.status.idle":"2023-04-09T22:55:37.231545Z","shell.execute_reply.started":"2023-04-09T22:55:36.084565Z","shell.execute_reply":"2023-04-09T22:55:37.229976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(stats.ks_2samp(AccML_Walk, AccML_Turn))\nprint(cohend(AccML_Walk, AccML_Turn))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:55:37.233355Z","iopub.execute_input":"2023-04-09T22:55:37.234234Z","iopub.status.idle":"2023-04-09T22:55:37.624585Z","shell.execute_reply.started":"2023-04-09T22:55:37.234175Z","shell.execute_reply":"2023-04-09T22:55:37.623171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(stats.ks_2samp(AccML_Walk, AccML_Hesitate))\nprint(cohend(AccML_Walk, AccML_Hesitate))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:55:37.625940Z","iopub.execute_input":"2023-04-09T22:55:37.626291Z","iopub.status.idle":"2023-04-09T22:55:37.728005Z","shell.execute_reply.started":"2023-04-09T22:55:37.626247Z","shell.execute_reply":"2023-04-09T22:55:37.726888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(stats.ks_2samp(AccML_Turn, AccML_Hesitate))\nprint(cohend(AccML_Turn, AccML_Hesitate))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:55:37.729209Z","iopub.execute_input":"2023-04-09T22:55:37.729682Z","iopub.status.idle":"2023-04-09T22:55:38.154243Z","shell.execute_reply.started":"2023-04-09T22:55:37.729634Z","shell.execute_reply":"2023-04-09T22:55:38.152566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## The AccAP variable","metadata":{}},{"cell_type":"code","source":"Unique_States = df['state'].unique()\nfor State in Unique_States:\n    stats.probplot(df[df['state'] == State]['AccAP'], dist = 'norm', plot = plt)\n    plt.title(\"Probability Plot - \" + State)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:55:38.156008Z","iopub.execute_input":"2023-04-09T22:55:38.156393Z","iopub.status.idle":"2023-04-09T22:55:59.020289Z","shell.execute_reply.started":"2023-04-09T22:55:38.156356Z","shell.execute_reply":"2023-04-09T22:55:59.018871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(2,1, figsize = (22,14))\nax = axes.ravel()\n\n#['AKinetic' 'Turn' 'Hesitate' 'Walk']\n\ndf.loc[df.state == 'AKinetic'].AccAP.plot(kind = 'hist', ax = ax[0], alpha = 0.5, bins = 50, sharex = True, label = 'A-Kinetic', edgecolor = 'black', linewidth = 3)\ndf.loc[df.state == 'Turn'].AccAP.plot(kind = 'hist', ax = ax[0], alpha = 0.5, bins = 50, sharex = True, label = 'Turn', edgecolor = 'blue', linewidth = 3)\ndf.loc[df.state == 'Hesitate'].AccAP.plot(kind = 'hist', ax = ax[0], alpha = 0.5, bins = 50, sharex = True, label = 'Hesitate', edgecolor = 'red', linewidth = 3)\ndf.loc[df.state == 'Walk'].AccAP.plot(kind = 'hist', ax = ax[0], alpha = 0.5, bins = 50, sharex = True, label = 'Walk', edgecolor = 'green', linewidth = 3)\n\n\n\nfig.suptitle(\"Histogram and Density plot of AccAP by state\", fontsize = 20)\n\ndf.loc[df.state == 'AKinetic'].AccAP.plot(kind = 'density', ax = ax[1], alpha = 0.5, label = 'A-Kinetic', linewidth = 3)\ndf.loc[df.state == 'Turn'].AccAP.plot(kind = 'density', ax = ax[1], alpha = 0.5, label = 'Turn', linewidth = 3)\ndf.loc[df.state == 'Hesitate'].AccAP.plot(kind = 'density', ax = ax[1], alpha = 0.5, label = 'Hesitate', linewidth = 3)\ndf.loc[df.state == 'Walk'].AccAP.plot(kind = 'density', ax = ax[1], alpha = 0.5, label = 'Walk', linewidth = 3)\n\nax[0].legend(fontsize = 18)\nax[0].set_xlim(-8, 8)\nax[0].set_title('Histogram of AccAP based on state')\nax[0].tick_params(axis = 'both', which = 'major', labelsize = 14)\nax[0].tick_params(axis = 'both', which = 'minor', labelsize = 14)\n\nax[1].legend(fontsize = 18)\nax[1].set_xlim(-8,8)\nax[1].set_title('Density of AccAP based on state')\nax[1].tick_params(axis = 'both', which = 'major', labelsize = 14)\nax[1].tick_params(axis = 'both', which = 'minor', labelsize = 14)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:55:59.022037Z","iopub.execute_input":"2023-04-09T22:55:59.022444Z","iopub.status.idle":"2023-04-09T22:58:04.509197Z","shell.execute_reply.started":"2023-04-09T22:55:59.022403Z","shell.execute_reply":"2023-04-09T22:58:04.507861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AccAP_AKin = df.loc[df['state']=='AKinetic','AccAP']\n\nAccAP_Walk = df.loc[df['state']=='Walk','AccAP']\n\nAccAP_Turn = df.loc[df['state']=='Turn','AccAP']\n\nAccAP_Hesitate = df.loc[df['state']=='Hesitate','AccAP']","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:58:04.510922Z","iopub.execute_input":"2023-04-09T22:58:04.511327Z","iopub.status.idle":"2023-04-09T22:58:06.310161Z","shell.execute_reply.started":"2023-04-09T22:58:04.511288Z","shell.execute_reply":"2023-04-09T22:58:06.308823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"boxplot = df.boxplot(column=['AccAP'], by='state')","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:58:06.312673Z","iopub.execute_input":"2023-04-09T22:58:06.313084Z","iopub.status.idle":"2023-04-09T22:58:17.274007Z","shell.execute_reply.started":"2023-04-09T22:58:06.313047Z","shell.execute_reply":"2023-04-09T22:58:17.272720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(stats.ks_2samp(AccAP_AKin, AccAP_Walk))\nprint(cohend(AccAP_AKin, AccAP_Walk))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:58:17.275678Z","iopub.execute_input":"2023-04-09T22:58:17.276089Z","iopub.status.idle":"2023-04-09T22:58:18.382993Z","shell.execute_reply.started":"2023-04-09T22:58:17.276051Z","shell.execute_reply":"2023-04-09T22:58:18.381293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(stats.ks_2samp(AccAP_AKin, AccAP_Turn))\nprint(cohend(AccAP_AKin, AccAP_Turn))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:58:18.384590Z","iopub.execute_input":"2023-04-09T22:58:18.385073Z","iopub.status.idle":"2023-04-09T22:58:19.816389Z","shell.execute_reply.started":"2023-04-09T22:58:18.385035Z","shell.execute_reply":"2023-04-09T22:58:19.814894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(stats.ks_2samp(AccAP_AKin, AccAP_Hesitate))\nprint(cohend(AccAP_AKin, AccAP_Hesitate))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:58:19.817850Z","iopub.execute_input":"2023-04-09T22:58:19.818217Z","iopub.status.idle":"2023-04-09T22:58:20.946258Z","shell.execute_reply.started":"2023-04-09T22:58:19.818183Z","shell.execute_reply":"2023-04-09T22:58:20.944955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(stats.ks_2samp(AccAP_Walk, AccAP_Turn))\nprint(cohend(AccAP_Walk, AccAP_Turn))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:58:20.947784Z","iopub.execute_input":"2023-04-09T22:58:20.948162Z","iopub.status.idle":"2023-04-09T22:58:21.328823Z","shell.execute_reply.started":"2023-04-09T22:58:20.948125Z","shell.execute_reply":"2023-04-09T22:58:21.327444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(stats.ks_2samp(AccAP_Walk, AccAP_Hesitate))\nprint(cohend(AccAP_Walk, AccAP_Hesitate))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:58:21.330577Z","iopub.execute_input":"2023-04-09T22:58:21.331486Z","iopub.status.idle":"2023-04-09T22:58:21.557676Z","shell.execute_reply.started":"2023-04-09T22:58:21.331386Z","shell.execute_reply":"2023-04-09T22:58:21.556225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(stats.ks_2samp(AccAP_Turn, AccAP_Hesitate))\nprint(cohend(AccAP_Turn, AccAP_Hesitate))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:58:21.559044Z","iopub.execute_input":"2023-04-09T22:58:21.559374Z","iopub.status.idle":"2023-04-09T22:58:21.957967Z","shell.execute_reply.started":"2023-04-09T22:58:21.559341Z","shell.execute_reply":"2023-04-09T22:58:21.956669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option(\"display.max_rows\", 999)\ndf.groupby('state')[['AccV','AccML','AccAP']].describe().unstack()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T22:58:21.959404Z","iopub.execute_input":"2023-04-09T22:58:21.959758Z","iopub.status.idle":"2023-04-09T22:58:23.845377Z","shell.execute_reply.started":"2023-04-09T22:58:21.959724Z","shell.execute_reply":"2023-04-09T22:58:23.843900Z"},"trusted":true},"execution_count":null,"outputs":[]}]}