{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-21T18:32:22.403042Z","iopub.execute_input":"2023-03-21T18:32:22.403564Z","iopub.status.idle":"2023-03-21T18:32:22.411136Z","shell.execute_reply.started":"2023-03-21T18:32:22.403523Z","shell.execute_reply":"2023-03-21T18:32:22.409628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Summary of what I found thus far\n* A Turn event is the most common in any given dataseries, with no other types of events occurring before or after (59%) followed by a Turn and Walking Event occurring in the same session (14.5%).\n* DeFOG dataseries has significantly more events per series/Id than the tDCS series. \n    * The DeFOG data series in terms of # of events per series/Id has a mean of 10.8, median of 7.5 and MAD of 10.6\n    * The tDCS data series in terms of # of events per series/Id has a mean of 2.9, median of 2 and MAD of 1.44.\n    * I need to look at the duration of the sessions to see if this was a big influence. ","metadata":{}},{"cell_type":"markdown","source":"\n\nTo start, let import the high level data, events, subjects, tasks and the metadata that we have from each data source.\n\n","metadata":{}},{"cell_type":"code","source":"events = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/events.csv')\n# For events, I'm going to add in a column that calculates the duration of the event, \n# as I'm curious as to how long these events are lasting.\nevents['eventDuration'] = events['Completion'] - events['Init']\nevents.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:22.434840Z","iopub.execute_input":"2023-03-21T18:32:22.436038Z","iopub.status.idle":"2023-03-21T18:32:22.462692Z","shell.execute_reply.started":"2023-03-21T18:32:22.435983Z","shell.execute_reply":"2023-03-21T18:32:22.461328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/subjects.csv')\nsubjects.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:22.470415Z","iopub.execute_input":"2023-03-21T18:32:22.470873Z","iopub.status.idle":"2023-03-21T18:32:22.494823Z","shell.execute_reply.started":"2023-03-21T18:32:22.470834Z","shell.execute_reply":"2023-03-21T18:32:22.493553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tasks = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tasks.csv')\ntasks.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:22.511826Z","iopub.execute_input":"2023-03-21T18:32:22.512328Z","iopub.status.idle":"2023-03-21T18:32:22.532290Z","shell.execute_reply.started":"2023-03-21T18:32:22.512285Z","shell.execute_reply":"2023-03-21T18:32:22.530864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Meta data\ndefogMeta = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv')\ndefogMeta.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:22.544293Z","iopub.execute_input":"2023-03-21T18:32:22.545662Z","iopub.status.idle":"2023-03-21T18:32:22.563085Z","shell.execute_reply.started":"2023-03-21T18:32:22.545593Z","shell.execute_reply":"2023-03-21T18:32:22.561682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfogMeta = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv')\ntdcsfogMeta.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:22.588812Z","iopub.execute_input":"2023-03-21T18:32:22.589872Z","iopub.status.idle":"2023-03-21T18:32:22.611001Z","shell.execute_reply.started":"2023-03-21T18:32:22.589793Z","shell.execute_reply":"2023-03-21T18:32:22.609280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dailyMeta = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/daily_metadata.csv')\ndailyMeta.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:22.613776Z","iopub.execute_input":"2023-03-21T18:32:22.614340Z","iopub.status.idle":"2023-03-21T18:32:22.633103Z","shell.execute_reply.started":"2023-03-21T18:32:22.614272Z","shell.execute_reply":"2023-03-21T18:32:22.632023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that three columns that we'll be able to join on are Id (event id), Subject, and Visit (visit only being available for daily and defog) to help us pull our data together. Another thing to note is that the daily events denote the recordings by the time of day, not by seconds during the study. We may want to convert this to make our data more uniform, depending on how we end up using it. (if initial and/or completion times become variables, or if we just use the duration of the event, etc)\n\nBefore we start joining our data together and looking for any basic trends, its worth while to check for missing data so that we're at least aware and can then decide how to handle it later. ","metadata":{}},{"cell_type":"code","source":"events.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:22.634757Z","iopub.execute_input":"2023-03-21T18:32:22.635454Z","iopub.status.idle":"2023-03-21T18:32:22.646935Z","shell.execute_reply.started":"2023-03-21T18:32:22.635413Z","shell.execute_reply":"2023-03-21T18:32:22.645583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It would appear that we found the data that has notype, where the annotator couldn't determien the type, so left it ambiguous.","metadata":{}},{"cell_type":"code","source":"subjects.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:22.654434Z","iopub.execute_input":"2023-03-21T18:32:22.654934Z","iopub.status.idle":"2023-03-21T18:32:22.668097Z","shell.execute_reply.started":"2023-03-21T18:32:22.654887Z","shell.execute_reply":"2023-03-21T18:32:22.666701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The NAs in the Visit column also checks out, as we were told only daily and defog would have visit available, so these must be the tdcsfog subjects. \n\nThe NAs for UPDRSII_On and UPDRSII_Off was not previously known. This si a rating scale while the patient is On and Off medication so its possible that those with NA just don't have scores either on or off medication. We can determine how we'll handle these missing values later on.","metadata":{}},{"cell_type":"code","source":"tasks.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:22.676475Z","iopub.execute_input":"2023-03-21T18:32:22.677199Z","iopub.status.idle":"2023-03-21T18:32:22.687900Z","shell.execute_reply.started":"2023-03-21T18:32:22.677158Z","shell.execute_reply":"2023-03-21T18:32:22.686484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defogMeta.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:22.703386Z","iopub.execute_input":"2023-03-21T18:32:22.704722Z","iopub.status.idle":"2023-03-21T18:32:22.716290Z","shell.execute_reply.started":"2023-03-21T18:32:22.704664Z","shell.execute_reply":"2023-03-21T18:32:22.715061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfogMeta.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:22.734660Z","iopub.execute_input":"2023-03-21T18:32:22.735912Z","iopub.status.idle":"2023-03-21T18:32:22.747231Z","shell.execute_reply.started":"2023-03-21T18:32:22.735852Z","shell.execute_reply":"2023-03-21T18:32:22.745585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dailyMeta.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:22.764468Z","iopub.execute_input":"2023-03-21T18:32:22.764995Z","iopub.status.idle":"2023-03-21T18:32:22.776435Z","shell.execute_reply.started":"2023-03-21T18:32:22.764952Z","shell.execute_reply":"2023-03-21T18:32:22.775325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Exploration for DeFoG Metadata, subjects and events","metadata":{}},{"cell_type":"markdown","source":"Ok perfect, now we know a little more about what we're working with. Lets start joining our data together so then we can start to get a sense for how its distributed and see if we can find any basic relationships between the subjects and the types of events, # of events, duration of events, etc. \n\nTo start, lets join the metadata of the at home defog test recording together with information about each subject.  To do this, we'll join on Subject and Visit. To check if this worked correctly, our joined df should have the same number of rows as the metadata from the at home defog recordings","metadata":{}},{"cell_type":"code","source":"defogMetaSubjects = pd.merge(defogMeta, subjects, on=['Subject', 'Visit'])\nprint(defogMeta.Id.count())\nprint(defogMetaSubjects.Id.count())\n\ndefogMetaSubjects.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:22.789434Z","iopub.execute_input":"2023-03-21T18:32:22.790259Z","iopub.status.idle":"2023-03-21T18:32:22.817406Z","shell.execute_reply.started":"2023-03-21T18:32:22.790210Z","shell.execute_reply":"2023-03-21T18:32:22.816067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=3, ncols=3)\n\ndefogMetaSubjects.Medication.value_counts().plot.bar(ax=axes[0,0])\ndefogMetaSubjects.Age.plot.box(ax=axes[0,1])\ndefogMetaSubjects.Sex.value_counts().plot.bar(ax=axes[0,2])\ndefogMetaSubjects.YearsSinceDx.plot.box(ax=axes[1,0])\ndefogMetaSubjects.UPDRSIII_On.plot.box(ax=axes[1,1])\ndefogMetaSubjects.UPDRSIII_Off.plot.box(ax=axes[1,2])\ndefogMetaSubjects.NFOGQ.plot.box(ax=axes[2,1])\n\n\naxes[0,0].set_xlabel('Medication')\naxes[0,0].set_ylabel('Count')\naxes[0,1].set_ylabel('Age')\naxes[0,2].set_xlabel('Sex')\naxes[0,2].set_ylabel('Count')\naxes[1,0].set_ylabel('Years')\naxes[1,1].set_ylabel('Score')\naxes[1,2].set_ylabel('Score')\naxes[2,1].set_ylabel('Score')\nfig.set_figwidth(9)\nfig.set_figheight(9)\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:22.819486Z","iopub.execute_input":"2023-03-21T18:32:22.820110Z","iopub.status.idle":"2023-03-21T18:32:24.038458Z","shell.execute_reply.started":"2023-03-21T18:32:22.820071Z","shell.execute_reply":"2023-03-21T18:32:24.036716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Then we can join it together with the events and tasks data to get a full picture of each recording and the events that occured in each. ","metadata":{}},{"cell_type":"code","source":"defogMetaSubjectsEvents = pd.merge(defogMetaSubjects, events, on=['Id'])\nprint(defogMetaSubjectsEvents.Id.count())\ndefogMetaSubjectsEvents.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:24.041285Z","iopub.execute_input":"2023-03-21T18:32:24.041848Z","iopub.status.idle":"2023-03-21T18:32:24.074195Z","shell.execute_reply.started":"2023-03-21T18:32:24.041792Z","shell.execute_reply":"2023-03-21T18:32:24.072973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defogMetaSubjectsEvents.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:24.075834Z","iopub.execute_input":"2023-03-21T18:32:24.076753Z","iopub.status.idle":"2023-03-21T18:32:24.091375Z","shell.execute_reply.started":"2023-03-21T18:32:24.076707Z","shell.execute_reply":"2023-03-21T18:32:24.089970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Just our unannotated defog data again. We don't know what type of event occurred, but just that one did. May make sense to just assume its one type of event or another to spend our training set if one type of event is dominant.","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=1, ncols=2)\n\neventsPerRecording = defogMetaSubjectsEvents.groupby('Id').Type.count()\n\neventsPerRecording.plot.box(ax=axes[0])\nq_low = defogMetaSubjectsEvents.eventDuration.quantile(0.01)\nq_hi  = defogMetaSubjectsEvents.eventDuration.quantile(0.99)\n\ndefogMetaSubjectsEvents[(defogMetaSubjectsEvents.eventDuration < q_hi) & (defogMetaSubjectsEvents.eventDuration > q_low)].eventDuration.plot.hist(ax=axes[1])\n\naxes[0].set_xlabel('# of Events per Recording')\naxes[0].set_ylabel('Count')\n\n\nfig.set_figwidth(9)\nfig.set_figheight(3)\nfig.tight_layout()\neventsPerRecordingBySex = defogMetaSubjectsEvents.groupby(['Sex','Id']).aggregate({'Type': 'count'})\nprint(eventsPerRecordingBySex.Type.groupby('Sex').mean())","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:24.095911Z","iopub.execute_input":"2023-03-21T18:32:24.096469Z","iopub.status.idle":"2023-03-21T18:32:24.470977Z","shell.execute_reply.started":"2023-03-21T18:32:24.096424Z","shell.execute_reply":"2023-03-21T18:32:24.469515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=1, ncols=2)\nax = sns.violinplot(defogMetaSubjectsEvents, x='Type', y='eventDuration', inner='quartile', ax=axes[0])\n\neventsPerRecording = defogMetaSubjectsEvents.groupby('Id').Type.count()\n\neventsPerRecording.plot.hist(ax=axes[1])\nq_low = defogMetaSubjectsEvents.eventDuration.quantile(0.01)\nq_hi  = defogMetaSubjectsEvents.eventDuration.quantile(0.99)\n\ndefogMetaSubjectsEvents[(defogMetaSubjectsEvents.eventDuration < q_hi) & (defogMetaSubjectsEvents.eventDuration > q_low)].eventDuration.plot.hist(ax=axes[1])\naxes[1].set_xlabel('Duration of Events\\n(excluding outliers)')\naxes[1].set_ylabel('Count')\nfig.set_figwidth(12)\nfig.set_figheight(4)\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:24.472848Z","iopub.execute_input":"2023-03-21T18:32:24.473785Z","iopub.status.idle":"2023-03-21T18:32:24.920125Z","shell.execute_reply.started":"2023-03-21T18:32:24.473740Z","shell.execute_reply":"2023-03-21T18:32:24.918893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=1, ncols=1)\nax = sns.violinplot(defogMetaSubjectsEvents, x='Sex', y='eventDuration',  inner='quartile', ax=axes)\nfig.set_figwidth(8)\nfig.set_figheight(4)\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:24.922662Z","iopub.execute_input":"2023-03-21T18:32:24.923201Z","iopub.status.idle":"2023-03-21T18:32:25.212328Z","shell.execute_reply.started":"2023-03-21T18:32:24.923146Z","shell.execute_reply":"2023-03-21T18:32:25.210722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defogByType = defogMetaSubjectsEvents.groupby('Type').aggregate({'Id': 'count'})\nprint(defogByType)\ndefogByType.plot.bar()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:25.214273Z","iopub.execute_input":"2023-03-21T18:32:25.215329Z","iopub.status.idle":"2023-03-21T18:32:25.468925Z","shell.execute_reply.started":"2023-03-21T18:32:25.215281Z","shell.execute_reply":"2023-03-21T18:32:25.467794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Turn events occurring by far the most often, 78% of the time in the defog dataset.\n\nNow lets check out events per subject.","metadata":{}},{"cell_type":"code","source":"defogByTypeBySubject = defogMetaSubjectsEvents.groupby(['Id', 'Subject', 'Type']).aggregate({'Id': 'count'})\ndefogByTypeBySubject.groupby('Type').median().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:25.470342Z","iopub.execute_input":"2023-03-21T18:32:25.470719Z","iopub.status.idle":"2023-03-21T18:32:25.669636Z","shell.execute_reply.started":"2023-03-21T18:32:25.470680Z","shell.execute_reply":"2023-03-21T18:32:25.668534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Turn events are the most common total and also the most common per subject per session. With StartHesitation and Walking events factors lower.\n\nGiven that we won't have much of this data at test time, combing through it may not be too helpful. That being said, 1/3rd of our defog training data is unannotated. We just got some useful info here, that 78% are Turn events. It may also be helpful to see if events can occur together or if they are generally distinct. ie will a subject have Turn event and then a Walking event. ","metadata":{}},{"cell_type":"code","source":"uniquedefogTypeSets = defogMetaSubjectsEvents[defogMetaSubjectsEvents['Type'].notna()].groupby(['Id','Subject']).aggregate({'Type': lambda x: set(x.unique())})\nuniquedefogTypeSets['Type'].value_counts().plot.bar()\ntotal = uniquedefogTypeSets['Type'].count() \nprint(total)\nprint(uniquedefogTypeSets['Type'].value_counts() / total)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:25.671299Z","iopub.execute_input":"2023-03-21T18:32:25.672030Z","iopub.status.idle":"2023-03-21T18:32:25.928923Z","shell.execute_reply.started":"2023-03-21T18:32:25.671985Z","shell.execute_reply":"2023-03-21T18:32:25.927580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Within the defog dataseries, Turn and Walking events most often occur together, happening 55%% of the time, where as a Turn event alone occurred 36% of the time.","metadata":{}},{"cell_type":"markdown","source":"# Data Exploration for TDCS FoG Metadata, subjects and events","metadata":{}},{"cell_type":"markdown","source":"Alright now lets give TDCS the same treatment","metadata":{}},{"cell_type":"code","source":"tdcsfogMetaSubjects = pd.merge(tdcsfogMeta, subjects, on=['Subject'])\ntdcsfogMetaSubjects.drop('Visit_y', inplace=True, axis=1)\ntdcsfogMetaSubjects.rename({'Visit_x': 'Visit'}, inplace=True, axis=1)\nprint(tdcsfogMeta.Id.count())\nprint(tdcsfogMetaSubjects.Id.count())\n\ntdcsfogMetaSubjects.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:25.932956Z","iopub.execute_input":"2023-03-21T18:32:25.933404Z","iopub.status.idle":"2023-03-21T18:32:25.966343Z","shell.execute_reply.started":"2023-03-21T18:32:25.933358Z","shell.execute_reply":"2023-03-21T18:32:25.964878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=3, ncols=3)\n\ntdcsfogMetaSubjects.Medication.value_counts().plot.bar(ax=axes[0,0])\ntdcsfogMetaSubjects.Age.plot.box(ax=axes[0,1])\ntdcsfogMetaSubjects.Sex.value_counts().plot.bar(ax=axes[0,2])\ntdcsfogMetaSubjects.YearsSinceDx.plot.box(ax=axes[1,0])\ntdcsfogMetaSubjects.UPDRSIII_On.plot.box(ax=axes[1,1])\ntdcsfogMetaSubjects.UPDRSIII_Off.plot.box(ax=axes[1,2])\ntdcsfogMetaSubjects.NFOGQ.plot.box(ax=axes[2,1])\n\n\naxes[0,0].set_xlabel('Medication')\naxes[0,0].set_ylabel('Count')\naxes[0,1].set_ylabel('Age')\naxes[0,2].set_xlabel('Sex')\naxes[0,2].set_ylabel('Count')\naxes[1,0].set_ylabel('Years')\naxes[1,1].set_ylabel('Score')\naxes[1,2].set_ylabel('Score')\naxes[2,1].set_ylabel('Score')\nfig.set_figwidth(9)\nfig.set_figheight(9)\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:25.968247Z","iopub.execute_input":"2023-03-21T18:32:25.969509Z","iopub.status.idle":"2023-03-21T18:32:27.195737Z","shell.execute_reply.started":"2023-03-21T18:32:25.969451Z","shell.execute_reply":"2023-03-21T18:32:27.194343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfogMetaSubjectsEvents = pd.merge(tdcsfogMetaSubjects, events, on=['Id'])\nprint(tdcsfogMetaSubjects.Id.count())\nprint(tdcsfogMetaSubjectsEvents.Id.count())\n\ntdcsfogMetaSubjectsEvents.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:27.197324Z","iopub.execute_input":"2023-03-21T18:32:27.197834Z","iopub.status.idle":"2023-03-21T18:32:27.231785Z","shell.execute_reply.started":"2023-03-21T18:32:27.197793Z","shell.execute_reply":"2023-03-21T18:32:27.230419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfogMetaSubjectsEvents.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:27.233307Z","iopub.execute_input":"2023-03-21T18:32:27.234544Z","iopub.status.idle":"2023-03-21T18:32:27.248626Z","shell.execute_reply.started":"2023-03-21T18:32:27.234487Z","shell.execute_reply":"2023-03-21T18:32:27.247160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=1, ncols=2)\n\neventsPerRecording = tdcsfogMetaSubjectsEvents.groupby('Id').Type.count()\n\neventsPerRecording.plot.box(ax=axes[0])\nq_low = tdcsfogMetaSubjectsEvents.eventDuration.quantile(0.01)\nq_hi  = tdcsfogMetaSubjectsEvents.eventDuration.quantile(0.99)\n\ntdcsfogMetaSubjectsEvents[(tdcsfogMetaSubjectsEvents.eventDuration < q_hi) & (tdcsfogMetaSubjectsEvents.eventDuration > q_low)].eventDuration.plot.hist(ax=axes[1])\n\naxes[0].set_xlabel('# of Events per Recording')\naxes[0].set_ylabel('Count')\naxes[1].set_xlabel('Duration of Events\\n(excluding outliers)')\naxes[1].set_ylabel('Count')\n\nfig.set_figwidth(9)\nfig.set_figheight(3)\nfig.tight_layout()\neventsPerRecordingBySex = tdcsfogMetaSubjectsEvents.groupby(['Sex','Id']).aggregate({'Type': 'count'})\nprint(eventsPerRecordingBySex.Type.groupby('Sex').mean())","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:27.250285Z","iopub.execute_input":"2023-03-21T18:32:27.251644Z","iopub.status.idle":"2023-03-21T18:32:27.622159Z","shell.execute_reply.started":"2023-03-21T18:32:27.251575Z","shell.execute_reply":"2023-03-21T18:32:27.620640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"More missing UPDRSIII_On and Off data, will need to decide best way to fill this in.\n","metadata":{}},{"cell_type":"code","source":"ax = sns.violinplot(tdcsfogMetaSubjectsEvents, x='Type', y='eventDuration', scale='count', inner='quartile')","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:27.623962Z","iopub.execute_input":"2023-03-21T18:32:27.624588Z","iopub.status.idle":"2023-03-21T18:32:27.818399Z","shell.execute_reply.started":"2023-03-21T18:32:27.624532Z","shell.execute_reply":"2023-03-21T18:32:27.816729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfogByType = tdcsfogMetaSubjectsEvents.groupby('Type').aggregate({'Id': 'count'})\nprint(tdcsfogByType)\ntdcsfogByType.plot.bar()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:27.820346Z","iopub.execute_input":"2023-03-21T18:32:27.821080Z","iopub.status.idle":"2023-03-21T18:32:28.064724Z","shell.execute_reply.started":"2023-03-21T18:32:27.821036Z","shell.execute_reply":"2023-03-21T18:32:28.063384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"uniquetdcsfogTypeSets = tdcsfogMetaSubjectsEvents.groupby(['Id','Subject']).aggregate({'Type': lambda x: set(x.unique())})\nuniquetdcsfogTypeSets['Type'].value_counts().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:28.066572Z","iopub.execute_input":"2023-03-21T18:32:28.067077Z","iopub.status.idle":"2023-03-21T18:32:28.356503Z","shell.execute_reply.started":"2023-03-21T18:32:28.067037Z","shell.execute_reply":"2023-03-21T18:32:28.355152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"allFog = tdcsfogMetaSubjectsEvents.append(defogMetaSubjectsEvents)\nuniqueallFog = allFog.groupby(['Id','Subject']).aggregate({'Type': lambda x: set(x.unique())})\ntotalUniqueSets = uniqueallFog.count().item()\nprint((uniqueallFog['Type'].value_counts() / totalUniqueSets) )\n\nuniqueallFog['Type'].value_counts().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:28.358310Z","iopub.execute_input":"2023-03-21T18:32:28.358836Z","iopub.status.idle":"2023-03-21T18:32:28.687031Z","shell.execute_reply.started":"2023-03-21T18:32:28.358793Z","shell.execute_reply":"2023-03-21T18:32:28.685690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"allEventsPerRecording = allFog.groupby('Id').Type.count()\nprint(allEventsPerRecording.mean())\nprint(allEventsPerRecording.median())\nprint(allEventsPerRecording.mad())\nprint(allEventsPerRecording.value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-03-21T18:32:28.688567Z","iopub.execute_input":"2023-03-21T18:32:28.688991Z","iopub.status.idle":"2023-03-21T18:32:28.705056Z","shell.execute_reply.started":"2023-03-21T18:32:28.688953Z","shell.execute_reply":"2023-03-21T18:32:28.703552Z"},"trusted":true},"execution_count":null,"outputs":[]}]}