{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport random\nimport seaborn as sns\nimport os\nimport matplotlib.pyplot as plt\n\n# each table has recordings with a different frequency\nfreq_dict = dict(\n    tdcsfog=128,\n    defog=100,\n    daily=100,\n    notype=100\n)\n\nevent_values = [\"StartHesitation\", \"Turn\", \"Walking\"]\nmagnitudes = [\"AccV\", \"AccML\", \"AccAP\"]\n\ninput_folder = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction\"\ntrain_folder = f\"{input_folder}/train\"\ntest_folder = f\"{input_folder}/test\"\nunlabeled_folder = f\"{input_folder}/unlabeled\"\n\ntrain_defog_folder = f\"{train_folder}/defog\"\ntest_defog_folder = f\"{test_folder}/defog\"\ntrain_tdcsfog_folder = f\"{train_folder}/tdcsfog\"\ntest_tdcsfog_folder = f\"{test_folder}/tdcsfog\"\ntrain_notype_folder = f\"{train_folder}/notype\"\n\ntrain_defog = os.listdir(train_defog_folder)\ntest_defog = os.listdir(test_defog_folder)\ntrain_tdcsfog = os.listdir(train_tdcsfog_folder)\ntest_tdcsfog = os.listdir(test_tdcsfog_folder)\ntrain_notype = os.listdir(train_notype_folder)\nunlabeled = os.listdir(unlabeled_folder)\n\nevents = pd.read_csv(f\"{input_folder}/events.csv\")\ntasks = pd.read_csv(f\"{input_folder}/tasks.csv\")\n\nprint(\"Train tables for defog:\", len(train_defog))\nprint(\"Test tables for defog:\", len(test_defog))\n\nprint(\"Train tables for tdcsfog:\", len(train_tdcsfog))\nprint(\"Test tables for tdcsfog:\", len(test_tdcsfog))\n\nprint(\"Train tables for notype:\", len(train_notype))\n\nprint(\"Unlabeled tables\", len(unlabeled))\n\ndef read_data_series(path, labeled=True):\n    file_id = path.split(\"/\")[-1].split(\".\")[0]\n    print(file_id)\n    df = pd.read_csv(path, index_col=\"Time\")\n    for series_type in [\"defog\", \"tdcsfog\", \"notype\"]:\n        if series_type in path:\n            freq = freq_dict[series_type]\n    df[\"seconds\"] = df.index / freq\n    print(f\"Total duration: {df.seconds.max()/60:.2f} minutes\")\n    if labeled:\n        for event in event_values:\n            print(f\"{df[df[event].eq(1)].size/freq/60:.2f} minutes of {event}\")\n    return df, file_id\n\ndef plot_data_series(df, file_id):\n    fig, ax = plt.subplots(figsize=(15, 3))\n    for var in magnitudes:\n        sns.lineplot(\n            data=df,\n            x=df.seconds,\n            y=var,\n            ax=ax\n        )\n    ax.set_ylabel(\"Acceleration\")\n    \n    type_color = dict(\n        StartHesitation=\"green\",\n        Turn=\"red\",\n        Walking=\"blue\"\n    )\n    \n    kin_dict = {0: \"akinetic\", 1: \"kinetic\"}\n    \n    for _, event in events[events.Id.eq(file_id)].iterrows():\n        ax.axvspan(xmin=event.Init,\n                   xmax=event.Completion,\n                   alpha=0.2,\n                   facecolor=type_color.get(event.Type, \"orange\"),\n                   label=f\"{event.Type}\"\n                  )\n\n    for _, task in tasks[tasks.Id.eq(file_id)].iterrows():\n        ax.axvspan(xmin=task.Begin,\n                   xmax=task.End,\n                   alpha=0,\n                   hatch=\"//\",\n                   label=\"Task\"\n                  )\n\n    handles, labels = plt.gca().get_legend_handles_labels()\n    by_label = dict(zip(labels, handles))\n    ax.legend(by_label.values(), by_label.keys())\n    #ax.legend()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-20T21:59:36.530515Z","iopub.execute_input":"2023-03-20T21:59:36.531028Z","iopub.status.idle":"2023-03-20T21:59:36.571315Z","shell.execute_reply.started":"2023-03-20T21:59:36.530986Z","shell.execute_reply":"2023-03-20T21:59:36.569776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Metadata\n\n##  tDCS FOG metadata\nIdentifies each series in the tdcsfog dataset by a unique Subject, Visit, Test, Medication condition.\n\n* `Visit` Lab visits consist of a baseline assessment, two post-treatment assessments for different treatment stages, and one follow-up assessment.\n* `Test` Which of three test types was performed, with 3 the most challenging.\n* `Medication` Subjects may have been either off or on anti-parkinsonian medication during the recording.\n\n","metadata":{}},{"cell_type":"code","source":"tdcsfog_metadata = pd.read_csv(f\"{input_folder}/tdcsfog_metadata.csv\")\ntdcsfog_metadata.groupby(\"Subject\").agg(lambda x: list(x)).sample(3)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T21:59:36.690762Z","iopub.execute_input":"2023-03-20T21:59:36.691228Z","iopub.status.idle":"2023-03-20T21:59:36.722748Z","shell.execute_reply.started":"2023-03-20T21:59:36.691186Z","shell.execute_reply":"2023-03-20T21:59:36.721262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## DeFOG metadata\nIdentifies each series in the defog dataset by a unique Subject, Visit, Medication condition.","metadata":{}},{"cell_type":"code","source":"defog_metadata = pd.read_csv(f\"{input_folder}/defog_metadata.csv\")\ndefog_metadata.groupby(\"Subject\").agg(lambda x: list(x)).sample(3)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T21:59:36.802472Z","iopub.execute_input":"2023-03-20T21:59:36.802936Z","iopub.status.idle":"2023-03-20T21:59:36.829485Z","shell.execute_reply.started":"2023-03-20T21:59:36.802892Z","shell.execute_reply":"2023-03-20T21:59:36.827939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Daily metadata\n\nEach series in the daily dataset is identified by the Subject id. This file also contains the time of day the recording began.","metadata":{}},{"cell_type":"code","source":"daily_metadata = pd.read_csv(f\"{input_folder}/daily_metadata.csv\")\ndaily_metadata.groupby(\"Subject\").agg(lambda x: list(x)).sample(3)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T21:59:36.916283Z","iopub.execute_input":"2023-03-20T21:59:36.917166Z","iopub.status.idle":"2023-03-20T21:59:36.942990Z","shell.execute_reply.started":"2023-03-20T21:59:36.917117Z","shell.execute_reply":"2023-03-20T21:59:36.941776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Subjects metadata\nMetadata for each `Subject` in the study, including their `Age` and `Sex` as well as:\n\n* `Visit` Only available for subjects in the `daily` and `defog` datasets.\n* `YearsSinceDx` Years since Parkinson's diagnosis.\n* `UPDRSIIIOn`/`UPDRSIIIOff` Unified Parkinson's Disease Rating Scale score during on/off medication respectively.\n* `NFOGQ` Self-report FoG questionnaire score. See:\n    https://pubmed.ncbi.nlm.nih.gov/19660949/","metadata":{}},{"cell_type":"code","source":"subjects = pd.read_csv(f\"{input_folder}/subjects.csv\")\nsubjects.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T21:59:37.031762Z","iopub.execute_input":"2023-03-20T21:59:37.032254Z","iopub.status.idle":"2023-03-20T21:59:37.053359Z","shell.execute_reply.started":"2023-03-20T21:59:37.032209Z","shell.execute_reply":"2023-03-20T21:59:37.052329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Events metadata\nMetadata for each FoG event in all data series. The event times agree with the labels in the data series.\n* `Id` The data series the event occured in.\n* `Init Time` (s) the event began.\n* `Completion` Time (s) the event ended.\n* `Type` Whether StartHesitation, Turn, or Walking.\n* `Kinetic` Whether the event was kinetic (1) and involved movement, or akinetic (0) and static.","metadata":{}},{"cell_type":"code","source":"events.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T21:59:37.089275Z","iopub.execute_input":"2023-03-20T21:59:37.089765Z","iopub.status.idle":"2023-03-20T21:59:37.108156Z","shell.execute_reply.started":"2023-03-20T21:59:37.089720Z","shell.execute_reply":"2023-03-20T21:59:37.106845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Tasks metadata\nTask metadata for series in the `defog` dataset. (Not relevant for the series in the `fog` or `daily` datasets.)\n* `Id` The data series where the task was measured.\n* `Begin` Time (s) the task began.\n* `End` Time (s) the task ended.\n* `Task` One of seven tasks types in the DeFOG protocol, described on [this page](https://www.kaggle.com/competitions/tlvmc-parkinsons-freezing-gait-prediction/overview/additional-data-documentation).","metadata":{}},{"cell_type":"code","source":"tasks.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T21:59:37.157024Z","iopub.execute_input":"2023-03-20T21:59:37.157486Z","iopub.status.idle":"2023-03-20T21:59:37.171486Z","shell.execute_reply.started":"2023-03-20T21:59:37.157437Z","shell.execute_reply":"2023-03-20T21:59:37.170147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Data\n\n## Train DeFOG\n\nThe DeFOG (defog) dataset, comprising data series collected in the subject's home, as subjects completed a FOG-provoking protocol.\n\nNote that the Valid and Task fields are only present in the defog dataset. They are not relevant for the tdcsfog data.","metadata":{}},{"cell_type":"code","source":"df_defog, df_defog_id = read_data_series(f\"{train_defog_folder}/{random.sample(train_defog, 1)[0]}\")\ndf_defog.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T21:59:37.264732Z","iopub.execute_input":"2023-03-20T21:59:37.265178Z","iopub.status.idle":"2023-03-20T21:59:37.543484Z","shell.execute_reply.started":"2023-03-20T21:59:37.265138Z","shell.execute_reply":"2023-03-20T21:59:37.541974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_data_series(df_defog, df_defog_id)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T21:59:37.545512Z","iopub.execute_input":"2023-03-20T21:59:37.545885Z","iopub.status.idle":"2023-03-20T21:59:40.875743Z","shell.execute_reply.started":"2023-03-20T21:59:37.545848Z","shell.execute_reply":"2023-03-20T21:59:40.874480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train tDCS FOG\nThe tDCS FOG (tdcsfog) dataset, comprising data series collected in the lab, as subjects completed a FOG-provoking protocol.","metadata":{}},{"cell_type":"code","source":"df_tdcsfog, df_tdcsfog_id = read_data_series(f\"{train_tdcsfog_folder}/{random.sample(train_tdcsfog, 1)[0]}\")\n\nplot_data_series(df_tdcsfog, df_tdcsfog_id)\n\ndf_tdcsfog.describe(include=\"all\")","metadata":{"execution":{"iopub.status.busy":"2023-03-20T21:59:40.877076Z","iopub.execute_input":"2023-03-20T21:59:40.878222Z","iopub.status.idle":"2023-03-20T21:59:41.391454Z","shell.execute_reply.started":"2023-03-20T21:59:40.878178Z","shell.execute_reply":"2023-03-20T21:59:41.390097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train NOTYPE\nSeries in the notype folder are from the defog dataset but lack event-type annotations.","metadata":{}},{"cell_type":"code","source":"df_notype, df_notype_id = read_data_series(f\"{train_notype_folder}/{random.sample(train_notype, 1)[0]}\", labeled=False)\nplot_data_series(df_notype, df_notype_id)\ndf_notype.describe(include=\"all\")","metadata":{"execution":{"iopub.status.busy":"2023-03-20T21:59:41.394727Z","iopub.execute_input":"2023-03-20T21:59:41.395398Z","iopub.status.idle":"2023-03-20T21:59:47.087449Z","shell.execute_reply.started":"2023-03-20T21:59:41.395340Z","shell.execute_reply":"2023-03-20T21:59:47.086042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test data\n## Test DeFOG","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(10, 2))\n_ = pd.read_csv(f\"{test_defog_folder}/{test_defog[0]}\", index_col=[\"Time\"]).plot(ax=ax)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T21:59:47.089169Z","iopub.execute_input":"2023-03-20T21:59:47.089658Z","iopub.status.idle":"2023-03-20T21:59:52.417145Z","shell.execute_reply.started":"2023-03-20T21:59:47.089617Z","shell.execute_reply":"2023-03-20T21:59:52.415540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(10, 2))\n_ = pd.read_csv(f\"{test_tdcsfog_folder}/{test_tdcsfog[0]}\", index_col=[\"Time\"]).plot(ax=ax)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T21:59:52.418732Z","iopub.execute_input":"2023-03-20T21:59:52.419139Z","iopub.status.idle":"2023-03-20T21:59:52.809050Z","shell.execute_reply.started":"2023-03-20T21:59:52.419098Z","shell.execute_reply":"2023-03-20T21:59:52.807589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf_unlabeled = pd.read_parquet(\n    f\"{unlabeled_folder}/{random.sample(unlabeled, 1)[0]}\"\n).set_index(\"Time\")\n\ndf_unlabeled.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T22:01:52.672173Z","iopub.execute_input":"2023-03-20T22:01:52.672607Z","iopub.status.idle":"2023-03-20T22:02:17.039125Z","shell.execute_reply.started":"2023-03-20T22:01:52.672555Z","shell.execute_reply":"2023-03-20T22:02:17.037751Z"},"trusted":true},"execution_count":null,"outputs":[]}]}