{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport glob\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-10T22:22:46.297649Z","iopub.execute_input":"2023-03-10T22:22:46.298099Z","iopub.status.idle":"2023-03-10T22:22:46.916116Z","shell.execute_reply.started":"2023-03-10T22:22:46.298057Z","shell.execute_reply":"2023-03-10T22:22:46.914472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_DIR = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction\"","metadata":{"execution":{"iopub.status.busy":"2023-03-10T22:22:47.723886Z","iopub.execute_input":"2023-03-10T22:22:47.724288Z","iopub.status.idle":"2023-03-10T22:22:47.730428Z","shell.execute_reply.started":"2023-03-10T22:22:47.724253Z","shell.execute_reply":"2023-03-10T22:22:47.729179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_train_files = list(glob.glob(os.path.join(BASE_DIR, \"train\", \"defog\", \"*.csv\")))\ntdcs_train_files = list(glob.glob(os.path.join(BASE_DIR, \"train\", \"tdcsfog\", \"*.csv\")))\nlen(defog_train_files), len(tdcs_train_files)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T22:22:48.605547Z","iopub.execute_input":"2023-03-10T22:22:48.606136Z","iopub.status.idle":"2023-03-10T22:22:48.960270Z","shell.execute_reply.started":"2023-03-10T22:22:48.606098Z","shell.execute_reply":"2023-03-10T22:22:48.958786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_labelled = []\nfor i, train_file in enumerate(tqdm(tdcs_train_files + defog_train_files)):\n    source = \"defog\" if \"defog\" in train_file else \"tdcs\"\n    train_df = pd.read_csv(train_file)\n    \n    # task, valid are just in defog? why? I assume in tdcs all records should be taken into account\n    if \"Task\" in train_df.columns and \"Valid\" in train_df.columns:\n        train_df = train_df[(train_df[\"Task\"]) & (train_df[\"Valid\"])]\n        \n    train_df['case_id'] = f'{source}_train_{i}'\n    train_df['source'] = source\n    all_labelled.append(train_df)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T22:24:28.803712Z","iopub.execute_input":"2023-03-10T22:24:28.804138Z","iopub.status.idle":"2023-03-10T22:25:11.990136Z","shell.execute_reply.started":"2023-03-10T22:24:28.804101Z","shell.execute_reply":"2023-03-10T22:25:11.988607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labelled_df = pd.concat(all_labelled)\nlabelled_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-10T22:34:01.174092Z","iopub.execute_input":"2023-03-10T22:34:01.174532Z","iopub.status.idle":"2023-03-10T22:34:02.655194Z","shell.execute_reply.started":"2023-03-10T22:34:01.174493Z","shell.execute_reply":"2023-03-10T22:34:02.653862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labelled_df.columns","metadata":{"execution":{"iopub.status.busy":"2023-03-10T22:34:05.129767Z","iopub.execute_input":"2023-03-10T22:34:05.130171Z","iopub.status.idle":"2023-03-10T22:34:05.138517Z","shell.execute_reply.started":"2023-03-10T22:34:05.130131Z","shell.execute_reply":"2023-03-10T22:34:05.136925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Accelerometer readings are a bit different between no event and a specific event happening","metadata":{}},{"cell_type":"code","source":"SAMPLE = 30000\nfor transform in [lambda x:x, np.abs]:\n    for e in [\"StartHesitation\", \"Turn\", \"Walking\"]:\n        any_positive_mask = np.any(labelled_df[[e]], axis=1)\n        any_positive = labelled_df[any_positive_mask]\n\n        non_positive = labelled_df[~any_positive_mask]\n        neg_fraction = len(any_positive) / len(non_positive)\n        non_positive_sample = non_positive.sample(frac = neg_fraction)\n        balanced = pd.concat([any_positive, non_positive_sample])\n        if len(balanced) > SAMPLE:\n            balanced = balanced.sample(n=SAMPLE)\n        \n        fig = plt.figure(layout=\"constrained\", figsize=(14, 6))\n        \n        ax_dict = fig.subplot_mosaic( [\n            [\"AccV\", \"AccML\", \"AccAP\"]\n        ])\n        func = \"abs(inputs)\" if transform == np.abs else \"inputs\" \n        plt.title(f\"distributions of {func} w.r.t. {e}\")\n        \n        for k, ax in ax_dict.items():    \n            balanced[k] = transform(balanced[k])\n            sns.histplot(balanced, x=k, hue=e, ax=ax)\n            \n\n\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-10T22:34:24.617916Z","iopub.execute_input":"2023-03-10T22:34:24.618334Z","iopub.status.idle":"2023-03-10T22:34:57.901232Z","shell.execute_reply.started":"2023-03-10T22:34:24.618264Z","shell.execute_reply":"2023-03-10T22:34:57.899860Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cases_events = labelled_df.groupby([\"case_id\"]).sum()\ncases_events","metadata":{"execution":{"iopub.status.busy":"2023-03-10T22:35:06.638267Z","iopub.execute_input":"2023-03-10T22:35:06.638727Z","iopub.status.idle":"2023-03-10T22:35:08.316849Z","shell.execute_reply.started":"2023-03-10T22:35:06.638674Z","shell.execute_reply":"2023-03-10T22:35:08.314256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Distribution of the number of each class per case","metadata":{}},{"cell_type":"code","source":"cases_events = labelled_df.groupby([\"case_id\"]).sum()\nfor k in [\"StartHesitation\", \"Turn\", \"Walking\"]:\n    sns.displot(cases_events, x=k)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-10T22:35:12.509523Z","iopub.execute_input":"2023-03-10T22:35:12.509930Z","iopub.status.idle":"2023-03-10T22:35:15.258019Z","shell.execute_reply.started":"2023-03-10T22:35:12.509894Z","shell.execute_reply":"2023-03-10T22:35:15.256394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## For each class - take a case that has the most datapoints with this class and plot accelerometer values","metadata":{}},{"cell_type":"code","source":"import plotly.express as px\nimport plotly.graph_objs as go\n\n#Code by Anmorgul https://www.kaggle.com/anmorgul/strange-pattern-cottonwood-willow\n#https://www.kaggle.com/code/mpwolke/roosevelt-forest-of-northern-colorado-charts\n#https://www.kaggle.com/code/kretes/freezing-of-gait-fog/edit\n\n# all_labelled\nfor e in [\"StartHesitation\", \"Turn\", \"Walking\"]:\n    case_id_the_most = cases_events.sort_values(e, ascending=False).index[0]\n    df = labelled_df[labelled_df[\"case_id\"] == case_id_the_most]\n    fig = px.scatter_3d(df, x='AccV', y='AccML', z='AccAP',\n                  color=e, size_max=8, width=1000, height=800, opacity=0.9, template=\"plotly_dark\",\n                        title=f'Acceleration in units of g, from a lower-back sensor on three axes, colored by {e}, for {case_id_the_most}')\n    fig.update_layout(\n        font_size=8,\n        legend_font_size=16,)\n\n\n    fig.show()     ","metadata":{"execution":{"iopub.status.busy":"2023-03-10T22:35:25.628474Z","iopub.execute_input":"2023-03-10T22:35:25.628879Z","iopub.status.idle":"2023-03-10T22:35:32.790895Z","shell.execute_reply.started":"2023-03-10T22:35:25.628845Z","shell.execute_reply":"2023-03-10T22:35:32.790133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}