{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from IPython.core.interactiveshell import InteractiveShell\nInteractiveShell.ast_node_interactivity = \"all\"\n","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:43:32.666906Z","iopub.execute_input":"2023-05-26T00:43:32.667262Z","iopub.status.idle":"2023-05-26T00:43:32.674549Z","shell.execute_reply.started":"2023-05-26T00:43:32.667232Z","shell.execute_reply":"2023-05-26T00:43:32.672714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:43:32.676621Z","iopub.execute_input":"2023-05-26T00:43:32.677208Z","iopub.status.idle":"2023-05-26T00:43:32.692687Z","shell.execute_reply.started":"2023-05-26T00:43:32.677175Z","shell.execute_reply":"2023-05-26T00:43:32.691583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-26T00:43:32.693897Z","iopub.execute_input":"2023-05-26T00:43:32.694335Z","iopub.status.idle":"2023-05-26T00:43:32.710023Z","shell.execute_reply.started":"2023-05-26T00:43:32.694310Z","shell.execute_reply":"2023-05-26T00:43:32.708369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction'\ntrain_path = 'train'\ntest_path = 'test'\ndefog_path = 'defog'\ntdcsfog_path = 'tdcsfog'\nsample = pd.read_csv(os.path.join(path,'sample_submission.csv'))\nsample.head()\nsample.info()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:43:32.712056Z","iopub.execute_input":"2023-05-26T00:43:32.712846Z","iopub.status.idle":"2023-05-26T00:43:32.921988Z","shell.execute_reply.started":"2023-05-26T00:43:32.712810Z","shell.execute_reply":"2023-05-26T00:43:32.921074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# subjects = pd.read_csv(os.path.join(path,'subjects.csv'))\n# subjects.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:43:32.923901Z","iopub.execute_input":"2023-05-26T00:43:32.925070Z","iopub.status.idle":"2023-05-26T00:43:32.929175Z","shell.execute_reply.started":"2023-05-26T00:43:32.925032Z","shell.execute_reply":"2023-05-26T00:43:32.928242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# subjects.Visit.unique()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:43:32.930455Z","iopub.execute_input":"2023-05-26T00:43:32.930939Z","iopub.status.idle":"2023-05-26T00:43:32.944102Z","shell.execute_reply.started":"2023-05-26T00:43:32.930911Z","shell.execute_reply":"2023-05-26T00:43:32.942966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tasks = pd.read_csv(os.path.join(path,'tasks.csv'))\n# tasks.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:43:32.945697Z","iopub.execute_input":"2023-05-26T00:43:32.946345Z","iopub.status.idle":"2023-05-26T00:43:32.954159Z","shell.execute_reply.started":"2023-05-26T00:43:32.946317Z","shell.execute_reply":"2023-05-26T00:43:32.953291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tasks = pd.read_csv(os.path.join(path,'events.csv'))\n# tasks.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:43:32.956985Z","iopub.execute_input":"2023-05-26T00:43:32.957226Z","iopub.status.idle":"2023-05-26T00:43:32.966796Z","shell.execute_reply.started":"2023-05-26T00:43:32.957206Z","shell.execute_reply":"2023-05-26T00:43:32.965706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_cs_path = os.path.join(path,train_path,tdcsfog_path)\ntdcsfog_df = [] \n\nfor dirname, _, filenames in os.walk(sample_cs_path):\n    for filename in filenames:\n        df = pd.read_csv(os.path.join(dirname, filename))\n        df['Id']=filename[0:filename.index('.')]\n        tdcsfog_df.append(df)\n\ntdcsfog_df = pd.concat(tdcsfog_df,ignore_index=True)\ntdcsfog_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:43:32.968002Z","iopub.execute_input":"2023-05-26T00:43:32.968271Z","iopub.status.idle":"2023-05-26T00:43:40.228239Z","shell.execute_reply.started":"2023-05-26T00:43:32.968250Z","shell.execute_reply":"2023-05-26T00:43:40.226821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_df.StartHesitation.unique()\ntdcsfog_df.Turn.unique()\ntdcsfog_df.Walking.unique()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:43:40.233329Z","iopub.execute_input":"2023-05-26T00:43:40.233671Z","iopub.status.idle":"2023-05-26T00:43:40.327229Z","shell.execute_reply.started":"2023-05-26T00:43:40.233645Z","shell.execute_reply":"2023-05-26T00:43:40.326018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_cs_path = os.path.join(path,train_path,defog_path)\ndefog_df = [] \n\nfor dirname, _, filenames in os.walk(sample_cs_path):\n    for filename in filenames:\n        df = pd.read_csv(os.path.join(dirname, filename))\n        df['Id']=filename[0:filename.index('.')]\n        defog_df.append(df)\n\ndefog_df = pd.concat(defog_df,ignore_index=True)\ndefog_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:43:40.328856Z","iopub.execute_input":"2023-05-26T00:43:40.329165Z","iopub.status.idle":"2023-05-26T00:43:48.937222Z","shell.execute_reply.started":"2023-05-26T00:43:40.329139Z","shell.execute_reply":"2023-05-26T00:43:48.936161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = ['Time','AccV','AccML','AccAP']\ntargets = ['StartHesitation','Turn','Walking']\naccmeasurs = ['AccV','AccML','AccAP']\ndefog_df.Valid.unique()\ndefog_df.Task.unique()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:43:48.938523Z","iopub.execute_input":"2023-05-26T00:43:48.938813Z","iopub.status.idle":"2023-05-26T00:43:49.032887Z","shell.execute_reply.started":"2023-05-26T00:43:48.938785Z","shell.execute_reply":"2023-05-26T00:43:49.031623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_df = defog_df.query('Valid==True and Task==True')\ndefog_df = defog_df.drop(['Valid','Task'],axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:43:49.034428Z","iopub.execute_input":"2023-05-26T00:43:49.034751Z","iopub.status.idle":"2023-05-26T00:43:49.486043Z","shell.execute_reply.started":"2023-05-26T00:43:49.034728Z","shell.execute_reply":"2023-05-26T00:43:49.485151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_train_data = pd.concat([tdcsfog_df,defog_df])\nall_train_data = all_train_data.astype({'Time':'int32','Turn':'int8','Walking':'int8',\\\n                                        'StartHesitation':'int8','AccV':'float16',\\\n                                        'AccML':'float16','AccAP':'float16'})\ndefog_df = None\ntdcsfog_df = None\nall_train_data.info()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:43:49.487045Z","iopub.execute_input":"2023-05-26T00:43:49.487792Z","iopub.status.idle":"2023-05-26T00:43:49.969650Z","shell.execute_reply.started":"2023-05-26T00:43:49.487746Z","shell.execute_reply":"2023-05-26T00:43:49.968816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_train_data.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:43:49.970784Z","iopub.execute_input":"2023-05-26T00:43:49.971476Z","iopub.status.idle":"2023-05-26T00:43:52.978436Z","shell.execute_reply.started":"2023-05-26T00:43:49.971447Z","shell.execute_reply":"2023-05-26T00:43:52.976853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nfrom scipy import signal ","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:43:52.979972Z","iopub.execute_input":"2023-05-26T00:43:52.980254Z","iopub.status.idle":"2023-05-26T00:43:52.984541Z","shell.execute_reply.started":"2023-05-26T00:43:52.980232Z","shell.execute_reply":"2023-05-26T00:43:52.983715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfor id,group in all_train_data.groupby('Id')[features+targets]:\n    sns.lineplot(x= 'Time',y='AccV',hue='StartHesitation',data=group.iloc[0:4000])\n    plt.show()\n    sns.lineplot(x= 'Time',y='AccV',hue='Turn',data=group.iloc[0:4000])\n    plt.show()\n    sns.lineplot(x= 'Time',y='AccV',hue='Walking',data=group.iloc[0:4000])\n    plt.show()\n    break","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:43:52.985636Z","iopub.execute_input":"2023-05-26T00:43:52.986359Z","iopub.status.idle":"2023-05-26T00:43:55.600299Z","shell.execute_reply.started":"2023-05-26T00:43:52.986314Z","shell.execute_reply":"2023-05-26T00:43:55.599167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{}},{"cell_type":"code","source":"def addPeakMetrics(df,colname):\n    \n    peakname = colname+'_peak'\n    prominancename = colname+'_peakprominance'\n    widthname = colname+'_peakwidth'\n    df[peakname] = 0\n    df[prominancename] = 0\n    df[widthname]=0\n    peaks,_ = signal.find_peaks(df[colname])\n    df.iloc[peaks,list(df.columns).index(peakname)] = 1\n    peakprom = signal.peak_prominences(df[colname],peaks)[0]\n    peakwidth = signal.peak_widths(df[colname],peaks)[0]\n    df.iloc[peaks,list(df.columns).index(prominancename)] = peakprom\n    df.iloc[peaks,list(df.columns).index(widthname)] = peakwidth\n\n    return df,peaks","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:43:55.601776Z","iopub.execute_input":"2023-05-26T00:43:55.602041Z","iopub.status.idle":"2023-05-26T00:43:55.609801Z","shell.execute_reply.started":"2023-05-26T00:43:55.602020Z","shell.execute_reply":"2023-05-26T00:43:55.608329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def featureEngineer(data):\n    \n    avgheight_features = ['AccV_avg_height', 'AccML_avg_height','AccAP_avg_height']\n    maxheight_features = ['AccV_max_height','AccML_max_height','AccAP_max_height']\n    minheight_features = ['AccV_min_height','AccML_min_height','AccAP_min_height']\n    std_features = ['AccV_std','AccML_std','AccAP_std']\n    first = None\n    engineered_dt = []\n    for cid,group in data:\n        df = group.set_index('Time')\n        df,accvpeaks = addPeakMetrics(df,'AccV')\n        accvfreq = round(np.diff(accvpeaks).mean())\n        df,accmlpeaks = addPeakMetrics(df,'AccML')\n        accmlfreq = round(np.diff(accmlpeaks).mean())\n        df,accappeaks = addPeakMetrics(df,'AccAP')\n        accapfreq = round(np.diff(accappeaks).mean())\n\n        window = max(accvfreq,accmlfreq,accapfreq)  # the min wavelength among the patient signals\n        \n        df[avgheight_features] = df[accmeasurs].rolling(window).mean()\n        df[avgheight_features] = df[avgheight_features].fillna(df.iloc[window,[list(df.columns).index(i) for i in df.columns if i in avgheight_features]])\n        df[avgheight_features] = df[avgheight_features].astype('float16')\n        df[maxheight_features] = df[accmeasurs].rolling(window).max()\n        df[maxheight_features] = df[maxheight_features].fillna(df.iloc[window,[list(df.columns).index(i) for i in df.columns if i in maxheight_features]])\n        df[maxheight_features] = df[maxheight_features].astype('float16')\n        df[minheight_features] = df[accmeasurs].rolling(window).min()\n        df[minheight_features] = df[minheight_features].fillna(df.iloc[window,[list(df.columns).index(i) for i in df.columns if i in minheight_features]])\n        df[minheight_features] = df[minheight_features].astype('float16')\n        df[std_features] = df[avgheight_features].rolling(window).std()\n        df[std_features] = df[std_features].fillna(df.iloc[window,[list(df.columns).index(i) for i in df.columns if i in std_features]])\n        df[std_features] = df[std_features].astype('float16')\n#         if first is None:  # plot the first iteration \n#             sns.scatterplot(x= 'Time',y='AccML',hue='StartHesitation',data=group[0:500])\n#             #sns.lineplot(df[maxheight_features])\n           \n#             plt.show()\n#             sns.lineplot(x= 'Time',y='AccML',hue='Turn',data=group)\n#             #sns.lineplot(df[maxheight_features])\n#             plt.show()\n#             sns.lineplot(x= 'Time',y='AccML',hue='Walking',data=group)\n#             #sns.lineplot(df[maxheight_features])\n#             plt.show()\n#             first = cid\n        df['Id']=cid\n        df = df.reset_index(names='Time')\n        engineered_dt.append(df)\n    engineered_dt = pd.concat(engineered_dt,ignore_index=True)\n    return engineered_dt\n","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:43:55.611792Z","iopub.execute_input":"2023-05-26T00:43:55.612803Z","iopub.status.idle":"2023-05-26T00:43:55.629216Z","shell.execute_reply.started":"2023-05-26T00:43:55.612740Z","shell.execute_reply":"2023-05-26T00:43:55.628494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def createSignals(data):\n#     samples = len(data)\n#     widths = [len(group) for cid,group in data]\n#     print(f'there are: {samples} visit data and the durations for them are: \\n{widths} ')\n        \n    ","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:43:55.630409Z","iopub.execute_input":"2023-05-26T00:43:55.631795Z","iopub.status.idle":"2023-05-26T00:43:55.647516Z","shell.execute_reply.started":"2023-05-26T00:43:55.631739Z","shell.execute_reply":"2023-05-26T00:43:55.646703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#createSignals(all_train_data.groupby('Id'))","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:43:55.648527Z","iopub.execute_input":"2023-05-26T00:43:55.648842Z","iopub.status.idle":"2023-05-26T00:43:55.662565Z","shell.execute_reply.started":"2023-05-26T00:43:55.648817Z","shell.execute_reply":"2023-05-26T00:43:55.661242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_train_data = featureEngineer(all_train_data.groupby('Id')[features+targets])\nall_train_data.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:43:55.663832Z","iopub.execute_input":"2023-05-26T00:43:55.664134Z","iopub.status.idle":"2023-05-26T00:44:35.200612Z","shell.execute_reply.started":"2023-05-26T00:43:55.664108Z","shell.execute_reply":"2023-05-26T00:44:35.199421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_train_data.corr()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:44:35.201850Z","iopub.execute_input":"2023-05-26T00:44:35.202121Z","iopub.status.idle":"2023-05-26T00:44:52.275943Z","shell.execute_reply.started":"2023-05-26T00:44:35.202098Z","shell.execute_reply":"2023-05-26T00:44:52.274458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_metadata = pd.read_csv(os.path.join(path,'tdcsfog_metadata.csv'))\ntdcsfog_metadata.Id.nunique()\ntdcsfog_metadata.Subject.nunique()\ntdcsfog_metadata.count()\ntdcsfog_metadata.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:44:52.277584Z","iopub.execute_input":"2023-05-26T00:44:52.277973Z","iopub.status.idle":"2023-05-26T00:44:52.309249Z","shell.execute_reply.started":"2023-05-26T00:44:52.277952Z","shell.execute_reply":"2023-05-26T00:44:52.308160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_test_dir = os.path.join(path,test_path,defog_path)\ntdcsfog_test_dir = os.path.join(path,test_path,tdcsfog_path)\ndefog_tst_df = []\ntdcsfog_tst_df = []\nfor dirname, _, filenames in os.walk(defog_test_dir):\n    for filename in filenames:\n        df= pd.read_csv(os.path.join(dirname,filename))\n        df['Id'] = filename[0:filename.index('.')]\n        defog_tst_df.append(df.copy())\nfor dirname, _, filenames in os.walk(tdcsfog_test_dir):\n    for filename in filenames:\n        df= pd.read_csv(os.path.join(dirname,filename))\n        df['Id'] = filename[0:filename.index('.')]\n        tdcsfog_tst_df.append(df.copy())\n\ndefog_tst_df = pd.concat(defog_tst_df)\ntdcsfog_tst_df = pd.concat(tdcsfog_tst_df)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:44:52.310446Z","iopub.execute_input":"2023-05-26T00:44:52.310895Z","iopub.status.idle":"2023-05-26T00:44:52.692943Z","shell.execute_reply.started":"2023-05-26T00:44:52.310858Z","shell.execute_reply":"2023-05-26T00:44:52.691286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_tst_df.head()\ntdcsfog_tst_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:44:52.694273Z","iopub.execute_input":"2023-05-26T00:44:52.694677Z","iopub.status.idle":"2023-05-26T00:44:52.718894Z","shell.execute_reply.started":"2023-05-26T00:44:52.694650Z","shell.execute_reply":"2023-05-26T00:44:52.716681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tst_df = pd.concat([tdcsfog_tst_df,defog_tst_df],ignore_index=True)\ntst_df = featureEngineer(tst_df.groupby('Id')[features])\ntdcsfog_tst_df = None\ndefog_tst_df = None\ntst_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:44:52.720400Z","iopub.execute_input":"2023-05-26T00:44:52.720789Z","iopub.status.idle":"2023-05-26T00:44:53.261102Z","shell.execute_reply.started":"2023-05-26T00:44:52.720726Z","shell.execute_reply":"2023-05-26T00:44:53.259673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count = all_train_data.groupby(['StartHesitation','Turn','Walking'])['Time'].agg('count')\nsampleCount = count.min()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:44:53.265998Z","iopub.execute_input":"2023-05-26T00:44:53.266349Z","iopub.status.idle":"2023-05-26T00:44:53.740226Z","shell.execute_reply.started":"2023-05-26T00:44:53.266311Z","shell.execute_reply":"2023-05-26T00:44:53.738482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"noGait = all_train_data.query('StartHesitation==0 and Walking==0 and Turn==0').sample(sampleCount)\nhesit = all_train_data.query('StartHesitation==1').sample(sampleCount)\nturn = all_train_data.query('Turn==1').sample(sampleCount)\nwalk = all_train_data.query('Walking==1').sample(sampleCount)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:44:53.742054Z","iopub.execute_input":"2023-05-26T00:44:53.742426Z","iopub.status.idle":"2023-05-26T00:44:56.956346Z","shell.execute_reply.started":"2023-05-26T00:44:53.742396Z","shell.execute_reply":"2023-05-26T00:44:56.955080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_train_data = pd.concat([noGait,hesit,turn,walk]).reset_index(drop=True)\nnoGait = None\nhesit = None\nturn = None\nwalk = None\nall_train_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:44:56.957853Z","iopub.execute_input":"2023-05-26T00:44:56.958259Z","iopub.status.idle":"2023-05-26T00:44:57.084000Z","shell.execute_reply.started":"2023-05-26T00:44:56.958227Z","shell.execute_reply":"2023-05-26T00:44:57.083061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fig,axes = plt.subplots(nrows=4,ncols=1)\n\n# sns.histplot(noGait['AccML'],ax=axes[0]);\n# sns.histplot(hesit['AccML'],ax=axes[1]);\n# sns.histplot(turn['AccML'],ax=axes[2]);\n# sns.histplot(walk['AccML'],ax=axes[3]);","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:44:57.085293Z","iopub.execute_input":"2023-05-26T00:44:57.086165Z","iopub.status.idle":"2023-05-26T00:44:57.090495Z","shell.execute_reply.started":"2023-05-26T00:44:57.086132Z","shell.execute_reply":"2023-05-26T00:44:57.089421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sns.histplot(x='AccV',hue='Turn',data=all_train_data)\n# plt.show()\n# sns.histplot(x='AccML',hue='Turn',data=all_train_data)\n# plt.show()\n# sns.histplot(x='AccAP',hue='Turn',data=all_train_data)\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:44:57.091884Z","iopub.execute_input":"2023-05-26T00:44:57.093081Z","iopub.status.idle":"2023-05-26T00:44:57.105959Z","shell.execute_reply.started":"2023-05-26T00:44:57.093033Z","shell.execute_reply":"2023-05-26T00:44:57.104839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## create multi class column","metadata":{}},{"cell_type":"code","source":"# model_classes = 'classes'\n# class_toindex = {1:0,2:1,3:2}\n\n# all_train_data[model_classes]=0\n# all_train_data.loc[all_train_data.StartHesitation==1,model_classes]=1\n# all_train_data.loc[all_train_data.Turn==1,model_classes]=2\n# all_train_data.loc[all_train_data.Walking==1,model_classes]=3\n# all_train_data[model_classes]= all_train_data[model_classes].astype('int8')\n# all_train_data.drop(columns=targets,index=1,inplace=True)\n# all_train_data.head()\nall_features = [feature for feature in all_train_data.columns if feature !='Id' and feature not in targets and feature != 'Time']\nall_features","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:44:57.107135Z","iopub.execute_input":"2023-05-26T00:44:57.107377Z","iopub.status.idle":"2023-05-26T00:44:57.123326Z","shell.execute_reply.started":"2023-05-26T00:44:57.107356Z","shell.execute_reply":"2023-05-26T00:44:57.122067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\nscalar = MinMaxScaler()\nscalar.fit(all_train_data[all_features].values)\nall_train_data[all_features] = scalar.transform(all_train_data[all_features].values)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:44:57.124602Z","iopub.execute_input":"2023-05-26T00:44:57.124905Z","iopub.status.idle":"2023-05-26T00:44:57.539489Z","shell.execute_reply.started":"2023-05-26T00:44:57.124883Z","shell.execute_reply":"2023-05-26T00:44:57.537719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_train_data.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:44:57.541107Z","iopub.execute_input":"2023-05-26T00:44:57.541441Z","iopub.status.idle":"2023-05-26T00:44:58.580881Z","shell.execute_reply.started":"2023-05-26T00:44:57.541403Z","shell.execute_reply":"2023-05-26T00:44:58.580086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflow-io","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:44:58.582189Z","iopub.execute_input":"2023-05-26T00:44:58.582455Z","iopub.status.idle":"2023-05-26T00:45:09.089269Z","shell.execute_reply.started":"2023-05-26T00:44:58.582424Z","shell.execute_reply":"2023-05-26T00:45:09.087834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras import Sequential\nfrom keras.layers import Dense, Dropout,Conv1D,Flatten,LSTM\nfrom keras.optimizers import Adam,SGD\nfrom keras.losses import CategoricalCrossentropy\nfrom keras.metrics import CategoricalAccuracy,Precision,SparseCategoricalCrossentropy\nimport tensorflow as tf\ntf.config.run_functions_eagerly(True)\nlookback = 3\nlosses = ['categorical_crossentropy']\nmetrics = [CategoricalAccuracy(),Precision()]\nmodel = Sequential(name='Prediction_Gait')\nmodel.add(LSTM(80,input_shape=(lookback,len(all_features),),return_sequences=True))\nmodel.add(LSTM(128,activation='relu'))\n\nmodel.add(Dense(80,activation='relu'))\nmodel.add(Dense(64,activation='relu'))\nmodel.add(Dense(32,activation='relu'))\n\nmodel.add(Dense(10,activation='sigmoid'))\n\nmodel.add(Dense(3,activation='softmax'))\nmodel.compile(optimizer=Adam(learning_rate=0.01),loss=losses,metrics=metrics)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:45:09.090909Z","iopub.execute_input":"2023-05-26T00:45:09.091226Z","iopub.status.idle":"2023-05-26T00:45:09.455743Z","shell.execute_reply.started":"2023-05-26T00:45:09.091200Z","shell.execute_reply":"2023-05-26T00:45:09.454420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:45:09.457148Z","iopub.execute_input":"2023-05-26T00:45:09.457418Z","iopub.status.idle":"2023-05-26T00:45:09.485281Z","shell.execute_reply.started":"2023-05-26T00:45:09.457392Z","shell.execute_reply":"2023-05-26T00:45:09.484290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nbatch = 5000\nepochs = 50\nprint(len(all_train_data.groupby('Id')))\nfor Id,group in all_train_data.groupby('Id'):\n    df = group.set_index('Time')\n    X = np.hstack([df[all_features].values[0:-2],\n                   df.iloc[1:][all_features].values[0:-1],\n                   df.iloc[2:][all_features].values])\n    X = np.reshape(X,(-1,lookback,len(all_features)))\n    Y = df[targets].values\n    \n    model.fit(X,Y,batch_size=batch,epochs=epochs,verbose=2,workers=32,validation_split=.2)\n    \n","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:45:09.486684Z","iopub.execute_input":"2023-05-26T00:45:09.487342Z","iopub.status.idle":"2023-05-26T00:51:12.091288Z","shell.execute_reply.started":"2023-05-26T00:45:09.487311Z","shell.execute_reply":"2023-05-26T00:51:12.088631Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model = RandomForestClassifier(verbose=1,n_jobs=8,random_state=69)\n# params = {'n_estimators':[30,50,70,100],'max_depth':[4,5,7,9]}\n# gsRF = GridSearchCV(model,params)\n# gsRF.fit(X_tr.values,Y_tr.values)\n# output_train = gsRF.predict(X_tr.values)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:51:12.093546Z","iopub.status.idle":"2023-05-26T00:51:12.094182Z","shell.execute_reply.started":"2023-05-26T00:51:12.093910Z","shell.execute_reply":"2023-05-26T00:51:12.093932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Evaluation \n# out_eval = model.predict(X_tst.values)\n# eval_precision = metrics.precision_score(Y_tst,out_eval,average='weighted')\n# eval_accuracy = metrics.accuracy_score(Y_tst,out_eval)\n# eval_confmat = metrics.confusion_matrix(Y_tst,out_eval)\n# print(f'the evaluation precision score is: {eval_precision}')\n# print(f'the evaluation accuracy score is: {eval_accuracy}')\n# print(f'the evaluation confusion matrix is : {eval_confmat}')","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:51:12.096011Z","iopub.status.idle":"2023-05-26T00:51:12.096333Z","shell.execute_reply.started":"2023-05-26T00:51:12.096192Z","shell.execute_reply":"2023-05-26T00:51:12.096206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntst_df[all_features] = scalar.transform(tst_df[all_features])\nresult = []\nfor Id,gr in tst_df.groupby('Id'):\n    gr= gr.set_index('Time')\n    df2 = gr.copy()\n    df2.iloc[0:-1] = gr.iloc[1:].copy()\n    df3 = gr.copy()\n    df3.iloc[0:-2] = gr.iloc[2:].copy()\n    Xt = np.hstack([gr[all_features].values,df2[all_features].values,df3[all_features].values]) \n    Xt = np.reshape(Xt,(-1,lookback,len(all_features)))\n    output_tst = model.predict(Xt).round()\n    result_df = pd.DataFrame(data=output_tst,columns=['StartHesitation','Turn','Walking'])\n    result_df['Id'] = Id +'_' +gr.index.to_series().apply(str)\n    result.append(result_df)\ndf = pd.concat(result,ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:57:43.460629Z","iopub.execute_input":"2023-05-26T00:57:43.461021Z","iopub.status.idle":"2023-05-26T01:01:31.661487Z","shell.execute_reply.started":"2023-05-26T00:57:43.460993Z","shell.execute_reply":"2023-05-26T01:01:31.659656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission_array = np.round(output_tst)\n# output_tst = None","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:51:12.100671Z","iopub.status.idle":"2023-05-26T00:51:12.101343Z","shell.execute_reply.started":"2023-05-26T00:51:12.101053Z","shell.execute_reply":"2023-05-26T00:51:12.101080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission_array.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:51:12.103019Z","iopub.status.idle":"2023-05-26T00:51:12.103508Z","shell.execute_reply.started":"2023-05-26T00:51:12.103264Z","shell.execute_reply":"2023-05-26T00:51:12.103287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df = pd.DataFrame(data = submission_array,columns=targets).astype('int8')\n# df['Id']=tst_df['Id']\ndf = df[['Id','StartHesitation','Turn','Walking']]\ndf[targets]=df[targets].astype('int8')\ndf.to_csv('/kaggle/working/submission.csv',index=False)\ndf.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T01:03:16.003309Z","iopub.execute_input":"2023-05-26T01:03:16.004270Z","iopub.status.idle":"2023-05-26T01:03:16.811549Z","shell.execute_reply.started":"2023-05-26T01:03:16.004220Z","shell.execute_reply":"2023-05-26T01:03:16.810273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T01:03:19.671904Z","iopub.execute_input":"2023-05-26T01:03:19.672298Z","iopub.status.idle":"2023-05-26T01:03:19.702318Z","shell.execute_reply.started":"2023-05-26T01:03:19.672269Z","shell.execute_reply":"2023-05-26T01:03:19.700978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-05-26T00:51:12.109523Z","iopub.status.idle":"2023-05-26T00:51:12.109975Z","shell.execute_reply.started":"2023-05-26T00:51:12.109797Z","shell.execute_reply":"2023-05-26T00:51:12.109814Z"},"trusted":true},"execution_count":null,"outputs":[]}]}