{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Investigation of Feature Engineering (for tdcsfog dataset)🤔\n--- \n[Objective]\n- Show time series data using plotly and investigate which feature is 　valide for ML.","metadata":{}},{"cell_type":"code","source":"# # Install tsflex and seglearn\n!pip install tsflex --no-index --find-links=file:///kaggle/input/time-series-tools\n!pip install seglearn --no-index --find-links=file:///kaggle/input/time-series-tools","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":20.008275,"end_time":"2023-05-21T02:12:36.148121","exception":false,"start_time":"2023-05-21T02:12:16.139846","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-27T15:43:03.770941Z","iopub.execute_input":"2023-05-27T15:43:03.771440Z","iopub.status.idle":"2023-05-27T15:43:30.397023Z","shell.execute_reply.started":"2023-05-27T15:43:03.771396Z","shell.execute_reply":"2023-05-27T15:43:30.395430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# =========================\n# Config\n# =========================\nidx = 798 # <- 0 -833のどれかを選択\ncol = \"AccAP\" # \"AccAP\"\n\n# --------------------------------------------------------------------------------\n# idx of Best5 of StartHesitaiton\n# --------------------------------------------------------------------------------\n# 44, 78, 241, 448, 798, \n# --------------------------------------------------------------------------------\n# idx of Best5 of Turn\n# --------------------------------------------------------------------------------\n# 61, 78, 246, 326, 509, \n# --------------------------------------------------------------------------------\n# idx of Best5 of Walking\n# --------------------------------------------------------------------------------\n# 242, 409, 561, 586, 686, ","metadata":{"execution":{"iopub.status.busy":"2023-05-27T15:43:30.400245Z","iopub.execute_input":"2023-05-27T15:43:30.400675Z","iopub.status.idle":"2023-05-27T15:43:30.407092Z","shell.execute_reply.started":"2023-05-27T15:43:30.400634Z","shell.execute_reply":"2023-05-27T15:43:30.405872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# =========================\n# Import libraries\n# =========================\n# default\nimport gc, os, glob, random\nfrom os import path\nfrom pathlib import Path\n# make data\nimport polars as pl\nimport pandas as pd\npd.set_option('display.max_columns', None); # pd.set_option('display.max_rows', None)\nimport numpy as np\nfrom tqdm.auto import tqdm\nimport ydata_profiling as pdp\n","metadata":{"papermill":{"duration":4.670362,"end_time":"2023-05-21T02:12:40.825689","exception":false,"start_time":"2023-05-21T02:12:36.155327","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-27T15:43:30.409180Z","iopub.execute_input":"2023-05-27T15:43:30.409547Z","iopub.status.idle":"2023-05-27T15:43:30.425732Z","shell.execute_reply.started":"2023-05-27T15:43:30.409517Z","shell.execute_reply":"2023-05-27T15:43:30.424295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_df(df, num=3, tail=True):\n    print(df.shape)\n    display(df.head(num))\n    if tail:\n        display(df.tail(num))","metadata":{"papermill":{"duration":0.014807,"end_time":"2023-05-21T02:12:40.845896","exception":false,"start_time":"2023-05-21T02:12:40.831089","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-27T15:43:30.428573Z","iopub.execute_input":"2023-05-27T15:43:30.428930Z","iopub.status.idle":"2023-05-27T15:43:30.441360Z","shell.execute_reply.started":"2023-05-27T15:43:30.428899Z","shell.execute_reply":"2023-05-27T15:43:30.440089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_path = glob.glob(\"../input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/*.csv\")\ntdcsfog_path = glob.glob(\"../input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/*.csv\")\nnotype_path = glob.glob(\"../input/tlvmc-parkinsons-freezing-gait-prediction/train/notype/*.csv\")\n\nprint(f\"length of defog_path   : {len(defog_path)}\")\nprint(f\"length of tdcsfog_path : {len(tdcsfog_path)}\")\nprint(f\"length of notype_path  : {len(notype_path)}\")","metadata":{"papermill":{"duration":0.091537,"end_time":"2023-05-21T02:12:40.942951","exception":false,"start_time":"2023-05-21T02:12:40.851414","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-27T15:43:30.443520Z","iopub.execute_input":"2023-05-27T15:43:30.443986Z","iopub.status.idle":"2023-05-27T15:43:30.466935Z","shell.execute_reply.started":"2023-05-27T15:43:30.443934Z","shell.execute_reply":"2023-05-27T15:43:30.465811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"st_path_list = ['c23e55e654','b19eb6e513','42ec812606','714dc454eb','9a7a5e1df2']\ntu_path_list = ['9a7a5e1df2', 'bfe8fee128', 'be8fdfa712', 'a68f95a17d', 'be2740b333']\nwa_path_list = ['bbe9ac572f', 'f66b28cfc4', 'a512e016a6', 'cd7a1efefb', '9ee7d24df1']\n\nprint(\"-\"*80);print('idx of Best5 of StartHesitaiton');print('-'*80)\nfor i, _path in enumerate(tdcsfog_path):\n    if os.path.basename(_path).split(\".cs\")[0] in st_path_list:\n        print(i, end=\", \")\nprint()\nprint(\"-\"*80);print('idx of Best5 of Turn');print('-'*80)\nfor i, _path in enumerate(tdcsfog_path):\n    if os.path.basename(_path).split(\".cs\")[0] in tu_path_list:\n        print(i, end=\", \")\nprint()        \nprint(\"-\"*80);print('idx of Best5 of Walking');print('-'*80)\nfor i, _path in enumerate(tdcsfog_path):\n    if os.path.basename(_path).split(\".cs\")[0] in wa_path_list:\n        print(i, end=\", \")","metadata":{"execution":{"iopub.status.busy":"2023-05-27T15:43:30.468220Z","iopub.execute_input":"2023-05-27T15:43:30.469102Z","iopub.status.idle":"2023-05-27T15:43:30.487509Z","shell.execute_reply.started":"2023-05-27T15:43:30.469070Z","shell.execute_reply":"2023-05-27T15:43:30.486309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ====================================\n# load defog_data\n# ====================================\ndf = pd.read_csv(tdcsfog_path[idx])\nshow_df(df)\n\n# データフレームが長い場合、グラフの表示する開始インデックスとインデックス長を指定\nst = 0 ; length = len(df)\nif len(df) >= 150000:\n    st = 0 ; length = 150000","metadata":{"papermill":{"duration":0.081247,"end_time":"2023-05-21T02:12:41.029695","exception":false,"start_time":"2023-05-21T02:12:40.948448","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-27T15:43:30.488759Z","iopub.execute_input":"2023-05-27T15:43:30.489797Z","iopub.status.idle":"2023-05-27T15:43:30.567765Z","shell.execute_reply.started":"2023-05-27T15:43:30.489761Z","shell.execute_reply":"2023-05-27T15:43:30.566646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# 可視化して確認 (Plotly)\n# ==============================\n# plotly \nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nimport plotly.express as px\n\ndef show_tdcsfog_plotly(df, xcol, ycol, flags, st, length):\n    tmp_df = df.copy()\n    if (st !=None) | (length != None):\n        tmp_df = tmp_df[st: st + length].copy()\n\n    row_length = len(ycol)\n    flag_length = len(flags)\n\n    fig = make_subplots(\n        rows=row_length+1, cols=1,\n        shared_xaxes=True\n        )\n\n    for i in range(row_length):\n        fig.add_trace(go.Scatter(\n            x=tmp_df[xcol], y=tmp_df[ycol[i]], \n            name=ycol[i], \n            mode=\"lines\",\n            ), row=i+1, col=1)\n        \n    for j in range(flag_length):\n        fig.add_trace(go.Scatter(\n            x=tmp_df[xcol], y=tmp_df[flags[j]], \n            name=flags[j], \n            mode=\"lines\",\n            ), row=i+2, col=1)    \n            \n    # Update xaxis properties\n    fig.update_xaxes(title_text= xcol, row=row_length+1, col=1)\n\n    # Update yaxis properties\n    for i in range(row_length):\n        fig.update_yaxes(title_text=ycol[i], row=i+1, col=1)\n    fig.update_yaxes(title_text=\"flags\", row=i+2, col=1)\n    \n    fig.update_xaxes(rangeslider={\"visible\":True}, row=row_length+1, col=1) # X軸に range slider を表示（下図参照\n    fig.update_layout(title=\"Time Series Analysis\") # グラフタイトルを設定\n    fig.update_layout(font={\"family\":\"Meiryo\", \"size\":12}) # フォントファミリとフォントサイズを指定\n    fig.update_layout(showlegend=True) # 凡例を強制的に表示\n    fig.update_layout(xaxis_type=\"linear\", yaxis_type=\"linear\") # lenear / log\n    fig.update_layout(xaxis_tickformat=',g')\n    fig.update_layout(width=1000, height=600)  # 図の高さを幅を指定\n    fig.update_layout(template=\"plotly_white\") # 白背景のテーマに変更\n\n    fig.show()","metadata":{"papermill":{"duration":0.664162,"end_time":"2023-05-21T02:12:41.700695","exception":false,"start_time":"2023-05-21T02:12:41.036533","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-27T15:43:30.569526Z","iopub.execute_input":"2023-05-27T15:43:30.569941Z","iopub.status.idle":"2023-05-27T15:43:30.588263Z","shell.execute_reply.started":"2023-05-27T15:43:30.569903Z","shell.execute_reply":"2023-05-27T15:43:30.586912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def normalize_column(df, col, low=0.05, high=0.95):\n    percentile_lo = df[col].quantile(low)\n    percentile_hi = df[col].quantile(high)\n    print(f\"percentile_lo:{percentile_lo}\")\n    print(f\"percentile_hi:{percentile_hi}\")\n    df[f\"{col}_0-1\"] = (df[col] - percentile_lo) / (percentile_hi - percentile_lo)\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-27T15:43:30.589782Z","iopub.execute_input":"2023-05-27T15:43:30.590147Z","iopub.status.idle":"2023-05-27T15:43:30.604219Z","shell.execute_reply.started":"2023-05-27T15:43:30.590113Z","shell.execute_reply":"2023-05-27T15:43:30.603188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ====================================\n# Line-plot: Start_Hesitation\n# ====================================\ndf[\"Time\"] = df.index / 128\ndf[\"AccV\"] /= 9.8\ndf[\"AccML\"] /= 9.8\ndf[\"AccAP\"] /= 9.8\ndf[\"StartHesitation\"] *= 0.8\ndf[\"Turn\"] *= 1.0\ndf[\"Walking\"] *= 1.2\nprint(f\"path: {defog_path[0]} {df.shape}\")\n\nxcol = \"Time\"\nycol = [\"AccV\", \"AccML\", \"AccAP\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_tdcsfog_plotly(df, xcol, ycol, flags, st, length)","metadata":{"papermill":{"duration":1.597362,"end_time":"2023-05-21T02:12:43.304318","exception":false,"start_time":"2023-05-21T02:12:41.706956","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-27T15:43:30.610448Z","iopub.execute_input":"2023-05-27T15:43:30.611210Z","iopub.status.idle":"2023-05-27T15:43:30.824702Z","shell.execute_reply.started":"2023-05-27T15:43:30.611174Z","shell.execute_reply":"2023-05-27T15:43:30.822180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# 積分値 (速度, 移動距離)\n# ==============================\n# df[\"AccV_cumsum\"] = df[\"AccV\"].cumsum()\n# df[\"AccML_cumsum\"] = df[\"AccML\"].cumsum()\ndf[\"AccAP_cumsum\"] = df[\"AccAP\"].cumsum().abs()\ndf[\"AccAP_dist\"] = df[\"AccAP_cumsum\"].cumsum()\n\nxcol = \"Time\"\nycol = [\"AccAP_cumsum\", \"AccAP_dist\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_tdcsfog_plotly(df, xcol, ycol, flags, st, length)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T15:43:30.827160Z","iopub.execute_input":"2023-05-27T15:43:30.827706Z","iopub.status.idle":"2023-05-27T15:43:30.988996Z","shell.execute_reply.started":"2023-05-27T15:43:30.827657Z","shell.execute_reply":"2023-05-27T15:43:30.987312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# 重ねて比較 (Plotly)\n# ==============================\ndef show_plotly_compare(df, xcol, ycol, flags, st, length):\n    tmp_df = df.copy()\n    if (st !=None) | (length != None):\n        tmp_df = tmp_df[st: st + length].copy()\n\n    row_length = len(ycol)\n    flag_length = len(flags)\n\n    fig = make_subplots(\n        rows=2, cols=1,\n        shared_xaxes=True\n        )\n\n    for i in range(row_length):\n        fig.add_trace(go.Scatter(\n            x=tmp_df[xcol], y=tmp_df[ycol[i]], \n            name=ycol[i], \n            mode=\"lines\",\n            ), row=1, col=1)\n        \n    for j in range(flag_length):\n        fig.add_trace(go.Scatter(\n            x=tmp_df[xcol], y=tmp_df[flags[j]], \n            name=flags[j], \n            mode=\"lines\",\n            ), row=2, col=1)    \n            \n    # Update xaxis properties\n    fig.update_xaxes(title_text= xcol, row=2, col=1)\n\n    # Update yaxis properties\n    fig.update_yaxes(title_text=\"ycol\", row=1, col=1)\n    fig.update_yaxes(title_text=\"flags\", row=2, col=1)\n    \n    fig.update_xaxes(rangeslider={\"visible\":True}, row=2, col=1) # X軸に range slider を表示（下図参照\n    fig.update_layout(title=\"Time Series Analysis\") # グラフタイトルを設定\n    fig.update_layout(font={\"family\":\"Meiryo\", \"size\":12}) # フォントファミリとフォントサイズを指定\n    fig.update_layout(showlegend=True) # 凡例を強制的に表示\n    fig.update_layout(xaxis_type=\"linear\", yaxis_type=\"linear\") # lenear / log\n    fig.update_layout(xaxis_tickformat=',g')\n    fig.update_layout(width=1000, height=600)  # 図の高さを幅を指定\n    fig.update_layout(template=\"plotly_white\") # 白背景のテーマに変更\n\n    fig.show()","metadata":{"papermill":{"duration":0.025615,"end_time":"2023-05-21T02:12:43.342071","exception":false,"start_time":"2023-05-21T02:12:43.316456","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-27T15:43:30.990471Z","iopub.execute_input":"2023-05-27T15:43:30.991441Z","iopub.status.idle":"2023-05-27T15:43:31.010185Z","shell.execute_reply.started":"2023-05-27T15:43:30.991397Z","shell.execute_reply":"2023-05-27T15:43:31.008415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# 移動平均\n# ==============================\ndf[f\"{col}_ma10\"] = df[f\"{col}\"].rolling(10, center=True).mean()\ndf[f\"{col}_ma50\"] = df[f\"{col}\"].rolling(50, center=True).mean()\ndf[f\"{col}_ma500\"] = df[f\"{col}\"].rolling(500, center=True).mean()\ndf[f\"{col}_ma5000\"] = df[f\"{col}\"].rolling(5000, center=True).mean()\n\nxcol = \"Time\"\nycol = [f\"{col}\", f\"{col}_ma10\", f\"{col}_ma50\", f\"{col}_ma500\", f\"{col}_ma5000\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"papermill":{"duration":0.104658,"end_time":"2023-05-21T02:12:43.459270","exception":false,"start_time":"2023-05-21T02:12:43.354612","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-27T15:43:31.012817Z","iopub.execute_input":"2023-05-27T15:43:31.013296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# 移動分散\n# ==============================\ndf[f\"{col}_std10\"] = df[f\"{col}\"].rolling(10, center=True).std()\ndf[f\"{col}_std50\"] = df[f\"{col}\"].rolling(50, center=True).std()\n\nxcol = \"Time\"\nycol = [f\"{col}\", f\"{col}_std10\", f\"{col}_std50\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"papermill":{"duration":0.094338,"end_time":"2023-05-21T02:12:43.575164","exception":false,"start_time":"2023-05-21T02:12:43.480826","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# 移動MAX, MIN\n# ==============================\ndf[f\"{col}_min50\"] = df[f\"{col}\"].rolling(50, center=True).min()\ndf[f\"{col}_max50\"] = df[f\"{col}\"].rolling(50, center=True).max()\n\nxcol = \"Time\"\nycol = [f\"{col}\", f\"{col}_min50\", f\"{col}_max50\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"papermill":{"duration":0.098204,"end_time":"2023-05-21T02:12:43.698878","exception":false,"start_time":"2023-05-21T02:12:43.600674","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# 生値と移動平均の差分\n# ==============================\ndf[f\"{col}_ma10_diff\"] = df[f\"{col}\"] - df[f\"{col}_ma10\"]\ndf[f\"{col}_ma50_diff\"] = df[f\"{col}\"] - df[f\"{col}_ma50\"]\n\nxcol = \"Time\"\nycol = [f\"{col}\", f\"{col}_ma10_diff\", f\"{col}_ma50_diff\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"papermill":{"duration":0.101883,"end_time":"2023-05-21T02:12:43.832530","exception":false,"start_time":"2023-05-21T02:12:43.730647","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# 生値と移動平均の差分の２乗\n# ==============================\ndf[f\"{col}_ma10_diff^2\"] = df[f\"{col}_ma10_diff\"]**2\ndf[f\"{col}_ma50_diff^2\"] = df[f\"{col}_ma50_diff\"]**2\n\nxcol = \"Time\"\nycol = [f\"{col}_ma10_diff^2\", f\"{col}_ma50_diff^2\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"papermill":{"duration":0.102764,"end_time":"2023-05-21T02:12:43.972007","exception":false,"start_time":"2023-05-21T02:12:43.869243","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = normalize_column(df, f\"{col}_ma50_diff^2\", low=0.00, high=1.00)\n\nxcol = \"Time\"\nycol = [f\"{col}_ma50_diff^2_0-1\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# 生値と移動平均の差分の２乗の移動平均\n# ==============================\ndf[f\"{col}_ma50_diff^2_ma200\"] = df[f\"{col}_ma50_diff^2\"].rolling(200, center=True).mean()\nxcol = \"Time\"\nycol = [f\"{col}_ma50_diff^2_ma200\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"papermill":{"duration":0.1018,"end_time":"2023-05-21T02:12:44.118051","exception":false,"start_time":"2023-05-21T02:12:44.016251","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# 生値と移動平均の差分の２乗の累積和\n# ==============================\ndf[f\"{col}_ma50_diff^2_cumsum\"] = df[f\"{col}_ma50_diff^2\"].cumsum()\nxcol = \"Time\"\nycol = [f\"{col}_ma50_diff^2_cumsum\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"papermill":{"duration":0.106941,"end_time":"2023-05-21T02:12:44.270582","exception":false,"start_time":"2023-05-21T02:12:44.163641","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def count_zerocrossing_zero(x):\n    arr = np.asarray(x)\n    return np.count_nonzero(np.diff(np.sign(arr))) - np.count_nonzero(arr==0)\n# np.signは正負(-:-1 or 0:0 or +:1)の判定","metadata":{"papermill":{"duration":0.052403,"end_time":"2023-05-21T02:12:44.372173","exception":false,"start_time":"2023-05-21T02:12:44.319770","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def count_01crossing(x):\n    arr = np.asarray(x) - 0.1\n    return np.count_nonzero(np.diff(np.sign(arr))) - np.count_nonzero(arr==0)\n# np.signは正負(-:-1 or 0:0 or +:1)の判定","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def count_05crossing(x):\n    arr = np.asarray(x) - 0.5\n    return np.count_nonzero(np.diff(np.sign(arr))) - np.count_nonzero(arr==0)\n# np.signは正負(-:-1 or 0:0 or +:1)の判定","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.interpolate(limit=int(1e+6), limit_direction='both', inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from seglearn.feature_functions import base_features, emg_features, all_features\nfrom tsflex.features import FeatureCollection, MultipleFeatureDescriptors\nfrom tsflex.features.integrations import seglearn_feature_dict_wrapper\n\n# ============================================\n# tsflex conditioning\n# ============================================\nfc = FeatureCollection([\n    MultipleFeatureDescriptors(\n        functions=[count_zerocrossing_zero],\n        series_names=[f'{col}_ma50_diff'],\n        windows=[500],\n        strides=[500],\n    ),\n    MultipleFeatureDescriptors(\n        functions=[count_zerocrossing_zero],\n        series_names=[f'{col}_ma50_diff'],\n        windows=[1000],\n        strides=[500],\n    ),\n    MultipleFeatureDescriptors(\n        functions=[count_zerocrossing_zero],\n        series_names=[f'{col}_ma50_diff'],\n        windows=[2000],\n        strides=[500],\n    ),\n]\n)","metadata":{"papermill":{"duration":0.512957,"end_time":"2023-05-21T02:12:44.929911","exception":false,"start_time":"2023-05-21T02:12:44.416954","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_feats = fc.calculate(\n    df, \n    return_df=True, # DataFrameでoutput\n    include_final_window=True, \n    approve_sparsity=True, \n    window_idx=\"middle\", # [\"begin\", \"middle\", \"end\"]\n    show_progress=True\n).astype(np.float32)\n\ndf_feats.index = df_feats.index.astype(\"int\")\ndf = df.merge(df_feats, how=\"left\", left_index=True, right_index=True)\ndf.interpolate(limit=int(1e+6), limit_direction='both', inplace=True)\nshow_df(df, 3, True)","metadata":{"papermill":{"duration":0.182251,"end_time":"2023-05-21T02:12:45.156731","exception":false,"start_time":"2023-05-21T02:12:44.974480","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xcol = \"Time\"\nycol = [f\"{col}_ma50_diff__count_zerocrossing_zero__w=500\", f\"{col}_ma50_diff__count_zerocrossing_zero__w=1000\", f\"{col}_ma50_diff__count_zerocrossing_zero__w=2000\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"papermill":{"duration":0.114198,"end_time":"2023-05-21T02:12:45.317007","exception":false,"start_time":"2023-05-21T02:12:45.202809","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ============================================\n# tsflex conditioning\n# ============================================\nfc = FeatureCollection([\n    MultipleFeatureDescriptors(\n        functions=[count_01crossing],\n        series_names=[f\"{col}_ma50_diff^2_0-1\"],\n        windows=[200],\n        strides=[200],\n    ),\n    MultipleFeatureDescriptors(\n        functions=[count_05crossing],\n        series_names=[f\"{col}_ma50_diff^2_0-1\"],\n        windows=[200],\n        strides=[200],\n    ),\n]\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_feats = fc.calculate(\n    df, \n    return_df=True, # DataFrameでoutput\n    include_final_window=True, \n    approve_sparsity=True, \n    window_idx=\"middle\", # [\"begin\", \"middle\", \"end\"]\n    show_progress=True\n).astype(np.float32)\n\ndf_feats.index = df_feats.index.astype(\"int\")\ndf = df.merge(df_feats, how=\"left\", left_index=True, right_index=True)\ndf.interpolate(limit=int(1e+6), limit_direction='both', inplace=True)\nshow_df(df, 3, True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xcol = \"Time\"\nycol = [f\"{col}_ma50_diff^2_0-1__count_01crossing__w=200\", f\"{col}_ma50_diff^2_0-1__count_05crossing__w=200\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[f\"{col}_ma50_diff^2_0-1_crossdiff\"] = df[f\"{col}_ma50_diff^2_0-1__count_01crossing__w=200\"] - df[f\"{col}_ma50_diff^2_0-1__count_05crossing__w=200\"]\nxcol = \"Time\"\nycol = [f\"{col}_ma50_diff^2_0-1_crossdiff\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_plotly_compare(df, xcol, ycol, flags, st, length) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom scipy.fft import fft\n# N = 1000  # データの長さ\n# dt = 0.01  # サンプリング間隔\n# t = np.arange(0, N) * dt  # 時間軸\n# x = np.random.randn(N)  # ランダムなデータ\n# x = df[f\"{col}\"][0:1000]\ndef fft_power_area_1_3(x):\n\n    N = len(x) # data_length\n    dt = 1/128 # 100, sampling_rate(sec)\n    t = np.arange(0, N) * dt\n\n    X = fft(x)\n    freq = np.fft.fftfreq(N, dt)\n    start_freq = 1  # 開始周波数\n    end_freq   = 3  # 終了周波数\n\n    start_index = np.argmax(freq >= start_freq)\n    end_index = np.argmax(freq >= end_freq)\n\n    power_spectrum = np.abs(X[start_index:end_index]) ** 2\n    integral_value = np.trapz(power_spectrum, freq[start_index:end_index])\n    return integral_value","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fft_power_area_3_8(x):\n\n    N = len(x) # data_length\n    dt = 1/128 # 100, sampling_rate(sec)\n    t = np.arange(0, N) * dt\n\n    X = fft(x)\n    freq = np.fft.fftfreq(N, dt)\n    start_freq = 3  # 開始周波数\n    end_freq   = 8  # 終了周波数\n\n    start_index = np.argmax(freq >= start_freq)\n    end_index = np.argmax(freq >= end_freq)\n\n    power_spectrum = np.abs(X[start_index:end_index]) ** 2\n    integral_value = np.trapz(power_spectrum, freq[start_index:end_index])\n    return integral_value","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ============================================\n# tsflex conditioning\n# ============================================\nfc = FeatureCollection([\n    MultipleFeatureDescriptors(\n        functions=[fft_power_area_1_3, fft_power_area_3_8],\n        series_names=[f\"{col}\"],\n        windows=[1000],\n        strides=[500],\n    ),\n]\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_feats = fc.calculate(\n    df, \n    return_df=True, # DataFrameでoutput\n    include_final_window=True, \n    approve_sparsity=True, \n    window_idx=\"middle\", # [\"begin\", \"middle\", \"end\"]\n    show_progress=True\n).astype(np.float32)\n\ndf_feats.index = df_feats.index.astype(\"int\")\ndf = df.merge(df_feats, how=\"left\", left_index=True, right_index=True)\ndf.interpolate(limit=int(1e+6), limit_direction='both', inplace=True)\nshow_df(df, 3, True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xcol = \"Time\"\nycol = [f\"{col}__fft_power_area_1_3__w=1000\", f\"{col}__fft_power_area_3_8__w=1000\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_plotly_compare(df, xcol, ycol, flags, st, length) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[f\"{col}_1_3_3_8_diff\"] = df[f\"{col}__fft_power_area_3_8__w=1000\"] - df[f\"{col}__fft_power_area_1_3__w=1000\"]\ndf[f\"{col}_1_3_3_8_ratio\"] = df[f\"{col}__fft_power_area_3_8__w=1000\"] / df[f\"{col}__fft_power_area_1_3__w=1000\"]\n\nxcol = \"Time\"\nycol = [f\"{col}_1_3_3_8_diff\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_plotly_compare(df, xcol, ycol, flags, st, length) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xcol = \"Time\"\nycol = [f\"{col}_1_3_3_8_ratio\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_plotly_compare(df, xcol, ycol, flags, st, length) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}