{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Investigation of Feature Engineering (for defog dataset)🤔\n--- \n[Objective]\n- Show time series data using plotly and investigate which feature is 　valide for ML.","metadata":{}},{"cell_type":"code","source":"# # Install tsflex and seglearn\n!pip install tsflex --no-index --find-links=file:///kaggle/input/time-series-tools\n!pip install seglearn --no-index --find-links=file:///kaggle/input/time-series-tools","metadata":{"execution":{"iopub.status.busy":"2023-05-29T13:41:06.692483Z","iopub.execute_input":"2023-05-29T13:41:06.692887Z","iopub.status.idle":"2023-05-29T13:41:30.036263Z","shell.execute_reply.started":"2023-05-29T13:41:06.692855Z","shell.execute_reply":"2023-05-29T13:41:30.034680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# =========================\n# Config\n# =========================\nidx = 33 # <- 0 -91のどれかを選択\ncol = \"AccV\" # AccV AccML AccAP\n\n# --------------------------------------------------------------------------------\n# idx of Best7 of StartHesitaiton\n# --------------------------------------------------------------------------------\n# 6, 11, 27, 57, 69, 88, 89, \n# --------------------------------------------------------------------------------\n# idx of Best5 of Turn\n# --------------------------------------------------------------------------------\n# 8, 29, 39, 43, 71, \n# --------------------------------------------------------------------------------\n# idx of Best5 of Walking\n# --------------------------------------------------------------------------------\n# 33, 37, 41, 51, 79, ","metadata":{"execution":{"iopub.status.busy":"2023-05-29T13:41:30.039373Z","iopub.execute_input":"2023-05-29T13:41:30.039868Z","iopub.status.idle":"2023-05-29T13:41:30.046702Z","shell.execute_reply.started":"2023-05-29T13:41:30.039824Z","shell.execute_reply":"2023-05-29T13:41:30.045567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# =========================\n# Import libraries\n# =========================\n# default\nimport gc, os, glob, random\nfrom os import path\nfrom pathlib import Path\n# make data\nimport polars as pl\nimport pandas as pd\npd.set_option('display.max_columns', None); # pd.set_option('display.max_rows', None)\nimport numpy as np\nfrom tqdm.auto import tqdm\nimport ydata_profiling as pdp","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2023-05-29T13:41:30.048234Z","iopub.execute_input":"2023-05-29T13:41:30.049165Z","iopub.status.idle":"2023-05-29T13:41:33.754781Z","shell.execute_reply.started":"2023-05-29T13:41:30.049133Z","shell.execute_reply":"2023-05-29T13:41:33.753548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_df(df, num=3, tail=True):\n    print(df.shape)\n    display(df.head(num))\n    if tail:\n        display(df.tail(num))","metadata":{"execution":{"iopub.status.busy":"2023-05-29T13:41:33.757794Z","iopub.execute_input":"2023-05-29T13:41:33.758599Z","iopub.status.idle":"2023-05-29T13:41:33.764505Z","shell.execute_reply.started":"2023-05-29T13:41:33.758556Z","shell.execute_reply":"2023-05-29T13:41:33.763518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_path = glob.glob(\"../input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/*.csv\")\ntdcsfog_path = glob.glob(\"../input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/*.csv\")\nnotype_path = glob.glob(\"../input/tlvmc-parkinsons-freezing-gait-prediction/train/notype/*.csv\")\n\nprint(f\"length of defog_path   : {len(defog_path)}\")\nprint(f\"length of tdcsfog_path : {len(tdcsfog_path)}\")\nprint(f\"length of notype_path  : {len(notype_path)}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-29T13:41:33.766054Z","iopub.execute_input":"2023-05-29T13:41:33.766477Z","iopub.status.idle":"2023-05-29T13:41:33.879913Z","shell.execute_reply.started":"2023-05-29T13:41:33.766423Z","shell.execute_reply":"2023-05-29T13:41:33.878655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"st_path_list = ['bf2fd0ff35','0d7ab3a9f9','8282009100','68e7e02a47','77d7d95074','1d99c2eecf','e069a57511']\ntu_path_list = ['3f970065e5', '3e6987cb2d', '6041cad8ec', '1ff78d55e9', '13a4fe5159']\nwa_path_list = ['54c6a21be6', 'da05ad7058', '8db3a7e46b', '32843e32b6', 'a2f1a8ab76']\n\nprint(\"-\"*80);print('idx of Best7 of StartHesitaiton');print('-'*80)\nfor i, _path in enumerate(defog_path):\n    if os.path.basename(_path).split(\".cs\")[0] in st_path_list:\n        print(i, end=\", \")\nprint()\nprint(\"-\"*80);print('idx of Best5 of Turn');print('-'*80)\nfor i, _path in enumerate(defog_path):\n    if os.path.basename(_path).split(\".cs\")[0] in tu_path_list:\n        print(i, end=\", \")\nprint()        \nprint(\"-\"*80);print('idx of Best5 of Walking');print('-'*80)\nfor i, _path in enumerate(defog_path):\n    if os.path.basename(_path).split(\".cs\")[0] in wa_path_list:\n        print(i, end=\", \")","metadata":{"execution":{"iopub.status.busy":"2023-05-29T13:41:33.881669Z","iopub.execute_input":"2023-05-29T13:41:33.882082Z","iopub.status.idle":"2023-05-29T13:41:33.892439Z","shell.execute_reply.started":"2023-05-29T13:41:33.882044Z","shell.execute_reply":"2023-05-29T13:41:33.891343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ====================================\n# load defog_data\n# ====================================\ndf = pd.read_csv(defog_path[idx])\nshow_df(df)\n\n# データフレームが長い場合、グラフの表示する開始インデックスとインデックス長を指定\nst = 0 ; length = len(df)\nif len(df) >= 150000:\n    st = 0 ; length = 150000","metadata":{"execution":{"iopub.status.busy":"2023-05-29T13:41:33.894109Z","iopub.execute_input":"2023-05-29T13:41:33.894816Z","iopub.status.idle":"2023-05-29T13:41:34.109065Z","shell.execute_reply.started":"2023-05-29T13:41:33.894775Z","shell.execute_reply":"2023-05-29T13:41:34.107946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# 可視化して確認 (Plotly)\n# ==============================\n# plotly \nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nimport plotly.express as px\n\ndef show_defog_plotly(df, xcol, ycol, flags, st, length):\n    tmp_df = df.copy()\n    if (st !=None) | (length != None):\n        tmp_df = tmp_df[st: st + length].copy()\n\n    row_length = len(ycol)\n    flag_length = len(flags)\n\n    fig = make_subplots(\n        rows=row_length+1, cols=1,\n        shared_xaxes=True\n        )\n\n    for i in range(row_length):\n        fig.add_trace(go.Scatter(\n            x=tmp_df[xcol], y=tmp_df[ycol[i]], \n            name=ycol[i], \n            mode=\"lines\",\n            ), row=i+1, col=1)\n        \n    for j in range(flag_length):\n        fig.add_trace(go.Scatter(\n            x=tmp_df[xcol], y=tmp_df[flags[j]], \n            name=flags[j], \n            mode=\"lines\",\n            ), row=i+2, col=1)    \n            \n    # Update xaxis properties\n    fig.update_xaxes(title_text= xcol, row=row_length+1, col=1)\n\n    # Update yaxis properties\n    for i in range(row_length):\n        fig.update_yaxes(title_text=ycol[i], row=i+1, col=1)\n    fig.update_yaxes(title_text=\"flags\", row=i+2, col=1)\n    \n    fig.update_xaxes(rangeslider={\"visible\":True}, row=row_length+1, col=1) # X軸に range slider を表示（下図参照\n    fig.update_layout(title=\"Time Series Analysis\") # グラフタイトルを設定\n    fig.update_layout(font={\"family\":\"Meiryo\", \"size\":12}) # フォントファミリとフォントサイズを指定\n    fig.update_layout(showlegend=True) # 凡例を強制的に表示\n    fig.update_layout(xaxis_type=\"linear\", yaxis_type=\"linear\") # lenear / log\n    fig.update_layout(xaxis_tickformat=',g')\n    fig.update_layout(width=1000, height=600)  # 図の高さを幅を指定\n    fig.update_layout(template=\"plotly_white\") # 白背景のテーマに変更\n\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T13:41:34.110182Z","iopub.execute_input":"2023-05-29T13:41:34.110488Z","iopub.status.idle":"2023-05-29T13:41:34.740553Z","shell.execute_reply.started":"2023-05-29T13:41:34.110463Z","shell.execute_reply":"2023-05-29T13:41:34.739423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ====================================\n# Line-plot: Start_Hesitation\n# ====================================\ndf[\"Time\"] = df.index / 100\ndf[\"StartHesitation\"] *= 0.8\ndf[\"Turn\"] *= 1.0\ndf[\"Walking\"] *= 1.2\ndf[\"Task\"] *= -0.8\ndf[\"Valid\"] *= -1.0\nprint(f\"path: {defog_path[0]} {df.shape}\")\n\nxcol = \"Time\"\nycol = [\"AccV\", \"AccML\", f\"{col}\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_defog_plotly(df, xcol, ycol, flags, st, length)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T13:41:34.741816Z","iopub.execute_input":"2023-05-29T13:41:34.742142Z","iopub.status.idle":"2023-05-29T13:41:36.502601Z","shell.execute_reply.started":"2023-05-29T13:41:34.742108Z","shell.execute_reply":"2023-05-29T13:41:36.500856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def normalize_column(df, col, low=0.05, high=0.95):\n    percentile_lo = df[col].quantile(low)\n    percentile_hi = df[col].quantile(high)\n    print(f\"percentile_lo:{percentile_lo}\")\n    print(f\"percentile_hi:{percentile_hi}\")\n    df[f\"{col}_0-1\"] = (df[col] - percentile_lo) / (percentile_hi - percentile_lo)\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-29T13:41:36.507765Z","iopub.execute_input":"2023-05-29T13:41:36.508686Z","iopub.status.idle":"2023-05-29T13:41:36.518915Z","shell.execute_reply.started":"2023-05-29T13:41:36.508635Z","shell.execute_reply":"2023-05-29T13:41:36.518074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ====================================\n# Line-plot: Start_Hesitation\n# ====================================\ndf = normalize_column(df, \"AccV\")\ndf = normalize_column(df, \"AccML\")\ndf = normalize_column(df, \"AccAP\")\ndf[\"AccV*ML*AP\"] = df[\"AccV_0-1\"] * df[\"AccML_0-1\"] * df[\"AccAP_0-1\"] \n\nxcol = \"Time\"\nycol = [\"AccV_0-1\", \"AccML_0-1\", \"AccAP_0-1\", \"AccV*ML*AP\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_defog_plotly(df, xcol, ycol, flags, st, length)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T13:41:36.520060Z","iopub.execute_input":"2023-05-29T13:41:36.520846Z","iopub.status.idle":"2023-05-29T13:41:36.934009Z","shell.execute_reply.started":"2023-05-29T13:41:36.520816Z","shell.execute_reply":"2023-05-29T13:41:36.929775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# 積分値 (速度, 移動距離)\n# ==============================\n# df[\"AccV_cumsum\"] = df[\"AccV\"].cumsum()\n# df[\"AccML_cumsum\"] = df[\"AccML\"].cumsum()\ndf[f\"{col}_cumsum\"] = df[f\"{col}\"].cumsum().abs()\ndf[f\"{col}_dist\"] = df[f\"{col}_cumsum\"].cumsum()\n\nxcol = \"Time\"\nycol = [f\"{col}_cumsum\", f\"{col}_dist\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\n# show_defog_plotly(df, xcol, ycol, flags, st, length)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# 重ねて比較 (Plotly)\n# ==============================\ndef show_plotly_compare(df, xcol, ycol, flags, st, length):\n    tmp_df = df.copy()\n    if (st !=None) | (length != None):\n        tmp_df = tmp_df[st: st + length].copy()\n\n    row_length = len(ycol)\n    flag_length = len(flags)\n\n    fig = make_subplots(\n        rows=2, cols=1,\n        shared_xaxes=True\n        )\n\n    for i in range(row_length):\n        fig.add_trace(go.Scatter(\n            x=tmp_df[xcol], y=tmp_df[ycol[i]], \n            name=ycol[i], \n            mode=\"lines\",\n            ), row=1, col=1)\n        \n    for j in range(flag_length):\n        fig.add_trace(go.Scatter(\n            x=tmp_df[xcol], y=tmp_df[flags[j]], \n            name=flags[j], \n            mode=\"lines\",\n            ), row=2, col=1)    \n            \n    # Update xaxis properties\n    fig.update_xaxes(title_text= xcol, row=2, col=1)\n\n    # Update yaxis properties\n    fig.update_yaxes(title_text=\"ycol\", row=1, col=1)\n    fig.update_yaxes(title_text=\"flags\", row=2, col=1)\n    \n    fig.update_xaxes(rangeslider={\"visible\":True}, row=2, col=1) # X軸に range slider を表示（下図参照\n    fig.update_layout(title=\"Time Series Analysis\") # グラフタイトルを設定\n    fig.update_layout(font={\"family\":\"Meiryo\", \"size\":12}) # フォントファミリとフォントサイズを指定\n    fig.update_layout(showlegend=True) # 凡例を強制的に表示\n    fig.update_layout(xaxis_type=\"linear\", yaxis_type=\"linear\") # lenear / log\n    fig.update_layout(xaxis_tickformat=',g')\n    fig.update_layout(width=1000, height=600)  # 図の高さを幅を指定\n    fig.update_layout(template=\"plotly_white\") # 白背景のテーマに変更\n\n    fig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# 移動平均\n# ==============================\ndf[f\"{col}_ma50\"] = df[f\"{col}\"].rolling(50, center=True).mean()\ndf[f\"{col}_ma500\"] = df[f\"{col}\"].rolling(500, center=True).mean()\n\nxcol = \"Time\"\nycol = [f\"{col}\", f\"{col}_ma50\", f\"{col}_ma500\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\n# show_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# 移動分散\n# ==============================\ndf[f\"{col}_std10\"] = df[f\"{col}\"].rolling(10, center=True).std()\ndf[f\"{col}_std50\"] = df[f\"{col}\"].rolling(50, center=True).std()\n\nxcol = \"Time\"\nycol = [f\"{col}_std10\", f\"{col}_std50\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\n# show_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# 移動MAX, MIN\n# ==============================\ndf[f\"{col}_min50\"] = df[f\"{col}\"].rolling(50, center=True).min()\ndf[f\"{col}_max50\"] = df[f\"{col}\"].rolling(50, center=True).max()\n\nxcol = \"Time\"\nycol = [f\"{col}_min50\", f\"{col}_max50\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\n# show_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# 生値と移動平均の差分\n# ==============================\ndf[f\"{col}_ma50_diff\"] = df[f\"{col}\"] - df[f\"{col}_ma50\"]\ndf[f\"{col}_ma500_diff\"] = df[f\"{col}\"] - df[f\"{col}_ma500\"]\n\nxcol = \"Time\"\nycol = [f\"{col}_ma50_diff\", f\"{col}_ma500_diff\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# 生値と移動平均の差分の２乗\n# ==============================\ndf[f\"{col}_ma50_diff^2\"] = df[f\"{col}_ma50_diff\"]**2\ndf[f\"{col}_ma500_diff^2\"] = df[f\"{col}_ma500_diff\"]**2\n\nxcol = \"Time\"\nycol = [f\"{col}_ma50_diff^2\", f\"{col}_ma500_diff^2\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = normalize_column(df, f\"{col}_ma50_diff^2\", low=0.00, high=1.00)\ndf = normalize_column(df, f\"{col}_ma500_diff^2\", low=0.00, high=1.00)\n\nxcol = \"Time\"\nycol = [f\"{col}_ma50_diff^2_0-1\", f\"{col}_ma500_diff^2_0-1\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\n# show_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# 生値と移動平均の差分の２乗の移動平均\n# ==============================\ndf[f\"{col}_ma50_diff^2_ma200\"] = df[f\"{col}_ma500_diff^2\"].rolling(200, center=True).mean()\nxcol = \"Time\"\nycol = [f\"{col}_ma50_diff^2_ma200\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# 生値と移動平均の差分の２乗の累積和\n# ==============================\ndf[f\"{col}_ma50_diff^2_cumsum\"] = df[f\"{col}_ma50_diff^2\"].cumsum()\nxcol = \"Time\"\nycol = [f\"{col}_ma50_diff^2_cumsum\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\n# show_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def count_zerocrossing_zero(x):\n    arr = np.asarray(x)\n    return np.count_nonzero(np.diff(np.sign(arr))) - np.count_nonzero(arr==0)\n# np.signは正負(-:-1 or 0:0 or +:1)の判定\n\ndef count_01crossing(x):\n    arr = np.asarray(x) - 0.1\n    return np.count_nonzero(np.diff(np.sign(arr))) - np.count_nonzero(arr==0)\n\ndef count_05crossing(x):\n    arr = np.asarray(x) - 0.5\n    return np.count_nonzero(np.diff(np.sign(arr))) - np.count_nonzero(arr==0)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.interpolate(limit=int(1e+6), limit_direction='both', inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from seglearn.feature_functions import base_features, emg_features, all_features\nfrom seglearn.feature_functions import zero_crossing\nfrom tsflex.features import FeatureCollection, MultipleFeatureDescriptors\nfrom tsflex.features.integrations import seglearn_feature_dict_wrapper\n\n# ============================================\n# tsflex conditioning\n# ============================================\nfc = FeatureCollection([\n    MultipleFeatureDescriptors(\n        functions=[count_zerocrossing_zero, np.sum],\n        series_names=[f'{col}_ma50_diff'],\n        windows=[500],\n        strides=[500],\n    ),\n]\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_feats = fc.calculate(\n    df, \n    return_df=True, # DataFrameでoutput\n    include_final_window=True, \n    approve_sparsity=True, \n    window_idx=\"middle\", # [\"begin\", \"middle\", \"end\"]\n    show_progress=True\n).astype(np.float32)\n\ndf_feats.index = df_feats.index.astype(\"int\")\ndf = df.merge(df_feats, how=\"left\", left_index=True, right_index=True)\ndf.interpolate(limit=int(1e+6), limit_direction='both', inplace=True)\nshow_df(df, 3, True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xcol = \"Time\"\nycol = [f\"{col}_ma50_diff__count_zerocrossing_zero__w=500\", f\"{col}_ma50_diff__count_zerocrossing_zero__w=1000\", f\"{col}_ma50_diff__count_zerocrossing_zero__w=2000\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\n# show_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ============================================\n# tsflex conditioning\n# ============================================\nfc = FeatureCollection([\n    MultipleFeatureDescriptors(\n        functions=[count_01crossing],\n        series_names=[f\"{col}_ma50_diff^2_0-1\"],\n        windows=[200],\n        strides=[200],\n    ),\n    MultipleFeatureDescriptors(\n        functions=[count_05crossing],\n        series_names=[f\"{col}_ma50_diff^2_0-1\"],\n        windows=[200],\n        strides=[200],\n    ),\n]\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_feats = fc.calculate(\n    df, \n    return_df=True, # DataFrameでoutput\n    include_final_window=True, \n    approve_sparsity=True, \n    window_idx=\"middle\", # [\"begin\", \"middle\", \"end\"]\n    show_progress=True\n).astype(np.float32)\n\ndf_feats.index = df_feats.index.astype(\"int\")\ndf = df.merge(df_feats, how=\"left\", left_index=True, right_index=True)\ndf.interpolate(limit=int(1e+6), limit_direction='both', inplace=True)\nshow_df(df, 3, True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xcol = \"Time\"\nycol = [f\"{col}_ma50_diff^2_0-1__count_01crossing__w=200\", f\"{col}_ma50_diff^2_0-1__count_05crossing__w=200\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\n# show_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[f\"{col}_ma50_diff^2_0-1_crossdiff\"] = df[f\"{col}_ma50_diff^2_0-1__count_01crossing__w=200\"] - df[f\"{col}_ma50_diff^2_0-1__count_05crossing__w=200\"]\nxcol = \"Time\"\nycol = [f\"{col}_ma50_diff^2_0-1_crossdiff\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_plotly_compare(df, xcol, ycol, flags, st, length) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fft_power_area_3_8(x):\n\n    N = len(x) # data_length\n    dt = 1/100 # 100, sampling_rate(sec)\n    t = np.arange(0, N) * dt\n\n    X = fft(x)\n    freq = np.fft.fftfreq(N, dt)\n    start_freq = 3  # 開始周波数\n    end_freq   = 8  # 終了周波数\n\n    start_index = np.argmax(freq >= start_freq)\n    end_index = np.argmax(freq >= end_freq)\n\n    power_spectrum = np.abs(X[start_index:end_index]) ** 2\n    integral_value = np.trapz(power_spectrum, freq[start_index:end_index])\n    return integral_value","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom scipy.fft import fft\n# N = 1000  # データの長さ\n# dt = 0.01  # サンプリング間隔\n# t = np.arange(0, N) * dt  # 時間軸\n# x = np.random.randn(N)  # ランダムなデータ\n# x = df[f\"{col}\"][0:1000]\ndef fft_power_area_1_3(x):\n\n    N = len(x) # data_length\n    dt = 1/100 # 100, sampling_rate(sec)\n    t = np.arange(0, N) * dt\n\n    X = fft(x)\n    freq = np.fft.fftfreq(N, dt)\n    start_freq = 1  # 開始周波数\n    end_freq   = 3  # 終了周波数\n\n    start_index = np.argmax(freq >= start_freq)\n    end_index = np.argmax(freq >= end_freq)\n\n    power_spectrum = np.abs(X[start_index:end_index]) ** 2\n    integral_value = np.trapz(power_spectrum, freq[start_index:end_index])\n    return integral_value","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ============================================\n# tsflex conditioning\n# ============================================\nfc = FeatureCollection([\n    MultipleFeatureDescriptors(\n        functions=[fft_power_area_1_3, fft_power_area_3_8],\n        series_names=[f\"{col}\"],\n        windows=[1000],\n        strides=[250, 500, 1000],\n    ),\n]\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_feats = fc.calculate(\n    df, \n    return_df=True, # DataFrameでoutput\n    include_final_window=True, \n    approve_sparsity=True, \n    window_idx=\"middle\", # [\"begin\", \"middle\", \"end\"]\n    show_progress=True\n).astype(np.float32)\n\ndf_feats.index = df_feats.index.astype(\"int\")\ndf = df.merge(df_feats, how=\"left\", left_index=True, right_index=True)\ndf.interpolate(limit=int(1e+6), limit_direction='both', inplace=True)\nshow_df(df, 3, True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xcol = \"Time\"\nycol = [f\"{col}__fft_power_area_1_3__w=1000\", f\"{col}__fft_power_area_3_8__w=1000\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_plotly_compare(df, xcol, ycol, flags, st, length) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[f\"{col}_1_3_3_8_diff\"] = df[f\"{col}__fft_power_area_3_8__w=1000\"] - df[f\"{col}__fft_power_area_1_3__w=1000\"]\ndf[f\"{col}_1_3_3_8_ratio\"] = df[f\"{col}__fft_power_area_3_8__w=1000\"] / df[f\"{col}__fft_power_area_1_3__w=1000\"]\n\nxcol = \"Time\"\nycol = [f\"{col}_1_3_3_8_diff\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_plotly_compare(df, xcol, ycol, flags, st, length) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xcol = \"Time\"\nycol = [f\"{col}_1_3_3_8_ratio\"]\nflags = [\"StartHesitation\", \"Turn\", \"Walking\"]\nshow_plotly_compare(df, xcol, ycol, flags, st, length) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# xcol = \"Time\"\n# ycol = [f\"{col}_ma50_diff__count_zerocrossing_zero__w=50\", f\"{col}_ma50_diff__count_zerocrossing_zero__w=500\", f\"{col}_ma500_diff__count_zerocrossing_zero__w=500\"]\n# flags = [\"StartHesitation\", \"Turn\", \"Walking\"]\n# show_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import librosa\n\n# # 音声ファイルからリズムに関する成分 (novelty function) を抽出する関数\n# def compute_spectral_based_novelty(y, hop_length):\n#     Y = np.log(np.abs(librosa.stft(y=y, hop_length=hop_length))+1)\n#     n_freq, n_tf = Y.shape\n#     spectral_novelty = np.zeros(n_tf-1)\n#     # Compute and accumulate novelty function each frequencies\n#     for f in range(0, n_freq):\n#         tmp = Y[f,1:] - Y[f,:-1]\n#         tmp[tmp<0.0] = 0.0\n#         spectral_novelty += tmp\n#     # Normalization\n#     spectral_novelty /= np.max(spectral_novelty)\n    \n#     return spectral_novelty\n\n\n# # novelty functionからフーリエテンポグラムを得る関数\n# def compute_fourier_tempogram(novelty_function, sampling_rate, hop_length, n_fft):\n#     ftg = np.abs(librosa.stft(novelty_function, hop_length=hop_length, n_fft=n_fft))\n#     sr_minite = 60 * sampling_rate\n#     ftg_bpms = np.linspace(0, float(sr_minite) / 2, int(1 + n_fft//2), endpoint=True)\n#     return ftg, ftg_bpms\n\n\n# # テンポの範囲を制限 (clipping) する関数\n# def adjust_tempograms(tg, bpms, bpm_min=-np.inf, bpm_max=np.inf):\n#     indices = np.where(np.logical_and(bpms>=bpm_min, bpms<=bpm_max))[0]\n#     return tg[indices], bpms[indices]\n\n\n# # テンポグラムの描画関数\n# def plot_tg(tg, bpms, ticks_base=10):\n#     plt.figure(figsize=(5, 3))\n#     plt.imshow(tg[::-1], aspect=\"auto\", cmap=\"jet\")\n#     plt.colorbar()\n#     plt.xlabel(\"time [s]\")\n#     plt.ylabel(\"BPM\")\n#     ticks = 2 * (len(bpms) // ticks_base)\n#     plt.yticks(np.arange(0,len(bpms), ticks), np.round(bpms[::-1][::ticks], 3))\n#     plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sampling_rate = 100  # novelty function抽出後のサンプリングレート（ハイパラ）\n# bpm_resolution = 4  # BPMの分解能（音声ファイルの長さ的にこれが限界の模様）\n# bpm_min = 40\n# bpm_max = 200\n\n# sr = 22050\n# hop_length = sr // sampling_rate\n# n_fft = sampling_rate * 60 // bpm_resolution ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ftg, ftg_bpms = compute_fourier_tempogram(df[f\"{col}_ma500\"].to_numpy(), sampling_rate, hop_length, n_fft)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def detect_bpm(x):\n#     sr =1024\n#     tempo, beats = librosa.beat.beat_track(y=x, sr=sr)\n#     return tempo","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fc = FeatureCollection([\n#     MultipleFeatureDescriptors(\n#         functions=[detect_bpm],\n#         series_names=[f'{col}_ma50_diff'],\n#         windows=[5000],\n#         strides=[5000],\n#     ),\n#     MultipleFeatureDescriptors(\n#         functions=[detect_bpm],\n#         series_names=[f'{col}_ma500_diff'],\n#         windows=[2500],\n#         strides=[2500],\n#     ),\n#     MultipleFeatureDescriptors(\n#         functions=[detect_bpm],\n#         series_names=[f'{col}_ma500_diff'],\n#         windows=[5000],\n#         strides=[5000],\n#     )\n# ])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_feats = fc.calculate(\n#     df, \n#     return_df=True, # DataFrameでoutput\n#     include_final_window=True, \n#     approve_sparsity=True, \n#     window_idx=\"middle\", # [\"begin\", \"middle\", \"end\"]\n#     show_progress=True\n# ).astype(np.float32)\n\n# df_feats.index = df_feats.index.astype(\"int\")\n# df = df.merge(df_feats, how=\"left\", left_index=True, right_index=True)\n# df.interpolate(limit=int(1e+6), limit_direction='both', inplace=True)\n# show_df(df, 3, True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# xcol = \"Time\"\n# ycol = [f\"{col}_ma500_diff__detect_bpm__w=2500\", f\"{col}_ma500_diff__detect_bpm__w=5000\", f\"{col}_ma50_diff__detect_bpm__w=5000\"]\n# flags = [\"StartHesitation\", \"Turn\", \"Walking\"]\n# show_plotly_compare(df, xcol, ycol, flags, st, length)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"scrolled":true},"execution_count":null,"outputs":[]}]}