{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport plotly.express as px\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objects as go\n\nimport os\nimport gc\nimport pickle\n\nfrom IPython.core.debugger import set_trace\n\nfrom tqdm import tqdm\nfrom sklearn import preprocessing\n\nimport lightgbm\nimport xgboost\nfrom sklearn.ensemble import RandomForestClassifier \nfrom sklearn.ensemble import AdaBoostClassifier \nfrom sklearn.linear_model import LogisticRegression\n\n\nimport catboost\nimport random\nrandom.seed(20)\n\n# Install tsflex and seglearn\n!pip install tsflex --no-index --find-links=file:///kaggle/input/tsflex\n!pip install seglearn --no-index --find-links=file:///kaggle/input/segalearn\n\n\nfrom seglearn.feature_functions import base_features, emg_features\n\nfrom tsflex.features import FeatureCollection, MultipleFeatureDescriptors\nfrom tsflex.features.integrations import seglearn_feature_dict_wrapper","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-30T11:06:27.213786Z","iopub.execute_input":"2023-05-30T11:06:27.214415Z","iopub.status.idle":"2023-05-30T11:06:51.571165Z","shell.execute_reply.started":"2023-05-30T11:06:27.214339Z","shell.execute_reply":"2023-05-30T11:06:51.569611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fix_invalid_events(df):\n    for e_type in [\"StartHesitation\", \"Turn\",'Walking']:\n        df.loc[(df[\"Valid\"] == False) | (df[\"Task\"] == False), e_type] = 0\n        \n    return df\ndef min_max_feature(df, feature):\n    new_feature = f\"precent_prograss_{feature}\"\n    df[new_feature] = (df[feature] - df[feature].min()) / (df[feature].max() - df[feature].min())\n    df[new_feature] = df[new_feature]                                                                                                     \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:06:51.574267Z","iopub.execute_input":"2023-05-30T11:06:51.574809Z","iopub.status.idle":"2023-05-30T11:06:51.584889Z","shell.execute_reply.started":"2023-05-30T11:06:51.574754Z","shell.execute_reply":"2023-05-30T11:06:51.583624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_memory_usage(df):\n    \n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                continue\n    mem_usg = df.memory_usage().sum() / 1024**2 \n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:06:51.586636Z","iopub.execute_input":"2023-05-30T11:06:51.587002Z","iopub.status.idle":"2023-05-30T11:06:51.605225Z","shell.execute_reply.started":"2023-05-30T11:06:51.586955Z","shell.execute_reply":"2023-05-30T11:06:51.604068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:06:51.606952Z","iopub.execute_input":"2023-05-30T11:06:51.607377Z","iopub.status.idle":"2023-05-30T11:06:51.825246Z","shell.execute_reply.started":"2023-05-30T11:06:51.607340Z","shell.execute_reply":"2023-05-30T11:06:51.824177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def FE_tsflex(df, is_td, isTest=False ):\n#     tdcsfog (128 timesteps per second)\n#     defog (100 timesteps per second).\n\n    if is_td:\n        second = 128\n    else:\n        second = 100\n\n    \n    if not isTest:\n        if \"Valid\" in df.columns:\n            df = fix_invalid_events(df)\n\n    for col in [\"Time\", \"AccV\", \"AccML\", \"AccAP\"]:\n        df = min_max_feature(df, col)\n    \n    #TODO TDFLEX FEATURES\n    basic_feats = MultipleFeatureDescriptors(\n        functions=seglearn_feature_dict_wrapper(base_features()),\n        series_names=['AccV', 'AccML', 'AccAP'],\n        windows=[5_000, 10_000],\n        strides=[5_000, 10_000],\n    )\n\n    emg_feats = emg_features()\n    del emg_feats['simple square integral'] # is same as abs_energy (which is in base_features)\n\n    emg_feats = MultipleFeatureDescriptors(\n        functions=seglearn_feature_dict_wrapper(emg_feats),\n        series_names=['AccV', 'AccML', 'AccAP'],\n        windows=[5_000, 10_000],\n        strides=[5_000, 10_000],\n    )\n\n    fc = FeatureCollection([basic_feats, emg_feats])\n    df_feats = fc.calculate(df, return_df=True, include_final_window=True, approve_sparsity=True, window_idx=\"begin\")\n    df = df.merge(df_feats, how=\"left\", left_index=True, right_index=True).fillna(method=\"ffill\")\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:06:51.829590Z","iopub.execute_input":"2023-05-30T11:06:51.830067Z","iopub.status.idle":"2023-05-30T11:06:51.841252Z","shell.execute_reply.started":"2023-05-30T11:06:51.830026Z","shell.execute_reply":"2023-05-30T11:06:51.839773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def describe_max_min(df, ranges_maxs):\n    new_cols = []\n    for col in [\"AccV\", \"AccML\", \"AccAP\"]:\n        past_1 = df[col].shift(1).fillna(method=\"bfill\")\n        future_1 = df[col].shift(-1).fillna(method=\"ffill\")\n\n        is_max = np.where((df[col] > past_1) & (df[col] > future_1), True, False)\n        is_min = np.where((df[col] < past_1) & (df[col] < future_1), True, False)\n\n        last_max_temp = df[col].where(is_max).ffill().fillna(0)\n        last_min_temp = df[col].where(is_min).ffill().fillna(0)\n\n        maxs_df = pd.DataFrame(df[col].where(is_max).dropna())\n        maxs_df['Time'] = df[\"Time\"].where(is_max).dropna().astype(\"int\")\n\n        cols_to_stat = []\n        for lags_max in ranges_maxs:\n            lags = list(range(1,lags_max+1))\n            for lag in lags:\n                if f\"{col}_max_{lag}_ago\" not in cols_to_stat:\n                    cols_to_stat.append(f\"{col}_max_{lag}_ago\")\n                    maxs_df[f\"{col}_max_{lag}_ago\"] = maxs_df[col].shift(lag).fillna(method=\"bfill\")\n\n            df[f\"{col}_min_past_{lags_max}_maxs\"] = maxs_df[cols_to_stat].min(axis=1)\n            df[f\"{col}_max_past_{lags_max}_maxs\"] = maxs_df[cols_to_stat].max(axis=1)\n            df[f\"{col}_mean_past_{lags_max}_maxs\"] = maxs_df[cols_to_stat].mean(axis=1)\n            df[f\"{col}_std_past_{lags_max}_maxs\"] = maxs_df[cols_to_stat].std(axis=1)\n\n            new_cols.extend([f\"{col}_min_past_{lags_max}_maxs\", \n                            f\"{col}_max_past_{lags_max}_maxs\",\n                            f\"{col}_mean_past_{lags_max}_maxs\",\n                            f\"{col}_std_past_{lags_max}_maxs\"])\n\n\n        mins_df = pd.DataFrame(df[col].where(is_min).dropna())\n        mins_df['Time'] = df[\"Time\"].where(is_min).dropna().astype(\"int\")\n\n        cols_to_stat = []\n        for lags_max in ranges_maxs:\n            lags = list(range(1,lags_max+1))\n            for lag in lags:\n                if f\"{col}_min_{lag}_ago\" not in cols_to_stat:\n                    cols_to_stat.append(f\"{col}_min_{lag}_ago\")\n                    mins_df[f\"{col}_min_{lag}_ago\"] = mins_df[col].shift(lag).fillna(method=\"bfill\")\n\n            df[f\"{col}_min_past_{lags_max}_mins\"] = mins_df[cols_to_stat].min(axis=1)\n            df[f\"{col}_max_past_{lags_max}_mins\"] = mins_df[cols_to_stat].max(axis=1)\n            df[f\"{col}_mean_past_{lags_max}_mins\"] = mins_df[cols_to_stat].mean(axis=1)\n            df[f\"{col}_std_past_{lags_max}_mins\"] = mins_df[cols_to_stat].std(axis=1)\n    \n            new_cols.extend([f\"{col}_min_past_{lags_max}_mins\", \n                f\"{col}_max_past_{lags_max}_mins\",\n                f\"{col}_mean_past_{lags_max}_mins\",\n                f\"{col}_std_past_{lags_max}_mins\"])\n    for col in new_cols:\n        df[col] = df[col].fillna(method=\"ffill\").fillna(method=\"bfill\")\n    del mins_df, maxs_df\n    gc.collect()\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:06:51.843426Z","iopub.execute_input":"2023-05-30T11:06:51.843960Z","iopub.status.idle":"2023-05-30T11:06:51.865067Z","shell.execute_reply.started":"2023-05-30T11:06:51.843904Z","shell.execute_reply":"2023-05-30T11:06:51.863574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lag_feature(df,col_to_lag, lag, second,lag_scale=\"Time\" ):\n    if lag_scale == \"Time\":\n        df[f\"{col_to_lag}_{lag}_{lag_scale}_ago\"] = df[col_to_lag].shift(lag*second).fillna(method=\"bfill\")\n        df[f\"{col_to_lag}_{lag}_{lag_scale}_from_now\"] = df[col_to_lag].shift(-lag*second).fillna(method=\"ffill\")\n        if df[f\"{col_to_lag}_{lag}_{lag_scale}_ago\"].isna().sum() > 0:\n            df[f\"{col_to_lag}_{lag}_{lag_scale}_ago\"] = df[f\"{col_to_lag}_{lag}_{lag_scale}_ago\"].fillna(0)\n        if df[f\"{col_to_lag}_{lag}_{lag_scale}_from_now\"].isna().sum() > 0:\n            df[f\"{col_to_lag}_{lag}_{lag_scale}_from_now\"] = df[f\"{col_to_lag}_{lag}_{lag_scale}_from_now\"].fillna(0)\n        df[f\"{col_to_lag}_{lag}_{lag_scale}_ago\"] = df[f\"{col_to_lag}_{lag}_{lag_scale}_ago\"].astype(\"float32\")\n        df[f\"{col_to_lag}_{lag}_{lag_scale}_from_now\"] = df[f\"{col_to_lag}_{lag}_{lag_scale}_from_now\"].astype(\"float32\")\n                                                                                                               \n    return df\n\ndef min_max_feature(df, feature):\n    new_feature = f\"precent_prograss_{feature}\"\n    df[new_feature] = (df[feature] - df[feature].min()) / (df[feature].max() - df[feature].min())\n    df[new_feature] = df[new_feature]                                                                                                     \n    return df\n\ndef fix_invalid_events(df):\n    for e_type in [\"StartHesitation\", \"Turn\",'Walking']:\n        df.loc[(df[\"Valid\"] == False) | (df[\"Task\"] == False), e_type] = 0\n        \n    return df\n\n\"\"\"\n- Stats in fences, for example, 2 seconds ago - 4 seconds ago,\nwhat was the mean? std? max? min?\n\"\"\"\ndef fences_features(df,col, margin, width, stat, second):\n    stat_feature = f\"moving_{stat}_{col}_{width}\"\n    df[f\"fence_{col}_m{margin}_w{width}_past\"] = df[stat_feature].shift(margin*second).fillna(method=\"bfill\")\n    df[f\"fence_{col}_m{margin}_w{width}_future\"] = df[stat_feature].shift(-margin*second).fillna(method=\"ffill\")\n    \n    df[f\"fence_{col}_m{margin}_w{width}_past\"] = df[f\"fence_{col}_m{margin}_w{width}_past\"].fillna(0)\n    df[f\"fence_{col}_m{margin}_w{width}_future\"] = df[f\"fence_{col}_m{margin}_w{width}_future\"].fillna(0)\n    df[f\"fence_{col}_m{margin}_w{width}_future\"]= df[f\"fence_{col}_m{margin}_w{width}_future\"].replace([np.inf, -np.inf], np.nan).fillna(method=\"bfill\").fillna(-1).astype(\"float32\")\n    df[f\"fence_{col}_m{margin}_w{width}_past\"]= df[f\"fence_{col}_m{margin}_w{width}_past\"].replace([np.inf, -np.inf], np.nan).fillna(method=\"bfill\").fillna(-1).astype(\"float32\")\n\n    return df\n\n# def whole file stats - std/min/max/precentiles\ndef whole_file_stats(df):\n    for col in [\"AccV\", \"AccML\", \"AccAP\"]:\n        df[f\"graph_{col}_mean\"] = df[col].mean()\n        df[f\"graph_{col}_std\"] = df[col].std()\n        df[f\"graph_{col}_min\"] = df[col].min()\n        df[f\"graph_{col}_max\"] = df[col].max()\n        \n        half_file = df.shape[0] // 2\n        \n        first_half = df.iloc[:half_file]\n        second_half = df.iloc[half_file:]\n        \n        df[f\"graph_first_half_{col}_mean\"] = first_half[col].mean()\n        df[f\"graph_first_half_{col}_std\"] = first_half[col].std()\n        df[f\"graph_first_half_{col}_min\"] = first_half[col].min()\n        df[f\"graph_first_half_{col}_max\"] = first_half[col].max()\n        \n        df[f\"graph_second_half_{col}_mean\"] = second_half[col].mean()\n        df[f\"graph_second_half_{col}_std\"] = second_half[col].std()\n        df[f\"graph_second_half_{col}_min\"] = second_half[col].min()\n        df[f\"graph_second_half_{col}_max\"] = second_half[col].max()\n    return df\n        \ndef FE(df, is_td, isTest=False ):\n#     tdcsfog (128 timesteps per second)\n#     defog (100 timesteps per second).\n\n    df = describe_max_min(df, [10,20, 30, 40])\n    df = whole_file_stats(df)\n\n    if is_td:\n        second = 128\n    else:\n        second = 100\n\n    \n    if not isTest:\n        if \"Valid\" in df.columns:\n            df = fix_invalid_events(df)\n                                                                                                               \n    for col in [\"Time\", \"AccV\", \"AccAP\"]:\n        df = min_max_feature(df, col)\n    \n    \n    for col in [\"AccV\", \"AccAP\"]:\n        new_col = col + \"_pct\"\n        df[new_col] = df[col].pct_change()\n        df[new_col] = df[new_col].replace([np.inf, -np.inf], np.nan).fillna(method=\"bfill\")\n        df[new_col] = df[new_col].fillna(0).astype(\"float32\") # 0 to 0 returns null\n       \n                        \n        for margin in [5,10, 15, 25, 35, 45, 55, 65, 75, 85, 95, 105]:\n            for w in[5,10]:\n                real_winow= second * w\n                df[f\"moving_std_{new_col}_{w}\"] = df[new_col].rolling(window = real_winow).std().fillna(method=\"bfill\")\n                # in case the file is smaller than w\n                df[f\"moving_std_{new_col}_{w}\"] = df[f\"moving_std_{new_col}_{w}\"].fillna(-1)\n                df = fences_features(df,new_col, margin=margin, width=w, stat=\"std\", second=second)\n                df[f\"moving_std_{new_col}_{w}\"] =  df[f\"moving_std_{new_col}_{w}\"].astype(\"float32\")\n        \n        df[f\"moving_std_all_{col}\"] = df[col].expanding().std().fillna(method=\"bfill\")\n        df = min_max_feature(df, f\"moving_std_all_{col}\")\n        \n        df[f\"moving_mean_all_{col}\"] = df[col].expanding().mean().fillna(method=\"bfill\")\n        df = min_max_feature(df, f\"moving_mean_all_{col}\")\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:06:51.866729Z","iopub.execute_input":"2023-05-30T11:06:51.867088Z","iopub.status.idle":"2023-05-30T11:06:51.905362Z","shell.execute_reply.started":"2023-05-30T11:06:51.867055Z","shell.execute_reply":"2023-05-30T11:06:51.904129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_metadata = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv\")\n# rememver, Visit only relevant for defog\ntdcsfog_metadata[\"Medication\"] = np.where(tdcsfog_metadata[\"Medication\"] == \"on\", 1, 0)\ntdcsfog_subject_dict =  dict(zip(tdcsfog_metadata[\"Id\"], tdcsfog_metadata[\"Subject\"]))\ntdcsfog_medication_dict = dict(zip(tdcsfog_metadata[\"Id\"], tdcsfog_metadata[\"Medication\"]))\ntdcsfog_Id_Visit = dict(zip(tdcsfog_metadata[\"Id\"], tdcsfog_metadata[\"Visit\"]))\ntdcsfog_Id_Test  =dict(zip(tdcsfog_metadata[\"Id\"], tdcsfog_metadata[\"Test\"]))","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:06:51.907535Z","iopub.execute_input":"2023-05-30T11:06:51.908069Z","iopub.status.idle":"2023-05-30T11:06:51.925015Z","shell.execute_reply.started":"2023-05-30T11:06:51.908015Z","shell.execute_reply":"2023-05-30T11:06:51.923510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_metadata = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv\")\nsubjects = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/subjects.csv\")\nsubjects[\"Visit\"] = subjects[\"Visit\"].fillna(1)\nsubjects = subjects.drop_duplicates(subset=[\"Subject\", \"Visit\"])\n\ndefog_metadata[\"Medication\"] = np.where(defog_metadata[\"Medication\"] == \"on\", 1, 0)\ndefog_subject_dict = dict(zip(defog_metadata[\"Id\"], defog_metadata[\"Subject\"]))\ndefog_medication_dict = dict(zip(defog_metadata[\"Id\"], defog_metadata[\"Medication\"]))\ndefog_Id_Visit = dict(zip(defog_metadata[\"Id\"], defog_metadata[\"Visit\"]))\ndefog_Id_Test  =dict(zip(defog_metadata[\"Id\"], np.zeros(defog_metadata.shape[0])))","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:06:51.927324Z","iopub.execute_input":"2023-05-30T11:06:51.928023Z","iopub.status.idle":"2023-05-30T11:06:51.949791Z","shell.execute_reply.started":"2023-05-30T11:06:51.927943Z","shell.execute_reply":"2023-05-30T11:06:51.948668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects[\"UPDRSIII_On\"] = subjects[\"UPDRSIII_On\"].fillna(0)\nsubjects[\"UPDRSIII_Off\"] = subjects[\"UPDRSIII_Off\"].fillna(0)","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:06:51.951885Z","iopub.execute_input":"2023-05-30T11:06:51.952444Z","iopub.status.idle":"2023-05-30T11:06:51.962372Z","shell.execute_reply.started":"2023-05-30T11:06:51.952387Z","shell.execute_reply":"2023-05-30T11:06:51.961268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def flat_outliers(df):\n    for col in ['AccV','AccML','AccV']:\n        max_value = df[col].quantile(q=0.99)\n        min_value = df[col].quantile(q=0.01)\n        df[col] = np.where(df[col] > max_value, max_value, df[col])\n        df[col] = np.where(df[col] < min_value, min_value, df[col])\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:06:51.964430Z","iopub.execute_input":"2023-05-30T11:06:51.964936Z","iopub.status.idle":"2023-05-30T11:06:51.974276Z","shell.execute_reply.started":"2023-05-30T11:06:51.964882Z","shell.execute_reply":"2023-05-30T11:06:51.972547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_target(df):\n    class_dict = {0: \"StartHesitation\", 1: \"Turn\", 2:\"Walking\", 3:\"None\"}\n    df[\"target\"] = 3\n    df[\"target\"] = np.where(df[\"StartHesitation\"] == 1, 0, df[\"target\"] )\n    df[\"target\"] = np.where(df[\"Turn\"] == 1, 1, df[\"target\"] )\n    df[\"target\"] = np.where(df[\"Walking\"] == 1, 2, df[\"target\"] )\n    \n    df = df.drop([\"StartHesitation\", \"Turn\", \"Walking\"], axis = 1)\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:06:51.976296Z","iopub.execute_input":"2023-05-30T11:06:51.977286Z","iopub.status.idle":"2023-05-30T11:06:51.988069Z","shell.execute_reply.started":"2023-05-30T11:06:51.977215Z","shell.execute_reply":"2023-05-30T11:06:51.986829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_df(base,isTest=False, black_list = []):\n    train= pd.DataFrame()\n    if \"tdcsfog\" in base:\n        is_td = True\n\n    else:\n        is_td = False\n    \n    for train_path in tqdm(os.listdir(base)):\n        file_path = base + '/'+train_path\n        df = pd.read_csv(file_path)\n        \n        df = flat_outliers(df)\n        df = FE(df, is_td)\n        \n        df_time = df[\"Time\"].copy()\n        df = df.set_index(\"Time\")\n        df[\"Time\"] = df_time\n        del df_time\n        gc.collect()\n        \n        df = FE_tsflex(df, is_td)\n        \n        if not isTest:\n            df = create_target(df)\n            \n        df[\"file\"] = train_path.split(\".\")[0]\n        df[\"id\"] = df[\"file\"].astype(\"str\") + \"_\" + df[\"Time\"].astype(\"str\")\n        \n        dot_index = train_path.index(\".\")\n        file_id = train_path[:dot_index]\n        \n        if \"tdcsfog\" in base:\n            df[\"Subject\"] = tdcsfog_subject_dict[file_id]\n            df[\"Medication\"] =  tdcsfog_medication_dict[file_id]\n            df[\"Visit\"] = tdcsfog_Id_Visit[file_id]\n            df[\"Test_level\"] =tdcsfog_Id_Test[file_id]\n\n        else:\n            df[\"Subject\"] = defog_subject_dict[file_id]\n            df[\"Medication\"] = defog_medication_dict[file_id]\n            df[\"Visit\"] = defog_Id_Visit[file_id]\n            df[\"Test_level\"] =defog_Id_Test[file_id]\n\n        if train.shape[0] == 0:\n            cur_black_list = [c for c in black_list if c in df.columns]\n            coverted_first = reduce_memory_usage(df.drop(cur_black_list, axis = 1))\n            new_dtypes_dict = coverted_first.dtypes.to_dict()\n            train_cols = [c for c in df.columns if c not in cur_black_list]\n            \n        train = train.append(df[train_cols].astype(new_dtypes_dict))[train_cols]\n        del df\n        gc.collect()\n\n    train = reduce_memory_usage(train)\n    train.reset_index(drop=True, inplace=True)     \n    for col in train.columns:\n        if train[col].dtype != \"object\":\n            if train[col].max() > 99999 or train[col].min()<-9999:\n                train[col]=train[col].replace([np.inf,-np.inf], np.nan).fillna(0)\n    return train","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:06:51.989812Z","iopub.execute_input":"2023-05-30T11:06:51.990194Z","iopub.status.idle":"2023-05-30T11:06:52.007287Z","shell.execute_reply.started":"2023-05-30T11:06:51.990156Z","shell.execute_reply":"2023-05-30T11:06:52.006295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_white_list(models):\n    white_list = set([])\n    for model_name in models: \n        model = models[model_name][0]\n        if \"xgboost\" in model_name :\n            model_cols = model.get_booster().feature_names\n\n        elif \"lgbm\" in model_name:\n            model_cols = model.feature_name_\n\n        elif \"rf\" in model_name or \"adaboost\" in model_name: \n            model_cols =  model.feature_names_in_\n        else:\n            model_cols = model.feature_names_\n\n        white_list = white_list.union(set(model_cols)).union([\"Subject\", \"Visit\",\"id\"])\n        \n    return white_list","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:06:52.011807Z","iopub.execute_input":"2023-05-30T11:06:52.012237Z","iopub.status.idle":"2023-05-30T11:06:52.023107Z","shell.execute_reply.started":"2023-05-30T11:06:52.012188Z","shell.execute_reply":"2023-05-30T11:06:52.022076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"catboost_tsflex_model_dict = pickle.load(open(\"/kaggle/input/make-final-model-full-catboost-v9-tsflex/model_dict_catboost_full.pkl\", \"rb\"))\nxgboost_tsflex_model_dict = pickle.load(open(\"/kaggle/input/make-final-model-full-xgboost-v9-tsflex/model_dict_xgboost_full.pkl\", \"rb\"))\nlgbm_tsflex_model_dict = pickle.load(open(\"/kaggle/input/make-final-model-full-lgbm-v9-tsflex/model_dict_lgbm_full.pkl\", \"rb\"))\nrf_tsflex_model_dict = pickle.load(open(\"/kaggle/input/make-final-model-full-rf-v9-tsflex/model_dict_rf_full.pkl\", \"rb\"))\nadaboost_tsflex_model_dict = pickle.load(open(\"/kaggle/input/make-final-model-full-adaboost-v9-tsflex/model_dict_adaboost_full.pkl\", \"rb\"))\n\ncatboost_model_dict = pickle.load(open(\"/kaggle/input/make-final-model-full-catboost-v9/model_dict_catboost_full.pkl\", \"rb\"))\nxgboost_model_dict = pickle.load(open(\"/kaggle/input/make-final-model-full-xgboost-v9/model_dict_xgboost_full.pkl\", \"rb\"))\nlgbm_model_dict = pickle.load(open(\"/kaggle/input/make-final-model-full-lgbm-v9/model_dict_lgbm_full.pkl\", \"rb\"))\nrf_model_dict = pickle.load(open(\"/kaggle/input/make-final-model-full-rf-v9/model_dict_rf_full.pkl\", \"rb\"))\nadaboost_model_dict = pickle.load(open(\"/kaggle/input/make-final-model-full-adaboost-v9/model_dict_adaboost_full.pkl\", \"rb\"))\n\nwhite_list = pickle.load(open(\"/kaggle/input/black-list-v9/white_list.pkl\",\"rb\"))","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:06:52.024705Z","iopub.execute_input":"2023-05-30T11:06:52.025217Z","iopub.status.idle":"2023-05-30T11:06:52.159211Z","shell.execute_reply.started":"2023-05-30T11:06:52.025174Z","shell.execute_reply":"2023-05-30T11:06:52.157533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"black_list = pickle.load(open(\"/kaggle/input/black-list-v9/black_list.pkl\", \"rb\"))\ntd_test = make_df(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog\",black_list= black_list, isTest=True)\nde_test = make_df(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog\", black_list = black_list,isTest=True)\n\n\nss_de = ss.loc[ss.Id.isin(de_test.id)].reset_index(drop=True)\nss_td = ss.loc[ss.Id.isin(td_test.id)].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:06:52.161174Z","iopub.execute_input":"2023-05-30T11:06:52.161562Z","iopub.status.idle":"2023-05-30T11:07:43.422866Z","shell.execute_reply.started":"2023-05-30T11:06:52.161521Z","shell.execute_reply":"2023-05-30T11:07:43.421304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(td_test.shape)\n# merging only on Subject because Visit is relevant for defog only\ntd_test = td_test.merge(subjects.drop(\"Visit\",axis = 1),\n                                    on=[\"Subject\"], how=\"left\")\nprint(td_test.shape)\n\nprint(de_test.shape)\nde_test = de_test.merge(subjects, on=[\"Subject\",\"Visit\"], how=\"left\")\nprint(de_test.shape)\n\ntd_test['Sex'] = np.where(td_test['Sex'] == \"M\", 1, 0)\nde_test['Sex'] = np.where(de_test['Sex'] == \"M\", 1, 0)\n\ntd_test = reduce_memory_usage(td_test)\nde_test = reduce_memory_usage(de_test)","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:07:43.425000Z","iopub.execute_input":"2023-05-30T11:07:43.425416Z","iopub.status.idle":"2023-05-30T11:07:47.890502Z","shell.execute_reply.started":"2023-05-30T11:07:43.425374Z","shell.execute_reply":"2023-05-30T11:07:47.889076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fix_y(y):\n    return y - 1 # (1,2,3) -> (0,1,2)\n\ndef predict(test, train_cols, model, model_name=\"dummy\", scaler = None):\n    preds = model.predict_proba(test[train_cols])\n\n    preds_df = pd.DataFrame()\n    preds_df[\"Id\"] = test[\"id\"]\n\n\n    preds_df[\"StartHesitation\"] = [p[0] for p in preds]\n    preds_df[\"Turn\"] = [p[1] for p in preds]\n    preds_df[\"Walking\"] = [p[2] for p in preds]\n\n    preds_df.index=test.index\n    return preds_df\n\ndef combine_preds(preds_td, preds_de, ss):\n    ss = ss.drop([\"StartHesitation\", \"Turn\", \"Walking\"], axis = 1)\n    \n    ss = ss.merge(preds_td, on=\"Id\", how=\"left\")\n    ss = ss.merge(preds_de, on=\"Id\", how=\"left\",  suffixes = (\"_td\", \"_de\"))\n    \n    ss[\"StartHesitation\"] = np.where(ss[\"StartHesitation_td\"].isna(),ss[\"StartHesitation_de\"],  ss[\"StartHesitation_td\"])\n    ss[\"Turn\"] = np.where(ss[\"Turn_td\"].isna(),ss[\"Turn_de\"],  ss[\"Turn_td\"])\n    ss[\"Walking\"] = np.where(ss[\"Walking_td\"].isna(),ss[\"Walking_de\"],  ss[\"Walking_td\"])\n    \n    return ss[[\"Id\",\"StartHesitation\", \"Turn\", \"Walking\"]]","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:07:47.892284Z","iopub.execute_input":"2023-05-30T11:07:47.893342Z","iopub.status.idle":"2023-05-30T11:07:47.904832Z","shell.execute_reply.started":"2023-05-30T11:07:47.893294Z","shell.execute_reply":"2023-05-30T11:07:47.903847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# catboost\npreds_td_catboost = predict(td_test, train_cols=catboost_model_dict[\"features\"], model=catboost_model_dict[\"model\"])\npreds_de_catboost = predict(de_test, train_cols=catboost_model_dict[\"features\"], model=catboost_model_dict[\"model\"])\nfinal_ss_catboost = combine_preds(preds_td_catboost, preds_de_catboost, ss.copy())\n\n#xgb\npreds_td_xgboost = predict(td_test, train_cols=xgboost_model_dict[\"features\"], model=xgboost_model_dict[\"model\"])\npreds_de_xgboost = predict(de_test,train_cols=xgboost_model_dict[\"features\"], model=xgboost_model_dict[\"model\"])\nfinal_ss_xgboost = combine_preds(preds_td_xgboost, preds_de_xgboost, ss.copy())\n\n\n#lgbm\npreds_td_lgbm = predict(td_test, train_cols=lgbm_model_dict[\"features\"], model=lgbm_model_dict[\"model\"])\npreds_de_lgbm = predict(de_test, train_cols=lgbm_model_dict[\"features\"], model = lgbm_model_dict[\"model\"])\nfinal_ss_lgbm = combine_preds(preds_td_lgbm, preds_de_lgbm, ss.copy())\n\n\n#rf\npreds_td_rf = predict(td_test, train_cols=rf_model_dict[\"features\"], model=rf_model_dict[\"model\"])\npreds_de_rf = predict(de_test,train_cols=rf_model_dict[\"features\"], model=rf_model_dict[\"model\"])\nfinal_ss_rf = combine_preds(preds_td_rf, preds_de_rf, ss.copy())\n\n#adaboost\npreds_td_adaboost = predict(td_test,  train_cols=adaboost_model_dict[\"features\"], model=adaboost_model_dict[\"model\"])\npreds_de_adaboost = predict(de_test, train_cols=adaboost_model_dict[\"features\"], model=adaboost_model_dict[\"model\"])\nfinal_ss_adaboost = combine_preds(preds_td_adaboost, preds_de_adaboost, ss.copy())","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:07:47.925410Z","iopub.execute_input":"2023-05-30T11:07:47.925738Z","iopub.status.idle":"2023-05-30T11:08:29.617504Z","shell.execute_reply.started":"2023-05-30T11:07:47.925706Z","shell.execute_reply":"2023-05-30T11:08:29.616155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# catboost\n\npreds_td_catboost_tsflex = predict(td_test, train_cols=catboost_tsflex_model_dict[\"features\"], model=catboost_tsflex_model_dict[\"model\"])\npreds_de_catboost_tsflex = predict(de_test, train_cols=catboost_tsflex_model_dict[\"features\"], model=catboost_tsflex_model_dict[\"model\"])\nfinal_ss_tsflex_catboost = combine_preds(preds_td_catboost_tsflex, preds_de_catboost_tsflex, ss.copy())\n\n#xgb\npreds_td_xgboost_tsflex = predict(td_test, train_cols=xgboost_tsflex_model_dict[\"features\"], model=xgboost_tsflex_model_dict[\"model\"])\npreds_de_xgboost_tsflex = predict(de_test,train_cols=xgboost_tsflex_model_dict[\"features\"], model=xgboost_tsflex_model_dict[\"model\"])\nfinal_ss_tsflex_xgboost = combine_preds(preds_td_xgboost_tsflex, preds_de_xgboost_tsflex, ss.copy())\n\n\n#lgbm\npreds_td_lgbm_tsflex = predict(td_test, train_cols=lgbm_tsflex_model_dict[\"features\"], model=lgbm_tsflex_model_dict[\"model\"])\npreds_de_lgbm_tsflex = predict(de_test, train_cols=lgbm_tsflex_model_dict[\"features\"], model = lgbm_tsflex_model_dict[\"model\"])\nfinal_ss_tsflex_lgbm = combine_preds(preds_td_lgbm_tsflex, preds_de_lgbm_tsflex, ss.copy())\n\n\n#rf\npreds_td_rf_tsflex = predict(td_test, train_cols=rf_tsflex_model_dict[\"features\"], model=rf_tsflex_model_dict[\"model\"])\npreds_de_rf_tsflex = predict(de_test,train_cols=rf_tsflex_model_dict[\"features\"], model=rf_tsflex_model_dict[\"model\"])\nfinal_ss_tsflex_rf = combine_preds(preds_td_rf_tsflex, preds_de_rf_tsflex, ss.copy())\n\n#adaboost\npreds_td_adaboost_tsflex = predict(td_test,  train_cols=adaboost_tsflex_model_dict[\"features\"], model=adaboost_tsflex_model_dict[\"model\"])\npreds_de_adaboost_tsflex = predict(de_test, train_cols=adaboost_tsflex_model_dict[\"features\"], model=adaboost_tsflex_model_dict[\"model\"])\nfinal_ss_tsflex_adaboost = combine_preds(preds_td_adaboost_tsflex, preds_de_adaboost_tsflex, ss.copy())","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:08:29.619182Z","iopub.execute_input":"2023-05-30T11:08:29.619563Z","iopub.status.idle":"2023-05-30T11:08:53.485478Z","shell.execute_reply.started":"2023-05-30T11:08:29.619525Z","shell.execute_reply":"2023-05-30T11:08:53.484204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"finals = {\"final_ss_catboost\": final_ss_catboost, \"final_ss_xgboost\": final_ss_xgboost,\n         \"final_ss_lgbm\": final_ss_lgbm,\"final_ss_rf\":final_ss_rf, 'final_ss_adaboost':final_ss_adaboost,\n        \"final_ss_tsflex_catboost\":final_ss_tsflex_catboost, \"final_ss_tsflex_xgboost\":final_ss_tsflex_xgboost,\n          \"final_ss_tsflex_lgbm\":final_ss_tsflex_lgbm, \"final_ss_tsflex_rf\":final_ss_tsflex_rf,\n             \"final_ss_tsflex_adaboost\":final_ss_tsflex_adaboost }\n # from baging lab\nweights = {'final_ss_rf': 0.3, 'final_ss_lgbm': 2,\n           'final_ss_xgboost': 3, 'final_ss_catboost': 3, 'final_ss_adaboost': 0.0654058370958252,\n           'final_ss_tsflex_rf': 0.3, 'final_ss_tsflex_lgbm': 2, \n           'final_ss_tsflex_xgboost': 3.5, 'final_ss_tsflex_catboost': 3, \n           'final_ss_tsflex_adaboost': 0.7}\n","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:08:53.487398Z","iopub.execute_input":"2023-05-30T11:08:53.487857Z","iopub.status.idle":"2023-05-30T11:08:53.495057Z","shell.execute_reply.started":"2023-05-30T11:08:53.487805Z","shell.execute_reply":"2023-05-30T11:08:53.493791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"c = 0\ntotal_models = 0\nto_sub=pd.DataFrame()\nfor final_name in finals:\n    final = finals[final_name]\n    sh_cols = [c for c in final.columns if \"StartH\" in c]\n    turn_cols= [c for c in final.columns if \"Turn\" in c]\n    walking_cols= [c for c in final.columns if \"Walking\" in c]\n    \n    total_models += len(sh_cols)\n    if c== 0:\n        to_sub[\"StartHesitation\"]=final[sh_cols].sum(axis=1) * weights[final_name]\n        to_sub[\"Turn\"]=final[turn_cols].sum(axis=1) * weights[final_name]\n        to_sub[\"Walking\"]=final[walking_cols].sum(axis=1) * weights[final_name]\n    \n    else:\n        to_sub[\"StartHesitation\"] += final[sh_cols].sum(axis=1) * weights[final_name]\n        to_sub[\"Turn\"] += final[turn_cols].sum(axis=1) * weights[final_name]\n        to_sub[\"Walking\"] += final[walking_cols].sum(axis=1) * weights[final_name]\n    c += 1\n\nto_sub = to_sub / total_models\nto_sub.insert(0, \"Id\",final_ss_xgboost[\"Id\"])","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:08:53.496809Z","iopub.execute_input":"2023-05-30T11:08:53.497177Z","iopub.status.idle":"2023-05-30T11:08:53.747286Z","shell.execute_reply.started":"2023-05-30T11:08:53.497140Z","shell.execute_reply":"2023-05-30T11:08:53.745953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to_sub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-30T11:08:53.749056Z","iopub.execute_input":"2023-05-30T11:08:53.749433Z","iopub.status.idle":"2023-05-30T11:08:54.386324Z","shell.execute_reply.started":"2023-05-30T11:08:53.749396Z","shell.execute_reply":"2023-05-30T11:08:54.385324Z"},"trusted":true},"execution_count":null,"outputs":[]}]}