{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport random\nSEED = 12\nnp.random.seed(SEED)\nrandom.seed(SEED)\nimport pandas as pd\n\nimport matplotlib.pyplot as plt\n\nimport os\nimport gc\nimport glob\n\nfrom sklearn.model_selection import StratifiedGroupKFold\n\nfrom tqdm.notebook import tqdm\n\ndef extract_id(x: str) -> str:\n    return os.path.basename(x).split(\".\")[0]","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-25T08:33:13.027603Z","iopub.execute_input":"2023-05-25T08:33:13.028051Z","iopub.status.idle":"2023-05-25T08:33:14.530729Z","shell.execute_reply.started":"2023-05-25T08:33:13.028017Z","shell.execute_reply":"2023-05-25T08:33:14.529613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_DIR = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction\"\n\npaths = {\n    \"train_tdcsfog\": glob.glob(os.path.join(BASE_DIR, \"train\", \"tdcsfog\", \"*.csv\")),\n    \"train_defog\": glob.glob(os.path.join(BASE_DIR, \"train\", \"defog\", \"*.csv\")),\n    \"test_tdcsfog\": glob.glob(os.path.join(BASE_DIR, \"test\", \"tdcsfog\", \"*.csv\")),\n    \"test_defog\": glob.glob(os.path.join(BASE_DIR, \"test\", \"defog\", \"*.csv\")),\n    \"meta_defog\": os.path.join(BASE_DIR, \"defog_metadata.csv\"),\n    \"meta_tdcsfog\": os.path.join(BASE_DIR, \"tdcsfog_metadata.csv\"),\n    \"subjects\": os.path.join(BASE_DIR, \"subjects.csv\"),\n}\n\ntf_meta = pd.read_csv(paths[\"meta_tdcsfog\"])\ndf_meta = pd.read_csv(paths['meta_defog'])\n\n\nsubjects = pd.read_csv(paths[\"subjects\"])\n\ntest_id = list(map(extract_id, paths[\"test_tdcsfog\"]))\ntrain_id = list(map(extract_id, paths[\"train_tdcsfog\"]))\n\ntest_id_df = list(map(extract_id, paths[\"test_defog\"]))\ntrain_id_df = list(map(extract_id, paths[\"train_defog\"]))","metadata":{"execution":{"iopub.status.busy":"2023-05-25T08:33:14.533007Z","iopub.execute_input":"2023-05-25T08:33:14.534100Z","iopub.status.idle":"2023-05-25T08:33:14.686129Z","shell.execute_reply.started":"2023-05-25T08:33:14.534056Z","shell.execute_reply":"2023-05-25T08:33:14.685024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare meta","metadata":{}},{"cell_type":"code","source":"def prepare_meta(df_meta, tf_meta, subjects):\n    subjects_meta = subjects.groupby(\"Subject\", as_index=False)[[\"Sex\", \"Age\"]].max()\n    \n    subjects_meta[\"Sex\"] = (subjects_meta[\"Sex\"] == \"M\").astype(\"int64\")\n    \n    tf_meta1 = (\n        tf_meta.merge(subjects_meta, how=\"left\", on=\"Subject\")\n        .assign(Medication=lambda x: (x[\"Medication\"] == \"on\").astype(\"int64\")))\n    \n    df_meta1 = (\n        df_meta.merge(subjects_meta, how=\"left\", on=\"Subject\")\n        .assign(Medication=lambda x: (x[\"Medication\"] == \"on\").astype(\"int64\"))\n    \n    )\n    \n    df_meta1[\"Test\"] = -1\n    \n    return (\n        pd.concat([tf_meta1, df_meta1], ignore_index=True))\n    \nmeta = prepare_meta(df_meta, tf_meta, subjects)","metadata":{"execution":{"iopub.status.busy":"2023-05-25T08:33:14.689952Z","iopub.execute_input":"2023-05-25T08:33:14.691052Z","iopub.status.idle":"2023-05-25T08:33:14.752920Z","shell.execute_reply.started":"2023-05-25T08:33:14.691006Z","shell.execute_reply":"2023-05-25T08:33:14.751974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Scaling metadata","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nimport joblib\nmeta_scaler = StandardScaler()\n\nmeta_index = [\"Id\", \"Subject\"]\nmeta_features = ['Age', 'Medication', 'Sex', 'Test', 'Visit']\n\nmeta_scaled_features = meta_scaler.fit_transform(meta.drop(meta_index, axis=1).values)\n\nmeta_scaled = meta[meta_index].join(pd.DataFrame(meta_scaled_features, columns=meta_features))\n\njoblib.dump(meta_scaler, \"meta_scaler.sav\")","metadata":{"execution":{"iopub.status.busy":"2023-05-25T08:33:14.756963Z","iopub.execute_input":"2023-05-25T08:33:14.757937Z","iopub.status.idle":"2023-05-25T08:33:14.778812Z","shell.execute_reply.started":"2023-05-25T08:33:14.757874Z","shell.execute_reply":"2023-05-25T08:33:14.777478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_scaled.to_parquet(\"meta.parquet\")","metadata":{"execution":{"iopub.status.busy":"2023-05-25T08:33:14.780422Z","iopub.execute_input":"2023-05-25T08:33:14.781003Z","iopub.status.idle":"2023-05-25T08:33:15.026873Z","shell.execute_reply.started":"2023-05-25T08:33:14.780961Z","shell.execute_reply":"2023-05-25T08:33:15.025960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Define data transformations","metadata":{}},{"cell_type":"code","source":"def rolling_agg(\n    dt: pd.DataFrame, step: int, \n    aggfunc: str, cols: list, back: bool = False) -> pd.DataFrame:\n    \n    if back:\n        rolling = dt[cols][::-1].rolling(step, min_periods=0)\n        suffix = f\"_back_rolling_{step}_{aggfunc}\"\n    else:\n        rolling = dt[cols].rolling(step, min_periods=0)\n        suffix = f\"_rolling_{step}_{aggfunc}\"\n        \n    if aggfunc.startswith(\"quantile\"):\n        quantile = int(aggfunc.split(\"_\")[1]) / 100\n        \n        return (\n            rolling.quantile(quantile)\n            .add_suffix(suffix))\n    else:\n        return (\n            rolling.agg(aggfunc)\n            .add_suffix(suffix))\n\n\ndef create_dataset(data, defog=False, verbose=False):\n#     dt = data.merge(meta, how=\"left\", on=\"Id\")\n    \n    cols = ['AccV', 'AccML', 'AccAP']\n    dt = data.copy()\n    \n    if not defog:\n        data[cols] = data[cols] / 9.80665\n    dt[\"defog\"] = int(defog)\n    if verbose: print(\"Global stats\")\n    for aggfunc in [\"mean\", \"max\", \"min\", \"std\", \"median\"]:\n        dt = dt.join(\n            dt[cols].groupby(dt.assign(dummy=1).dummy)\n            .transform(aggfunc).add_suffix(f\"_{aggfunc}\")\n        )\n    \n    step1 = 500\n    step2 = 100\n    if verbose: print(\"Shifts stats\")\n    for shift in [1, 2, -1, -2]:\n        \n        if shift > 0:\n            suffix_name = f\"_lag_{shift}\"\n            fill_data = dt[cols].iloc[: shift]\n        else:\n            suffix_name = f\"_lead_{abs(shift)}\"\n            fill_data = dt[cols].iloc[shift:]\n        \n        dt = dt.join(\n            dt[cols]\n            .shift(shift)\n            .fillna(fill_data)\n            .add_suffix(suffix_name)\n        )\n    \n    aggfuncs = [\n        \"mean\", \"std\", \"max\", \"min\", \"median\", \n        \"quantile_75\", \"quantile_25\", \"quantile_99\"\n    ]\n    if verbose: print(\"Rolling stats, step 1\")\n\n    for aggfunc in aggfuncs:\n        \n        dt = dt.join(\n            rolling_agg(dt, step1, aggfunc, cols)\n        )\n    \n    if verbose: print(\"Rolling stats, step 2\")\n\n    for aggfunc in aggfuncs:\n        funcname = aggfunc if isinstance(aggfunc, str) else aggfunc.__name__\n        dt = dt.join(\n            rolling_agg(dt, step2, aggfunc, cols)\n        )\n    if verbose: print(\"Back Rolling stats, step 1\")\n\n    for aggfunc in aggfuncs:\n        dt = dt.join(\n            rolling_agg(dt, step1, aggfunc, cols, back=True)\n        )\n    if verbose: print(\"Back Rolling stats, step 2\")\n\n    for aggfunc in aggfuncs:\n        dt = dt.join(\n            rolling_agg(dt[::-1], step2, aggfunc, cols, back=True)\n        )\n    if verbose: print(\"Calculating diffs\")\n\n    diff = dt[cols].transform(\"diff\").add_suffix(\"_diff\")\n    diff = diff.fillna(diff.iloc[0])\n    cols = [\"AccV_diff\", \"AccML_diff\", \"AccAP_diff\"]\n    if verbose: print(\"Diff rolling stat step 1\")\n    \n    for aggfunc in aggfuncs:\n        dt = dt.join(\n            rolling_agg(diff, step1, aggfunc, cols)\n        )\n        \n    if verbose: print(\"Diff rolling stat step 2\")\n    for aggfunc in aggfuncs:\n        dt = dt.join(\n            rolling_agg(diff, step2, aggfunc, cols)\n        )\n        \n    if verbose: print(\"Back Diff rolling stat step 1\")\n\n    for aggfunc in aggfuncs:\n        dt = dt.join(\n            rolling_agg(diff, step1, aggfunc, cols, back=True)\n        )\n    \n    if verbose: print(\"Back Diff rolling stat step 2\")\n\n    for aggfunc in aggfuncs:\n        dt = dt.join(\n            rolling_agg(diff, step2, aggfunc, cols, back=True)\n        )\n    \n    if verbose: print(\"Sign change\")\n    sign_change = (\n        diff.apply(np.sign)\n        .transform(\"diff\")\n        .apply(np.abs)\n        .divide(2)\n        .fillna(0)\n        .add_suffix(\"_sc\")\n    )\n    \n    cols = [\"AccV_diff_sc\", \"AccML_diff_sc\", \"AccAP_diff_sc\"]\n    aggfuncs = [\"mean\"]\n    if verbose: print(\"Sign change rolling stat step 1\")\n\n    for aggfunc in aggfuncs:\n        dt = dt.join(\n            rolling_agg(sign_change, step1, aggfunc, cols)\n        )\n    if verbose: print(\"Sign change rolling stat step 2\")\n\n    for aggfunc in aggfuncs:\n        dt = dt.join(\n            rolling_agg(sign_change, step2, aggfunc, cols)\n        )\n    if verbose: print(\"Back Sign change rolling stat step 1\")\n\n    for aggfunc in aggfuncs:\n        dt = dt.join(\n            rolling_agg(sign_change, step1, aggfunc, cols, back=True)\n        )\n    if verbose: print(\"Back Sign change rolling stat step 2\")\n\n        \n    for aggfunc in aggfuncs:\n        dt = dt.join(\n            rolling_agg(sign_change, step2, aggfunc, cols, back=True)\n        )\n        \n    if verbose: print(\"time spent\")\n    \n    dt[\"time_spent\"] = dt.Time.divide(dt.Time.max())\n    return dt.drop(\"Time\", axis=1).fillna(0)","metadata":{"execution":{"iopub.status.busy":"2023-05-25T08:33:15.029264Z","iopub.execute_input":"2023-05-25T08:33:15.029717Z","iopub.status.idle":"2023-05-25T08:33:15.061415Z","shell.execute_reply.started":"2023-05-25T08:33:15.029676Z","shell.execute_reply":"2023-05-25T08:33:15.060171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Learn scale data","metadata":{}},{"cell_type":"code","source":"paths_train = paths[\"train_tdcsfog\"] + paths[\"train_defog\"]","metadata":{"execution":{"iopub.status.busy":"2023-05-25T08:34:49.002624Z","iopub.execute_input":"2023-05-25T08:34:49.003262Z","iopub.status.idle":"2023-05-25T08:34:49.009291Z","shell.execute_reply.started":"2023-05-25T08:34:49.003217Z","shell.execute_reply":"2023-05-25T08:34:49.008024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"OUT_DIR = \"out_data\"\n\nif not os.path.exists(OUT_DIR):\n    os.mkdir(OUT_DIR)","metadata":{"execution":{"iopub.status.busy":"2023-05-25T08:34:50.188586Z","iopub.execute_input":"2023-05-25T08:34:50.189030Z","iopub.status.idle":"2023-05-25T08:34:50.195247Z","shell.execute_reply.started":"2023-05-25T08:34:50.188987Z","shell.execute_reply":"2023-05-25T08:34:50.194178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask_columns = [\"Valid\", \"Task\"]\ntargets = ['StartHesitation', 'Turn', 'Walking']\n\nfrom tqdm import tqdm\ndata_scaler = StandardScaler()\n\nfor path in tqdm(paths_train):\n#     print(\"Processing\", path)\n    defog = \"defog\" in path\n    data = pd.read_csv(path)\n    proc = create_dataset(data, defog=defog)\n    if \"StartHesitation\" in data.columns:\n        df_target = data[targets].copy()\n        proc = proc.drop(targets, axis=1)\n        df_target[\"NoEvent\"] = (~df_target[targets].any(axis=1)).astype(\"int\")\n        \n    Id = extract_id(path)\n    \n    if defog:\n        mask = proc[mask_columns]\n        proc = proc.drop(mask_columns, axis=1)\n    \n    data_scaler.partial_fit(proc.values)\n    \n    data_proc_meta = (\n        proc.assign(Id=Id)\n        .merge(meta_scaled.drop(\"Subject\", axis=1), how=\"inner\", on=\"Id\")\n        .drop(\"Id\", axis=1)\n    )\n    \n    if defog:\n        data_proc_meta = data_proc_meta.join(mask)\n    \n    data_proc_meta = data_proc_meta.join(df_target)\n    \n    out_path = os.path.join(OUT_DIR, f\"{Id}.feather\")\n    \n    data_proc_meta.to_feather(out_path)","metadata":{"execution":{"iopub.status.busy":"2023-05-25T08:34:50.537401Z","iopub.execute_input":"2023-05-25T08:34:50.537845Z","iopub.status.idle":"2023-05-25T08:35:06.969777Z","shell.execute_reply.started":"2023-05-25T08:34:50.537810Z","shell.execute_reply":"2023-05-25T08:35:06.968159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = proc.columns\nnon_features = (\n    data_proc_meta.columns.difference(features).difference(mask_columns)\n    .difference(targets + [\"NoEvent\"]))","metadata":{"execution":{"iopub.status.busy":"2023-05-25T08:35:43.273236Z","iopub.execute_input":"2023-05-25T08:35:43.273665Z","iopub.status.idle":"2023-05-25T08:35:43.281986Z","shell.execute_reply.started":"2023-05-25T08:35:43.273635Z","shell.execute_reply":"2023-05-25T08:35:43.280716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\nwith open(\"features.json\", \"w\", encoding=\"utf-8\") as file:\n    json.dump(features.tolist(), file)\n\nwith open(\"non_features.json\", \"w\", encoding=\"utf-8\") as file:\n    json.dump(non_features.tolist(), file)","metadata":{"execution":{"iopub.status.busy":"2023-05-25T08:35:45.326135Z","iopub.execute_input":"2023-05-25T08:35:45.326523Z","iopub.status.idle":"2023-05-25T08:35:45.334212Z","shell.execute_reply.started":"2023-05-25T08:35:45.326493Z","shell.execute_reply":"2023-05-25T08:35:45.332806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import joblib\njoblib.dump(data_scaler, \"data_scaler.sav\")","metadata":{"execution":{"iopub.status.busy":"2023-05-25T08:35:46.873790Z","iopub.execute_input":"2023-05-25T08:35:46.874206Z","iopub.status.idle":"2023-05-25T08:35:46.881759Z","shell.execute_reply.started":"2023-05-25T08:35:46.874175Z","shell.execute_reply":"2023-05-25T08:35:46.880973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Scale and save data","metadata":{}},{"cell_type":"code","source":"out_paths = glob.glob(os.path.join(OUT_DIR, \"*.feather\"))","metadata":{"execution":{"iopub.status.busy":"2023-05-25T09:03:19.589619Z","iopub.execute_input":"2023-05-25T09:03:19.590095Z","iopub.status.idle":"2023-05-25T09:03:19.596801Z","shell.execute_reply.started":"2023-05-25T09:03:19.590056Z","shell.execute_reply":"2023-05-25T09:03:19.595442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask_columns = [\"Valid\", \"Task\"]\ntargets = ['StartHesitation', 'Turn', 'Walking', \"NoEvent\"]\nfor path in tqdm(out_paths):\n#     print(\"Processing 1\", path)\n    defog = \"defog\" in path\n    data = pd.read_feather(path)\n    \n    proc_scaled = pd.DataFrame(\n        data_scaler.transform(data[features].values), columns=features)\n    \n    result = data[non_features].join(proc_scaled)\n    \n    if \"Valid\" in data.columns:\n        result = result.join(data[mask_columns])\n        \n    if \"StartHesitation\" in data.columns:\n        result = result.join(data[targets])\n        \n    shape0 = result.shape[0]\n\n    if shape0 % 10000 != 0:\n        new_shape = ((shape0 // 10000) + 1) * 10000\n\n    data_reindex = result.reindex(range(new_shape))\n\n    if \"Valid\" in result.columns:\n        data_reindex = data_reindex.fillna({\n            \"Valid\": False,\n            \"Task\": False\n        }).fillna(0)\n\n        data_reindex[\"mask\"] = data_reindex[\"Valid\"] & data_reindex[\"Task\"]\n        data_reindex = data_reindex.drop(mask_columns, axis=1)\n    else:\n        data_reindex[\"mask\"] = data_reindex.defog.notna()\n        data_reindex = data_reindex.fillna(0)\n\n    num = new_shape // 10000\n\n    split = [\n        data_reindex[i * 10000: (i + 1) * 10000].reset_index(drop=True) for i in range(num)\n    ]\n    \n    os.remove(path)\n    path_part1 = path.split(\".\")[0]\n    for i, item in enumerate(split):\n        new_path = f\"{path_part1}_{i}.feather\"\n        item.to_feather(new_path)","metadata":{"execution":{"iopub.status.busy":"2023-05-25T09:03:19.751762Z","iopub.execute_input":"2023-05-25T09:03:19.753069Z","iopub.status.idle":"2023-05-25T09:03:27.473159Z","shell.execute_reply.started":"2023-05-25T09:03:19.753019Z","shell.execute_reply":"2023-05-25T09:03:27.468367Z"},"trusted":true},"execution_count":null,"outputs":[]}]}