{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Dataset visualization with UMAP","metadata":{}},{"cell_type":"markdown","source":"I'm exploring the usage of UMAP for visualizing high-dimensional datasets.\n\nIn this particular case we have exactly 3 columns so is possible to compare the original dataset and the transformed one.\nFor both I plotted the color of each point based on the associated event (no event, StartHesitation, Turn, Walking).\n\nNote that I'm using de and tdcs datasets together.\n\nI was quite surprised that the transformed dataset is so different from the original one.\n\nVisually, the classes seem much more separable in the transformed dataset.\nDo you have a guess on why this is the case?","metadata":{}},{"cell_type":"code","source":"import time\nfrom typing import List, Iterable, Callable\nimport pandas as pd, numpy as np\nfrom joblib import Parallel, parallel_backend, delayed\nimport os\n\nfrom tqdm import tqdm\nimport umap\nimport plotly.express as px\nimport plotly.graph_objects as go\nimport joblib\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import make_pipeline, Pipeline\nfrom xgboost.sklearn import XGBClassifier","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-25T11:30:51.033606Z","iopub.execute_input":"2023-05-25T11:30:51.034050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Helper functions","metadata":{}},{"cell_type":"code","source":"def get_files_in_folder(folder:str) -> List[str]:\n    return list(os.walk(folder))[0][2]\n\n\ndef get_ids(train_or_test:str, de_or_tdcs:str) -> List[str]:\n    this_folder = f\"{path_to_data}/{train_or_test}/{de_or_tdcs}fog\"\n    files = get_files_in_folder(this_folder)\n    ts_ids = [f.replace('.csv', '') for f in files]\n    return ts_ids\n\n\ndef get_dataset_stats(de_or_tdcs:str) -> pd.DataFrame:\n    ts_ids = get_ids(train_or_test='train', de_or_tdcs=de_or_tdcs)\n    dataset = load_timeseries(ts_ids, n_jobs=-1)\n    out = dataset.groupby('Id')[['StartHesitation', 'Turn', 'Walking']].sum()\n    out = out.join(dataset.groupby('Id')['StartHesitation'].agg(count='count'))\n    out = out.assign(no_events=out['count'] - out['StartHesitation'] - out['Turn'] - out['Walking'])\n    out = out[['no_events'] + out.columns.drop('no_events').to_list()]\n    # events mean duration\n    for event in ['StartHesitation', 'Turn', 'Walking']:\n        a = (dataset[event] != dataset[event].shift()).cumsum()\n        a = a[dataset[event] == 1].to_frame()\n        a = a.reset_index(level='Time').groupby(['Id', event])['Time'].count()\n        a = a.rename(f\"{event}_mean_n_samples\").to_frame()\n        a = a.groupby('Id')[f\"{event}_mean_n_samples\"].mean().to_frame()\n        out = out.join(a, how='outer')\n    return out\n\n\ndef find(ts_id: str) -> (str, str):\n    if ts_id in get_ids(train_or_test='train', de_or_tdcs='de'):\n        train_or_test, de_or_tdcs = 'train', 'de'\n    elif ts_id in get_ids(train_or_test='train', de_or_tdcs='tdcs'):\n        train_or_test, de_or_tdcs = 'train', 'tdcs'\n    elif ts_id in get_ids(train_or_test='test', de_or_tdcs='de'):\n        train_or_test, de_or_tdcs = 'test', 'de'\n    elif ts_id in get_ids(train_or_test='test', de_or_tdcs='tdcs'):\n        train_or_test, de_or_tdcs = 'test', 'tdcs'\n    else:\n        raise Exception(f'{ts_id} not found!')\n    return train_or_test, de_or_tdcs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_ids_with_events(de_or_tdcs:str, max_noevents_fraction:float=1) -> np.ndarray:\n    assert 0 < max_noevents_fraction <= 1  # 1 equals all ids\n    stats = get_dataset_stats(de_or_tdcs)\n    stats['no_events_fraction'] = stats['no_events'] / stats['count']\n    with_events = stats[stats['no_events_fraction'] <= max_noevents_fraction]\n    farc_events = with_events[['no_events'] + target_colnames].sum() / with_events['count'].sum()\n    return with_events.index.values","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_timeseries(ts_ids:Iterable[str], n_jobs:int=4) -> pd.DataFrame:\n    with parallel_backend('loky'):\n        ts = pd.concat(\n            Parallel(n_jobs=n_jobs)(delayed(load_single_timeseries)(ts_id) for ts_id in ts_ids)\n        )\n    return ts\n\n\ndef load_single_timeseries(ts_id:str) -> pd.DataFrame:\n    assert '.csv' not in ts_id\n    train_or_test, de_or_tdcs = find(ts_id)\n    path_to_timeseries = f\"{path_to_data}/{train_or_test}/{de_or_tdcs}fog/{ts_id}.csv\"\n    df = pd.read_csv(path_to_timeseries)\n    df['Id'] = ts_id\n    df = df.set_index('Id')\n    df = df.set_index('Time', append=True)\n    #df = df.set_index('time_seconds', append=True)\n    if train_or_test == 'train' and de_or_tdcs == 'de':\n        df = df[(df['Valid']) & df['Task']]\n        df = df.drop(columns=['Valid', 'Task'])\n    if de_or_tdcs == 'tdcs':\n        df.loc[:, ['AccV', 'AccML', 'AccAP']] = df[['AccV', 'AccML', 'AccAP']] / G_MS2  # Convert to G units (no scaling needed?)\n    return df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sample_Xy(Xy:pd.DataFrame, n:int):\n    Xy_out = pd.concat([\n        Xy[Xy['StartHesitation'].eq(0) & Xy['Turn'].eq(0) & Xy['Walking'].eq(0)].sample(n=n),\n        Xy[Xy['StartHesitation'].eq(1)].sample(n=min(len(Xy[Xy['StartHesitation'].eq(1)]), n)),\n        Xy[Xy['Turn'].eq(1)].sample(n=min(len(Xy[Xy['Turn'].eq(1)]), n)),\n        Xy[Xy['Walking'].eq(1)].sample(n=min(len(Xy[Xy['Walking'].eq(1)]), n)),\n    ]).sort_index()\n    return Xy_out\n\n\ndef get_Xy() -> (pd.DataFrame, pd.DataFrame):\n    Xy = pd.DataFrame()\n    for de_or_tdcs in ['de', 'tdcs']:\n        ts_ids = get_ids_with_events(de_or_tdcs, max_noevent_fraction)\n        Xy_this = load_timeseries(ts_ids)\n        Xy_this = sample_Xy(Xy_this, n=n_samples[de_or_tdcs])\n        Xy = pd.concat([Xy, Xy_this])\n    return Xy\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot(X, y):\n    fig = go.Figure()\n    for target in ['no_event'] + target_colnames:\n        if target == 'no_event':\n            xyz = X[y['StartHesitation'].eq(0) & y['Turn'].eq(0) & y['Walking'].eq(0)]\n        else:\n            xyz = X[y[target].eq(1)]\n        fig.add_trace(\n            go.Scatter3d(\n                x=xyz.iloc[:, 0], y=xyz.iloc[:, 1], z=xyz.iloc[:, 2], mode='markers', marker=dict(size=2),\n                name=target\n            ),\n        )\n    fig.update_layout(\n        scene=dict(\n            xaxis_title=xyz.columns[0],\n            yaxis_title=xyz.columns[1],\n            zaxis_title=xyz.columns[2],\n        ),\n    )\n    return fig","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Preparation","metadata":{}},{"cell_type":"code","source":"path_to_data = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction\"\nacc_colnames = ['AccV', 'AccML', 'AccAP']\ntarget_colnames = ['StartHesitation', 'Turn', 'Walking']\ntdcs_sampling = 128  # Hz\nde_sampling = 100  # Hz\ntdcs_dt       = 1/tdcs_sampling # s\nde_dt         = 1/de_sampling # s\nG_MS2 = 9.8066","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_samples = {'de': 5_000, 'tdcs': 10_000}\nfrac_neigh_samples = 1/10\nmax_noevent_fraction = .9","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_neigh = int((n_samples['de'] + n_samples['tdcs']) * frac_neigh_samples)\nreducer = umap.UMAP(n_neighbors=n_neigh, n_components=3, n_jobs=-1)\n\nXy = get_Xy()\nX, y = Xy[acc_colnames], Xy[target_colnames]\n# TODO: scaler?\nreducer = reducer.fit(X)\nprint('umap fitted')\nprint(f\"classes after sampling: {y.sum(axis=0)}\")\n\nX_red = pd.DataFrame(index=X.index, data=reducer.transform(X))\nprint('dimensions reduced')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualization of the original dataset","metadata":{}},{"cell_type":"code","source":"fig = plot(X, y)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualization of the transformed dataset","metadata":{}},{"cell_type":"code","source":"fig = plot(X_red, y)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}