{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# time-series 🤝 tsflex 🚀","metadata":{}},{"cell_type":"markdown","source":"### **Thanks to great works from JEROENVDD.**\n\n### improvements\n\n1. add metadata infomation and subject infomation, fix a little bug of subject feature in JEROENVDD‘s origin notebook\n2. make GroupKfold Cross Validation\n\n#### reference\n\n* @JEROENVDD\n    * Origin notebook link https://www.kaggle.com/code/jeroenvdd/time-series-tsflex\n    \n<br>","metadata":{}},{"cell_type":"markdown","source":"<div style=\"background-color:#f2f2f2; padding:20px; border-radius: 10px;\">\n    <h2 style=\"color:#595959;\">Check out <a href=\"https://github.com/predict-idlab/tsflex\" target=\"_blank\" style=\"color:#0099cc;\">tsflex</a>!</h2>\n    <h4 style=\"color:#737373;\">tsflex is a Python package for flexible and efficient time series feature extraction. It's great for data preprocessing and feature engineering for time series data. Check it out on <a href=\"https://github.com/predict-idlab/tsflex\" target=\"_blank\" style=\"color:#0099cc;\">GitHub</a> today!</h4>\n    \n<p style=\"color:#737373;\">This notebook is a fork of the <a href=\"https://www.kaggle.com/code/jazivxt/familiar-solvs\" target=\"_blank\" style=\"color:#0099cc;\">familiar-solvs notebook of jazivxt</a> and adds tsflex to extract some basic <a href=\"https://github.com/dmbee/seglearn\" target=\"_blank\" style=\"color:#0099cc;\">seglearn</a> features.</p>\n    \n</div>","metadata":{}},{"cell_type":"code","source":"# Install tsflex and seglearn\n!pip install tsflex --no-index --find-links=file:///kaggle/input/time-series-tools\n!pip install seglearn --no-index --find-links=file:///kaggle/input/time-series-tools","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:07:12.990676Z","iopub.execute_input":"2023-06-08T19:07:12.991269Z","iopub.status.idle":"2023-06-08T19:07:38.358588Z","shell.execute_reply.started":"2023-06-08T19:07:12.991215Z","shell.execute_reply":"2023-06-08T19:07:38.357048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom tqdm.auto import tqdm\nfrom sklearn import *\nimport glob\n\np = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/'\n\ntrain = glob.glob(p+'train/**/**')\ntest = glob.glob(p+'test/**/**')\nsubjects = pd.read_csv(p+'subjects.csv')\ntasks = pd.read_csv(p+'tasks.csv')\n\n\ntdcsfog_metadata=pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv')\ndefog_metadata=pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv')\n# daily_metadata=pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/daily_metadata.csv')\ntdcsfog_metadata['Module']='tdcsfog'\ndefog_metadata['Module']='defog'\n# daily_metadata['Module']='daily'\nmetadata=pd.concat([tdcsfog_metadata,defog_metadata])\nmetadata","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-08T19:07:38.362381Z","iopub.execute_input":"2023-06-08T19:07:38.362958Z","iopub.status.idle":"2023-06-08T19:07:40.640544Z","shell.execute_reply.started":"2023-06-08T19:07:38.362896Z","shell.execute_reply":"2023-06-08T19:07:40.639185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tasks","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:07:40.642696Z","iopub.execute_input":"2023-06-08T19:07:40.643568Z","iopub.status.idle":"2023-06-08T19:07:40.66301Z","shell.execute_reply.started":"2023-06-08T19:07:40.643508Z","shell.execute_reply":"2023-06-08T19:07:40.661574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/code/jazivxt/familiar-solvs\ntasks['Duration'] = tasks['End'] - tasks['Begin'] # variable Duration creada\ntasks = pd.pivot_table(tasks, values=['Duration'], index=['Id'], columns=['Task'], aggfunc='sum', fill_value=0)\ntasks.columns = [c[-1] for c in tasks.columns]\ntasks = tasks.reset_index()\ntasks['t_kmeans'] = cluster.KMeans(n_clusters=10, random_state=3).fit_predict(tasks[tasks.columns[1:]])\n\nsubjects = subjects.fillna(0).groupby('Subject').median()\nsubjects = subjects.reset_index()\n# subjects.rename(columns={'Subject':'Id'}, inplace=True)\nsubjects['s_kmeans'] = cluster.KMeans(n_clusters=10, random_state=3).fit_predict(subjects[subjects.columns[1:]])\nsubjects=subjects.rename(columns={'Visit':'s_Visit','Age':'s_Age','YearsSinceDx':'s_YearsSinceDx','UPDRSIII_On':'s_UPDRSIII_On','UPDRSIII_Off':'s_UPDRSIII_Off','NFOGQ':'s_NFOGQ'})\n\ndisplay(tasks)\ndisplay(subjects)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:07:40.666311Z","iopub.execute_input":"2023-06-08T19:07:40.666809Z","iopub.status.idle":"2023-06-08T19:07:40.887193Z","shell.execute_reply.started":"2023-06-08T19:07:40.666741Z","shell.execute_reply":"2023-06-08T19:07:40.885857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"complex_featlist=['Visit','Test','Medication','s_Visit','s_Age','s_YearsSinceDx','s_UPDRSIII_On','s_UPDRSIII_Off','s_NFOGQ','s_kmeans']\nmetadata_complex=metadata.merge(subjects,how='left',on='Subject').copy()\nmetadata_complex['Medication']=metadata_complex['Medication'].factorize()[0]\n\ndisplay(metadata_complex)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:07:40.889087Z","iopub.execute_input":"2023-06-08T19:07:40.889927Z","iopub.status.idle":"2023-06-08T19:07:40.930789Z","shell.execute_reply.started":"2023-06-08T19:07:40.88987Z","shell.execute_reply":"2023-06-08T19:07:40.929296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create a tsflex feature collection","metadata":{}},{"cell_type":"code","source":"from seglearn.feature_functions import base_features, emg_features\n\nfrom tsflex.features import FeatureCollection, MultipleFeatureDescriptors\nfrom tsflex.features.integrations import seglearn_feature_dict_wrapper\n\n\nbasic_feats = MultipleFeatureDescriptors(\n    functions=seglearn_feature_dict_wrapper(base_features()),\n    series_names=['AccV', 'AccML', 'AccAP'],\n    windows=[20_000],\n    strides=[20_000],\n)\n\nemg_feats = emg_features()\ndel emg_feats['simple square integral'] # is same as abs_energy (which is in base_features)\n\nemg_feats = MultipleFeatureDescriptors(\n    functions=seglearn_feature_dict_wrapper(emg_feats),\n    series_names=['AccV', 'AccML', 'AccAP'],\n    windows=[20_000],\n    strides=[20_000],\n)\n\nfc = FeatureCollection([basic_feats, emg_feats])","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:07:40.932437Z","iopub.execute_input":"2023-06-08T19:07:40.933437Z","iopub.status.idle":"2023-06-08T19:07:40.989971Z","shell.execute_reply.started":"2023-06-08T19:07:40.933374Z","shell.execute_reply":"2023-06-08T19:07:40.988231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fc","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:07:40.992173Z","iopub.execute_input":"2023-06-08T19:07:40.993063Z","iopub.status.idle":"2023-06-08T19:07:41.001691Z","shell.execute_reply.started":"2023-06-08T19:07:40.993005Z","shell.execute_reply":"2023-06-08T19:07:41.000165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Extract the features","metadata":{}},{"cell_type":"code","source":"train = train[:40]\n","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:07:41.003607Z","iopub.execute_input":"2023-06-08T19:07:41.004245Z","iopub.status.idle":"2023-06-08T19:07:41.012582Z","shell.execute_reply.started":"2023-06-08T19:07:41.004186Z","shell.execute_reply":"2023-06-08T19:07:41.010831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:07:41.014365Z","iopub.execute_input":"2023-06-08T19:07:41.015229Z","iopub.status.idle":"2023-06-08T19:07:41.028158Z","shell.execute_reply.started":"2023-06-08T19:07:41.015165Z","shell.execute_reply":"2023-06-08T19:07:41.02716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pathlib\ndef reader(f):\n    try:\n        df = pd.read_csv(f, index_col=\"Time\", usecols=['Time', 'AccV', 'AccML', 'AccAP', 'StartHesitation', 'Turn' , 'Walking'])\n        df['Id'] = f.split('/')[-1].split('.')[0]\n        df['Module'] = pathlib.Path(f).parts[-2]\n        df = pd.merge(df, tasks[['Id','t_kmeans']], how='left', on='Id').fillna(-1)\n#         df = pd.merge(df, subjects[['Id','s_kmeans']], how='left', on='Id').fillna(-1)\n        df = pd.merge(df, metadata_complex[['Id','Subject']+['Visit','Test','Medication','s_kmeans']], how='left', on='Id').fillna(-1)\n        df_feats = fc.calculate(df, return_df=True, include_final_window=True, approve_sparsity=True, window_idx=\"begin\").astype(np.float32)\n        df = df.merge(df_feats, how=\"left\", left_index=True, right_index=True)\n        df.fillna(method=\"ffill\", inplace=True)\n        return df\n    except: pass\n    \ntrain = pd.concat([reader(f) for f in tqdm(train[:40])]).fillna(0); print(train.shape)\ncols = [c for c in train.columns if c not in ['Id','Subject','Module', 'Time', 'StartHesitation', 'Turn' , 'Walking', 'Valid', 'Task','Event']]\npcols = ['StartHesitation', 'Turn' , 'Walking']\nscols = ['Id', 'StartHesitation', 'Turn' , 'Walking']","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:07:41.032556Z","iopub.execute_input":"2023-06-08T19:07:41.032954Z","iopub.status.idle":"2023-06-08T19:08:17.9404Z","shell.execute_reply.started":"2023-06-08T19:07:41.032915Z","shell.execute_reply":"2023-06-08T19:08:17.938951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=train.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:08:17.942139Z","iopub.execute_input":"2023-06-08T19:08:17.942518Z","iopub.status.idle":"2023-06-08T19:08:19.284383Z","shell.execute_reply.started":"2023-06-08T19:08:17.942481Z","shell.execute_reply":"2023-06-08T19:08:19.283368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sqrt_df(x):\n    return np.sqrt(abs(x))\n\n\ndef feature_engineering(x):\n    # moving average\n    x[[\"AccV_ma\",\"AccML_ma\",\"AccAP_ma\"]] = x[[\"AccV\",\"AccML\",\"AccAP\"]].rolling(window=2).mean()\n    \n    # delta with time\n    x[\"AccV_delta\"] = (x.AccV - x.AccV.shift()).fillna(0)\n    x[\"AccML_delta\"] = (x.AccML - x.AccML.shift()).fillna(0)\n    x[\"AccAP_delta\"] = (x.AccAP - x.AccAP.shift()).fillna(0)\n    \n    # stride# add feature to train dataset\n\n    x[\"Stride\"] = x[\"AccV\"] + x[\"AccML\"] + x[\"AccAP\"]\n    \n    # step\n    x[\"Step\"] = x[\"Stride\"].apply(sqrt_df)\n    \n    # fillna    \n    cols = [c for c in x.columns if c not in ['Id', 'StartHesitation', 'Turn' , 'Walking', 'Valid', 'Task','Event']] # renew cols for new data from feature engineering\n    x[cols] = x[cols].fillna(0)\n    return x","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:08:19.285791Z","iopub.execute_input":"2023-06-08T19:08:19.286434Z","iopub.status.idle":"2023-06-08T19:08:19.296139Z","shell.execute_reply.started":"2023-06-08T19:08:19.286392Z","shell.execute_reply":"2023-06-08T19:08:19.294477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# add feature to train dataset\ntrain = feature_engineering(train)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:08:19.298467Z","iopub.execute_input":"2023-06-08T19:08:19.298913Z","iopub.status.idle":"2023-06-08T19:08:32.915322Z","shell.execute_reply.started":"2023-06-08T19:08:19.298872Z","shell.execute_reply":"2023-06-08T19:08:32.913854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_memory_usage(df):\n    \n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024**2 \n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:08:32.916886Z","iopub.execute_input":"2023-06-08T19:08:32.917398Z","iopub.status.idle":"2023-06-08T19:08:32.933038Z","shell.execute_reply.started":"2023-06-08T19:08:32.917348Z","shell.execute_reply":"2023-06-08T19:08:32.931623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = reduce_memory_usage(train)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:08:32.934751Z","iopub.execute_input":"2023-06-08T19:08:32.935963Z","iopub.status.idle":"2023-06-08T19:08:54.526149Z","shell.execute_reply.started":"2023-06-08T19:08:32.935913Z","shell.execute_reply":"2023-06-08T19:08:54.52472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1. Import the necessary libraries for decision tree and feature selection:\n","metadata":{}},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\nfrom sklearn.feature_selection import SelectFromModel\n","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:08:54.527597Z","iopub.execute_input":"2023-06-08T19:08:54.528011Z","iopub.status.idle":"2023-06-08T19:08:54.533605Z","shell.execute_reply.started":"2023-06-08T19:08:54.527972Z","shell.execute_reply":"2023-06-08T19:08:54.532181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"2. Separate the input features (X) and the target variable (y) from your training data:","metadata":{}},{"cell_type":"code","source":"train.columns","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:08:54.535304Z","iopub.execute_input":"2023-06-08T19:08:54.537168Z","iopub.status.idle":"2023-06-08T19:08:54.555817Z","shell.execute_reply.started":"2023-06-08T19:08:54.537099Z","shell.execute_reply":"2023-06-08T19:08:54.552639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_features = [c for c in train.columns if c not in pcols]\npcols = ['StartHesitation', 'Turn', 'Walking']\nX_selected = pd.get_dummies(train[selected_features])\ny_selected = train[pcols]\n\nclf = DecisionTreeClassifier()\nclf.fit(X_selected, y_selected)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:08:54.557401Z","iopub.execute_input":"2023-06-08T19:08:54.557844Z","iopub.status.idle":"2023-06-08T19:25:04.87711Z","shell.execute_reply.started":"2023-06-08T19:08:54.5578Z","shell.execute_reply":"2023-06-08T19:25:04.874143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model = SelectFromModel(clf, prefit=True)\n# selected_features = [c for c in X.columns if c not in pcols]\n\n# print(\"Selected Features:\")\n# print(selected_features)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:25:04.880256Z","iopub.execute_input":"2023-06-08T19:25:04.880734Z","iopub.status.idle":"2023-06-08T19:25:04.889051Z","shell.execute_reply.started":"2023-06-08T19:25:04.88069Z","shell.execute_reply":"2023-06-08T19:25:04.886593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_importances = clf.feature_importances_\n\n# Create a DataFrame to display the feature importances\nimportance_df = pd.DataFrame({'Feature': X_selected.columns, 'Importance': feature_importances})\nimportance_df = importance_df.sort_values('Importance', ascending=False)\n\n# Display the top 10 most important features\ntop_features = importance_df.head(20)\nprint(top_features)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:25:04.891788Z","iopub.execute_input":"2023-06-08T19:25:04.892333Z","iopub.status.idle":"2023-06-08T19:25:04.922901Z","shell.execute_reply.started":"2023-06-08T19:25:04.892274Z","shell.execute_reply":"2023-06-08T19:25:04.92133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming your data frame is named 'df' and the feature importance is stored in a column named 'Feature Importance'\nselected_features = [\n    'AccAP_ma', 'AccAP__var__w=20000', 'AccAP', 'AccV__kurt__w=20000', 'AccML_delta',\n    'AccV_delta', 'AccAP_delta', 'AccML_ma', 'AccV_ma', 'AccML', 'AccAP__skew__w=20000',\n    'AccV', 'Stride', 'AccAP__median__w=20000', 'Step', 'AccAP__kurt__w=20000',\n    'AccAP__minimum__w=20000', 'AccML__maximum__w=20000', 'AccV__skew__w=20000',\n    'AccML__mean_crossings__w=20000', 'Subject','StartHesitation', 'Turn' , 'Walking'\n]\n\ntrain_filtered = train[selected_features]","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:25:04.925088Z","iopub.execute_input":"2023-06-08T19:25:04.92615Z","iopub.status.idle":"2023-06-08T19:25:05.183885Z","shell.execute_reply.started":"2023-06-08T19:25:04.926091Z","shell.execute_reply":"2023-06-08T19:25:05.182496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Adding SVM","metadata":{}},{"cell_type":"code","source":"# train_filtered.head()\n# train_filtered.drop(columns=[\"IntegerEncoding\"], inplace=True)\ntrain_filtered.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:25:05.186194Z","iopub.execute_input":"2023-06-08T19:25:05.186704Z","iopub.status.idle":"2023-06-08T19:25:05.218825Z","shell.execute_reply.started":"2023-06-08T19:25:05.186633Z","shell.execute_reply":"2023-06-08T19:25:05.217494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subset = train_filtered[['StartHesitation', 'Turn', 'Walking']].copy()\n\nintegers = subset.apply(lambda x: np.argmax(x)+1 if sum(x)!=0 else 0, axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:25:05.220489Z","iopub.execute_input":"2023-06-08T19:25:05.220941Z","iopub.status.idle":"2023-06-08T19:26:16.335193Z","shell.execute_reply.started":"2023-06-08T19:25:05.220895Z","shell.execute_reply":"2023-06-08T19:26:16.333801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_filtered[\"IntegerEncoding\"] = integers\n","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:26:16.337954Z","iopub.execute_input":"2023-06-08T19:26:16.3393Z","iopub.status.idle":"2023-06-08T19:26:16.357832Z","shell.execute_reply.started":"2023-06-08T19:26:16.33924Z","shell.execute_reply":"2023-06-08T19:26:16.356817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_filtered.columns","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:26:16.359594Z","iopub.execute_input":"2023-06-08T19:26:16.360442Z","iopub.status.idle":"2023-06-08T19:26:16.36774Z","shell.execute_reply.started":"2023-06-08T19:26:16.360399Z","shell.execute_reply":"2023-06-08T19:26:16.366596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = [c for c in train_filtered.columns if c not in ['Subject', 'IntegerEncoding', 'StartHesitation', 'Turn' , 'Walking']]\npcols = ['IntegerEncoding']","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:26:16.369051Z","iopub.execute_input":"2023-06-08T19:26:16.369404Z","iopub.status.idle":"2023-06-08T19:26:16.379173Z","shell.execute_reply.started":"2023-06-08T19:26:16.36937Z","shell.execute_reply":"2023-06-08T19:26:16.377992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x1, x2, y1, y2 = model_selection.train_test_split(train_filtered[cols], train_filtered[pcols], test_size=.10, random_state=3, stratify=train_filtered[pcols])","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:26:16.380858Z","iopub.execute_input":"2023-06-08T19:26:16.381212Z","iopub.status.idle":"2023-06-08T19:26:47.017581Z","shell.execute_reply.started":"2023-06-08T19:26:16.381179Z","shell.execute_reply":"2023-06-08T19:26:47.016275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x1.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-08T19:29:57.803365Z","iopub.execute_input":"2023-06-08T19:29:57.804018Z","iopub.status.idle":"2023-06-08T19:29:57.814633Z","shell.execute_reply.started":"2023-06-08T19:29:57.803961Z","shell.execute_reply":"2023-06-08T19:29:57.813229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = svm.SVC()\nclf.fit(x1, y1)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T14:39:21.957639Z","iopub.execute_input":"2023-06-08T14:39:21.959005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# clf = svm.SVC(kernel=rbf)\n# clf.fit(x2, y2)\nprint(metrics.average_precision_score(y1, clf.predict(x1).clip(0.0,1.0)))\n# print(metrics.average_precision_score(y1[:1_000_000], reg.predict(x1[:1_000_000]).clip(0.0,1.0)))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### This is not mine\n","metadata":{}},{"cell_type":"markdown","source":"## Train the model","metadata":{}},{"cell_type":"code","source":"train_filtered =train_filtered.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T16:12:42.220224Z","iopub.execute_input":"2023-05-31T16:12:42.2207Z","iopub.status.idle":"2023-05-31T16:12:42.353799Z","shell.execute_reply.started":"2023-05-31T16:12:42.220658Z","shell.execute_reply":"2023-05-31T16:12:42.352343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_filtered","metadata":{"execution":{"iopub.status.busy":"2023-05-31T16:12:44.32344Z","iopub.execute_input":"2023-05-31T16:12:44.323904Z","iopub.status.idle":"2023-05-31T16:12:45.334766Z","shell.execute_reply.started":"2023-05-31T16:12:44.323845Z","shell.execute_reply":"2023-05-31T16:12:45.33345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = [c for c in train_filtered.columns if c not in ['Id','Subject','Module', 'Time', 'StartHesitation', 'Turn' , 'Walking', 'Valid', 'Task','Event']]\npcols = ['StartHesitation', 'Turn' , 'Walking']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import GroupKFold\n\nN_FOLDS=5\nkfold = GroupKFold(N_FOLDS)\ngroup_var = train_filtered.Subject\ngroups=kfold.split(train_filtered, groups=group_var)\nregs=[]\ncvs=[]\nfor fold, (tr_idx,te_idx ) in enumerate(tqdm(groups, total=N_FOLDS, desc=\"Folds\")):\n    tr_idx=pd.Series(tr_idx).sample(n=2000000,random_state=100).values\n    reg = ensemble.ExtraTreesRegressor(n_estimators=100, max_depth=7, n_jobs=-1, random_state=3)\n    x_tr,y_tr=train_filtered.loc[tr_idx,cols],train_filtered.loc[tr_idx,pcols]\n    x_te,y_te=train_filtered.loc[te_idx,cols],train_filtered.loc[te_idx,pcols]\n    reg.fit(x_tr,y_tr)\n    regs.append(reg)\n    cv=metrics.average_precision_score(y_te, reg.predict(x_te).clip(0.0,1.0))\n    cvs.append(cv)\nprint(cvs)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T16:12:53.811186Z","iopub.execute_input":"2023-05-31T16:12:53.812796Z","iopub.status.idle":"2023-05-31T16:33:17.334135Z","shell.execute_reply.started":"2023-05-31T16:12:53.812725Z","shell.execute_reply":"2023-05-31T16:33:17.332211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x1, x2, y1, y2 = model_selection.train_test_split(train_filtered[cols], train_filtered[pcols], test_size=.10, random_state=3, stratify=train_filtered[pcols])\nreg = ensemble.ExtraTreesRegressor(n_estimators=100, max_depth=7, n_jobs=-1, random_state=3)\nreg.fit(x2,y2)\nprint(metrics.average_precision_score(y1[:1_000_000], reg.predict(x1[:1_000_000]).clip(0.0,1.0)))","metadata":{"execution":{"iopub.status.busy":"2023-05-31T16:40:32.533151Z","iopub.execute_input":"2023-05-31T16:40:32.533831Z","iopub.status.idle":"2023-05-31T16:42:30.827411Z","shell.execute_reply.started":"2023-05-31T16:40:32.533768Z","shell.execute_reply":"2023-05-31T16:42:30.825906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"#  This should be some proper cross validation..\n\nx1, x2, y1, y2 = model_selection.train_test_split(train_filtered[cols], train_filtered[pcols], test_size=.10, random_state=3, stratify=train_filtered[pcols])\nreg = ensemble.ExtraTreesRegressor(n_estimators=100, max_depth=7, n_jobs=-1, random_state=3)\nreg.fit(x2,y2)\nprint(metrics.average_precision_score(y1[:1_000_000], reg.predict(x1[:1_000_000]).clip(0.0,1.0)))","metadata":{"execution":{"iopub.status.busy":"2023-03-19T10:22:30.727744Z","iopub.execute_input":"2023-03-19T10:22:30.72922Z","iopub.status.idle":"2023-03-19T10:36:24.571591Z","shell.execute_reply.started":"2023-03-19T10:22:30.729165Z","shell.execute_reply":"2023-03-19T10:36:24.569535Z"}}},{"cell_type":"markdown","source":"## Predict for test","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv(p+'sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-05-31T16:44:24.173789Z","iopub.execute_input":"2023-05-31T16:44:24.174288Z","iopub.status.idle":"2023-05-31T16:44:24.473686Z","shell.execute_reply.started":"2023-05-31T16:44:24.174246Z","shell.execute_reply":"2023-05-31T16:44:24.47245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2023-05-31T16:47:29.148512Z","iopub.execute_input":"2023-05-31T16:47:29.149063Z","iopub.status.idle":"2023-05-31T16:47:29.159294Z","shell.execute_reply.started":"2023-05-31T16:47:29.149015Z","shell.execute_reply":"2023-05-31T16:47:29.157446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['t'] = 0\nsubmission = []\nfor f in test:\n    df = pd.read_csv(f)\n    \n    df.set_index('Time', drop=True, inplace=True)\n    df['Id'] = f.split('/')[-1].split('.')[0]\n#     df = df.fillna(0).reset_index(drop=True)\n    df = pd.merge(df, tasks[['Id','t_kmeans']], how='left', on='Id').fillna(-1)\n#     df = pd.merge(df, subjects[['Id','s_kmeans']], how='left', on='Id').fillna(-1)\n    df = pd.merge(df, metadata_complex[['Id','Subject']+['Visit','Test','Medication','s_kmeans']], how='left', on='Id').fillna(-1)\n    df_feats = fc.calculate(df, return_df=True, include_final_window=True, approve_sparsity=True, window_idx=\"begin\")\n    df = df.merge(df_feats, how=\"left\", left_index=True, right_index=True)\n    df.fillna(method=\"ffill\", inplace=True)\n    df = feature_engineering(df)\n    \n#     res = pd.DataFrame(np.round(reg.predict(df[cols]).clip(0.0,1.0),3), columns=pcols)\n    \n    res_vals=[]\n    for i_fold in range(N_FOLDS):\n        res_val=np.round(regs[i_fold].predict(df[cols]).clip(0.0,1.0),3)\n        res_vals.append(np.expand_dims(res_val,axis=2))\n    res_vals=np.mean(np.concatenate(res_vals,axis=2),axis=2)\n    res = pd.DataFrame(res_vals, columns=pcols)\n    \n    df = pd.concat([df,res], axis=1).reset_index(drop=True)\n    df['Id'] = df['Id'].astype(str) + '_' + df.index.astype(str)\n    submission.append(df[scols])\nsubmission = pd.concat(submission)\nsubmission = pd.merge(sub[['Id','t']], submission, how='left', on='Id').fillna(0.0)\nsubmission[scols].to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T17:36:46.350489Z","iopub.execute_input":"2023-05-31T17:36:46.351722Z","iopub.status.idle":"2023-05-31T17:36:55.708428Z","shell.execute_reply.started":"2023-05-31T17:36:46.351656Z","shell.execute_reply":"2023-05-31T17:36:55.706574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_filtered = submission[scols]","metadata":{"execution":{"iopub.status.busy":"2023-05-31T17:37:20.057626Z","iopub.execute_input":"2023-05-31T17:37:20.058115Z","iopub.status.idle":"2023-05-31T17:37:20.0722Z","shell.execute_reply.started":"2023-05-31T17:37:20.058073Z","shell.execute_reply":"2023-05-31T17:37:20.071062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_filtered.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T17:37:37.593389Z","iopub.execute_input":"2023-05-31T17:37:37.593926Z","iopub.status.idle":"2023-05-31T17:37:39.160094Z","shell.execute_reply.started":"2023-05-31T17:37:37.593871Z","shell.execute_reply":"2023-05-31T17:37:39.158738Z"},"trusted":true},"execution_count":null,"outputs":[]}]}