{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# PARKINSON FREEZING GAIT PREDICTION ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-23T15:50:24.714540Z","iopub.execute_input":"2023-03-23T15:50:24.715857Z","iopub.status.idle":"2023-03-23T15:50:28.318226Z","shell.execute_reply.started":"2023-03-23T15:50:24.715805Z","shell.execute_reply":"2023-03-23T15:50:28.316945Z"}}},{"cell_type":"markdown","source":"****The goal of this competition is to detect freezing of gait (FOG), a debilitating symptom that afflicts many people with Parkinson’s disease.**** ","metadata":{}},{"cell_type":"markdown","source":"## READING DATA","metadata":{}},{"cell_type":"code","source":"!pip install tsflex --no-index --find-links=file:///kaggle/input/imp-packs\n!pip install seglearn --no-index --find-links=file:///kaggle/input/imp-packs","metadata":{"execution":{"iopub.status.busy":"2023-03-26T16:08:15.874742Z","iopub.execute_input":"2023-03-26T16:08:15.875172Z","iopub.status.idle":"2023-03-26T16:08:41.452457Z","shell.execute_reply.started":"2023-03-26T16:08:15.875132Z","shell.execute_reply":"2023-03-26T16:08:41.450926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os \nimport gc\nfrom sklearn.cluster import KMeans\nfrom sklearn.model_selection import StratifiedKFold\nfrom lightgbm import LGBMClassifier, early_stopping, log_evaluation\nfrom sklearn.metrics import roc_auc_score,log_loss\nimport warnings \nfrom sklearn.metrics import precision_score\nimport pickle\nfrom tqdm.auto import tqdm\nimport sys\n%matplotlib inline\nwarnings.filterwarnings(\"ignore\")\nsys.path.append('/kaggle/working/mysitepackages')\n","metadata":{"execution":{"iopub.status.busy":"2023-03-26T16:09:02.870416Z","iopub.execute_input":"2023-03-26T16:09:02.871638Z","iopub.status.idle":"2023-03-26T16:09:02.881807Z","shell.execute_reply.started":"2023-03-26T16:09:02.871588Z","shell.execute_reply":"2023-03-26T16:09:02.880472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reduce Memory Usage\n# reference : https://www.kaggle.com/code/arjanso/reducing-dataframe-memory-size-by-65 @ARJANGROEN\ndef reduce_memory_usage(df):\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024**2 \n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-03-26T09:16:05.117000Z","iopub.execute_input":"2023-03-26T09:16:05.117980Z","iopub.status.idle":"2023-03-26T09:16:05.131408Z","shell.execute_reply.started":"2023-03-26T09:16:05.117940Z","shell.execute_reply":"2023-03-26T09:16:05.130120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reading metadata files\ntdcsfog_metadata = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv')\ndefog_metadata = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv')\n# daily_metadata=pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/daily_metadata.csv')\ntdcsfog_metadata['Module'] ='tdcsfog'\ndefog_metadata['Module'] = 'defog'\n# daily_metadata['Module']='daily'\nmetadata = pd.concat([tdcsfog_metadata,defog_metadata])\nmetadata.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-26T09:16:05.961238Z","iopub.execute_input":"2023-03-26T09:16:05.961654Z","iopub.status.idle":"2023-03-26T09:16:06.022702Z","shell.execute_reply.started":"2023-03-26T09:16:05.961618Z","shell.execute_reply":"2023-03-26T09:16:06.021463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/code/jazivxt/familiar-solvs\nsubjects = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/subjects.csv')\ntasks = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tasks.csv')\ntasks['Duration'] = tasks['End'] - tasks['Begin']\ntasks = pd.pivot_table(tasks, values=['Duration'], index=['Id'], columns=['Task'], aggfunc='sum', fill_value=0)\ntasks.columns = [c[-1] for c in tasks.columns]\ntasks = tasks.reset_index()\ntasks['t_kmeans'] = KMeans(n_clusters=10, random_state=3).fit_predict(tasks[tasks.columns[1:]])\n\nsubjects = subjects.fillna(0).groupby('Subject').median()\nsubjects = subjects.reset_index()\nsubjects['s_kmeans'] = KMeans(n_clusters=10, random_state=3).fit_predict(subjects[subjects.columns[1:]])\nsubjects=subjects.rename(columns={'Visit':'s_Visit','Age':'s_Age','YearsSinceDx':'s_YearsSinceDx','UPDRSIII_On':'s_UPDRSIII_On','UPDRSIII_Off':'s_UPDRSIII_Off','NFOGQ':'s_NFOGQ'})\n","metadata":{"execution":{"iopub.status.busy":"2023-03-26T09:16:06.886363Z","iopub.execute_input":"2023-03-26T09:16:06.886779Z","iopub.status.idle":"2023-03-26T09:16:07.049888Z","shell.execute_reply.started":"2023-03-26T09:16:06.886741Z","shell.execute_reply":"2023-03-26T09:16:07.048842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## FEATURE ENGINEERING","metadata":{}},{"cell_type":"code","source":"#REF:\n# https://www.kaggle.com/code/xzj19013742/groupkfold-cross-validation-tsflex\n# https://www.kaggle.com/code/jeroenvdd/time-series-tsflex\n\nfrom seglearn.feature_functions import base_features, emg_features\n\nfrom tsflex.features import FeatureCollection, MultipleFeatureDescriptors\nfrom tsflex.features.integrations import seglearn_feature_dict_wrapper\n\n\nbasic_feats = MultipleFeatureDescriptors(\n    functions=seglearn_feature_dict_wrapper(base_features()),\n    series_names=['AccV', 'AccML', 'AccAP'],\n    windows=[5_000],\n    strides=[5_000],\n)\n\nemg_feats = emg_features()\ndel emg_feats['simple square integral'] # is same as abs_energy (which is in base_features)\n\nemg_feats = MultipleFeatureDescriptors(\n    functions=seglearn_feature_dict_wrapper(emg_feats),\n    series_names=['AccV', 'AccML', 'AccAP'],\n    windows=[5_000],\n    strides=[5_000],\n)\n\nfc = FeatureCollection([basic_feats, emg_feats])","metadata":{"execution":{"iopub.status.busy":"2023-03-26T09:39:50.695987Z","iopub.execute_input":"2023-03-26T09:39:50.696757Z","iopub.status.idle":"2023-03-26T09:39:50.758646Z","shell.execute_reply.started":"2023-03-26T09:39:50.696715Z","shell.execute_reply":"2023-03-26T09:39:50.757628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#complex_featlist = ['Visit','Test','Medication','s_Visit','s_Age','s_YearsSinceDx','s_UPDRSIII_On','s_UPDRSIII_Off','s_NFOGQ','s_kmeans']\nmetadata_complex = metadata.merge(subjects,how='left',on='Subject').copy()\nmetadata_complex['Medication'] = metadata_complex['Medication'].factorize()[0]","metadata":{"execution":{"iopub.status.busy":"2023-03-26T09:39:51.990498Z","iopub.execute_input":"2023-03-26T09:39:51.991221Z","iopub.status.idle":"2023-03-26T09:39:52.006751Z","shell.execute_reply.started":"2023-03-26T09:39:51.991160Z","shell.execute_reply":"2023-03-26T09:39:52.005394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#reference: https://www.kaggle.com/code/ghrangel/read-data-and-merge\nDATA_ROOT_DEFOG = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/'\ndefog = pd.DataFrame()\ntry:\n    for root, dirs, files in os.walk(DATA_ROOT_DEFOG):\n        for name in files:       \n            f = os.path.join(root, name)\n            df_list= pd.read_csv(f, usecols=['Time', 'AccV', 'AccML', 'AccAP', 'StartHesitation', 'Turn' , 'Walking','Valid','Task'])\n            words = name.split('.')[0]\n            df_list['Id']= words\n            df_list = pd.merge(df_list, tasks[['Id','t_kmeans']], how='left', on='Id').fillna(-1)\n            df_list = pd.merge(df_list, metadata_complex[['Id','Subject']+['Visit','Test','Medication','s_kmeans']], how='left', on='Id').fillna(-1)\n            defog_feats = fc.calculate(df_list, return_df=True, include_final_window=True, approve_sparsity=True, window_idx=\"begin\").astype(np.float32)\n            df_list = df_list.merge(defog_feats, how=\"left\", left_index=True, right_index=True)\n            df_list.fillna(method=\"ffill\", inplace=True)\n            defog = pd.concat([defog, df_list], axis=0)\nexcept Exception as e:\n    raise e\ndefog.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-26T09:39:54.983781Z","iopub.execute_input":"2023-03-26T09:39:54.984477Z","iopub.status.idle":"2023-03-26T09:42:51.254643Z","shell.execute_reply.started":"2023-03-26T09:39:54.984437Z","shell.execute_reply":"2023-03-26T09:42:51.253393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog = reduce_memory_usage(defog)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-26T09:45:45.024408Z","iopub.execute_input":"2023-03-26T09:45:45.025424Z","iopub.status.idle":"2023-03-26T09:46:34.155392Z","shell.execute_reply.started":"2023-03-26T09:45:45.025368Z","shell.execute_reply":"2023-03-26T09:46:34.153333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Shape of the defog dataset with all Valid entries \"+str(defog.shape))","metadata":{"execution":{"iopub.status.busy":"2023-03-26T09:46:38.684197Z","iopub.execute_input":"2023-03-26T09:46:38.685239Z","iopub.status.idle":"2023-03-26T09:46:38.693240Z","shell.execute_reply.started":"2023-03-26T09:46:38.685189Z","shell.execute_reply":"2023-03-26T09:46:38.692185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_ROOT_TDCSFOG = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/'\ntdcsfog = pd.DataFrame()\ntry:\n    for root, dirs, files in os.walk(DATA_ROOT_TDCSFOG):\n        for name in files:       \n            f = os.path.join(root, name)\n            df_list= pd.read_csv(f, usecols=['Time', 'AccV', 'AccML', 'AccAP', 'StartHesitation', 'Turn' , 'Walking'])\n            words = name.split('.')[0]\n            df_list['Id']= words\n            df_list = pd.merge(df_list, tasks[['Id','t_kmeans']], how='left', on='Id').fillna(-1)\n            df_list = pd.merge(df_list, metadata_complex[['Id','Subject']+['Visit','Test','Medication','s_kmeans']], how='left', on='Id').fillna(-1)\n            tdcsfog_feats = fc.calculate(df_list, return_df=True, include_final_window=True, approve_sparsity=True, window_idx=\"begin\").astype(np.float32)\n            df_list = df_list.merge(tdcsfog_feats, how=\"left\", left_index=True, right_index=True)\n            df_list.fillna(method=\"ffill\", inplace=True)\n            tdcsfog = pd.concat([tdcsfog, df_list], axis=0)\nexcept Exception as e:\n    raise e\ntdcsfog.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-26T09:46:44.072431Z","iopub.execute_input":"2023-03-26T09:46:44.072844Z","iopub.status.idle":"2023-03-26T09:59:08.372389Z","shell.execute_reply.started":"2023-03-26T09:46:44.072806Z","shell.execute_reply":"2023-03-26T09:59:08.370946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog = reduce_memory_usage(tdcsfog)","metadata":{"execution":{"iopub.status.busy":"2023-03-26T10:00:58.084152Z","iopub.execute_input":"2023-03-26T10:00:58.085274Z","iopub.status.idle":"2023-03-26T10:01:21.456738Z","shell.execute_reply.started":"2023-03-26T10:00:58.085223Z","shell.execute_reply":"2023-03-26T10:01:21.455855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-26T10:12:51.879642Z","iopub.execute_input":"2023-03-26T10:12:51.880248Z","iopub.status.idle":"2023-03-26T10:12:52.072254Z","shell.execute_reply.started":"2023-03-26T10:12:51.880197Z","shell.execute_reply":"2023-03-26T10:12:52.070874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_list,tdcsfog_feats","metadata":{"execution":{"iopub.status.busy":"2023-03-25T19:22:23.339261Z","iopub.execute_input":"2023-03-25T19:22:23.339817Z","iopub.status.idle":"2023-03-25T19:22:23.346761Z","shell.execute_reply.started":"2023-03-25T19:22:23.339757Z","shell.execute_reply":"2023-03-25T19:22:23.345557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EDA","metadata":{}},{"cell_type":"code","source":"def summary(df):\n    print(f'data shape: {df.shape}')\n    summ = pd.DataFrame(df.dtypes, columns=['data type'])\n    summ['#missing'] = df.isnull().sum().values * 100\n    summ['%missing'] = df.isnull().sum().values / len(df)\n    summ['#unique'] = df.nunique().values\n    desc = pd.DataFrame(df.describe(include='all').transpose())\n    summ['min'] = desc['min'].values\n    summ['max'] = desc['max'].values\n    return summ","metadata":{"execution":{"iopub.status.busy":"2023-03-25T10:23:48.244920Z","iopub.execute_input":"2023-03-25T10:23:48.245368Z","iopub.status.idle":"2023-03-25T10:23:48.252758Z","shell.execute_reply.started":"2023-03-25T10:23:48.245317Z","shell.execute_reply":"2023-03-25T10:23:48.251584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summary(defog[['Time', 'AccV', 'AccML', 'AccAP', 'StartHesitation', 'Turn' , 'Walking']])","metadata":{"execution":{"iopub.status.busy":"2023-03-25T10:23:50.954419Z","iopub.execute_input":"2023-03-25T10:23:50.955239Z","iopub.status.idle":"2023-03-25T10:23:58.788645Z","shell.execute_reply.started":"2023-03-25T10:23:50.955185Z","shell.execute_reply":"2023-03-25T10:23:58.787036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summary(tdcsfog[['Time', 'AccV', 'AccML', 'AccAP', 'StartHesitation', 'Turn' , 'Walking']])","metadata":{"execution":{"iopub.status.busy":"2023-03-25T10:23:58.790815Z","iopub.execute_input":"2023-03-25T10:23:58.791217Z","iopub.status.idle":"2023-03-25T10:24:02.620702Z","shell.execute_reply.started":"2023-03-25T10:23:58.791176Z","shell.execute_reply":"2023-03-25T10:24:02.619385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog = defog[(defog['Valid']==1) & (defog['Task']==1)]","metadata":{"execution":{"iopub.status.busy":"2023-03-26T10:14:29.202249Z","iopub.execute_input":"2023-03-26T10:14:29.202683Z","iopub.status.idle":"2023-03-26T10:14:33.283884Z","shell.execute_reply.started":"2023-03-26T10:14:29.202635Z","shell.execute_reply":"2023-03-26T10:14:33.282780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Shape for valid entries: \"+str(defog.shape))","metadata":{"execution":{"iopub.status.busy":"2023-03-26T10:14:34.289770Z","iopub.execute_input":"2023-03-26T10:14:34.291020Z","iopub.status.idle":"2023-03-26T10:14:34.296696Z","shell.execute_reply.started":"2023-03-26T10:14:34.290967Z","shell.execute_reply":"2023-03-26T10:14:34.295865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conditions = [\n    (defog['StartHesitation'] == 1),\n    (defog['Turn'] == 1),\n    (defog['Walking'] == 1)]\nchoices = ['StartHesitation', 'Turn', 'Walking']\ndefog['event'] = np.select(conditions, choices, default='Normal')\ndefog['event'].value_counts().to_frame().style.background_gradient()","metadata":{"execution":{"iopub.status.busy":"2023-03-26T10:13:02.730361Z","iopub.execute_input":"2023-03-26T10:13:02.730801Z","iopub.status.idle":"2023-03-26T10:13:09.252162Z","shell.execute_reply.started":"2023-03-26T10:13:02.730750Z","shell.execute_reply":"2023-03-26T10:13:09.250992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.rcParams['figure.figsize'] = 15,40\nplt.rcParams['figure.autolayout'] = True\ncat_cols = ['Medication','Visit']\nfor i,var in enumerate(cat_cols) :\n    ax = plt.subplot(3,1,i+1)\n    defog_count_df = (defog.groupby(var)['event'].value_counts(1)*100).to_frame('event_perc').reset_index()\n    sns.barplot(x = 'event', y = 'event_perc',hue = var,data = defog_count_df,ax=ax)\n    ax.set_title(var)\nplt.show()\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2023-03-24T22:11:56.292010Z","iopub.execute_input":"2023-03-24T22:11:56.292422Z","iopub.status.idle":"2023-03-24T22:11:57.933761Z","shell.execute_reply.started":"2023-03-24T22:11:56.292384Z","shell.execute_reply":"2023-03-24T22:11:57.932597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conditions = [\n    (tdcsfog['StartHesitation'] == 1),\n    (tdcsfog['Turn'] == 1),\n    (tdcsfog['Walking'] == 1)]\nchoices = ['StartHesitation', 'Turn', 'Walking']\ntdcsfog['event'] = np.select(conditions, choices, default='Normal')\ntdcsfog['event'].value_counts().to_frame().style.background_gradient()","metadata":{"execution":{"iopub.status.busy":"2023-03-26T10:13:13.234071Z","iopub.execute_input":"2023-03-26T10:13:13.235138Z","iopub.status.idle":"2023-03-26T10:13:15.426109Z","shell.execute_reply.started":"2023-03-26T10:13:13.235089Z","shell.execute_reply":"2023-03-26T10:13:15.424845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.rcParams['figure.figsize'] = 15,40\nplt.rcParams['figure.autolayout'] = True\ncat_cols = ['Medication','Visit']\nfor i,var in enumerate(cat_cols) :\n    ax = plt.subplot(3,1,i+1)\n    tdcs_count_df = (tdcsfog.groupby(var)['event'].value_counts(1)*100).to_frame('event_perc').reset_index()\n    sns.barplot(x = 'event', y = 'event_perc',hue = var,data = tdcs_count_df,ax=ax)\n    ax.set_title(var)\nplt.show()\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2023-03-24T22:12:01.365823Z","iopub.execute_input":"2023-03-24T22:12:01.366916Z","iopub.status.idle":"2023-03-24T22:12:03.935304Z","shell.execute_reply.started":"2023-03-24T22:12:01.366833Z","shell.execute_reply":"2023-03-24T22:12:03.934039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(12,6))\ndefog.plot(x='Time', y=['AccV', 'AccML', 'AccAP'], ax=ax);\nax.set_xlabel('Time (s)');\nax.set_ylabel('Acceleration (g)');\nax.set_title('Acceleration for defog over time');","metadata":{"execution":{"iopub.status.busy":"2023-03-24T22:12:04.724486Z","iopub.execute_input":"2023-03-24T22:12:04.724920Z","iopub.status.idle":"2023-03-24T22:12:19.603759Z","shell.execute_reply.started":"2023-03-24T22:12:04.724860Z","shell.execute_reply":"2023-03-24T22:12:19.602671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 1, sharex='col', sharey='row', figsize=(16,10))\nfig.suptitle('Sensor Values over Time')\ndata = {'V': defog['AccV'].astype(np.float32), 'ML': defog['AccML'].astype(np.float32), 'AP': defog['AccAP'].astype(np.float32)}\n\nfor i, (name, values) in enumerate(data.items()):\n    ax = axes[i]\n    ax.plot(values, label=name, color=f'C{i}')\n    a, b = np.polyfit(values.index, values, 1)\n    ax.plot(values.index, a*values.index+b, color='red')\n    ax.set_xlabel('Time')\n    ax.set_ylabel('Sensor Value')\n    ax.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-24T22:12:26.560532Z","iopub.execute_input":"2023-03-24T22:12:26.561551Z","iopub.status.idle":"2023-03-24T22:12:46.127035Z","shell.execute_reply.started":"2023-03-24T22:12:26.561507Z","shell.execute_reply":"2023-03-24T22:12:46.126238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(12,6))\ntdcsfog.plot(x='Time', y=['AccV', 'AccML', 'AccAP'], ax=ax);\nax.set_xlabel('Time (s)');\nax.set_ylabel('Acceleration (g)');\nax.set_title('Acceleration for tdcsfog dataset over time');","metadata":{"execution":{"iopub.status.busy":"2023-03-24T22:14:37.136210Z","iopub.execute_input":"2023-03-24T22:14:37.136652Z","iopub.status.idle":"2023-03-24T22:15:16.105336Z","shell.execute_reply.started":"2023-03-24T22:14:37.136616Z","shell.execute_reply":"2023-03-24T22:15:16.104321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 1, sharex='col', sharey='row', figsize=(16,10))\nfig.suptitle('Sensor Values over Time')\ndata = {'V': tdcsfog['AccV'].astype(np.float32), 'ML': tdcsfog['AccML'].astype(np.float32), 'AP': tdcsfog['AccAP'].astype(np.float32)}\n\nfor i, (name, values) in enumerate(data.items()):\n    ax = axes[i]\n    ax.plot(values, label=name, color=f'C{i}')\n    a, b = np.polyfit(values.index, values, 1)\n    ax.plot(values.index, a*values.index+b, color='red')\n    ax.set_xlabel('Time')\n    ax.set_ylabel('Sensor Value')\n    ax.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-24T22:15:25.722838Z","iopub.execute_input":"2023-03-24T22:15:25.723228Z","iopub.status.idle":"2023-03-24T22:16:02.691795Z","shell.execute_reply.started":"2023-03-24T22:15:25.723194Z","shell.execute_reply":"2023-03-24T22:16:02.690631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nl1 = LabelEncoder()\nl2 = LabelEncoder()","metadata":{"execution":{"iopub.status.busy":"2023-03-26T10:13:19.210697Z","iopub.execute_input":"2023-03-26T10:13:19.211553Z","iopub.status.idle":"2023-03-26T10:13:19.217960Z","shell.execute_reply.started":"2023-03-26T10:13:19.211502Z","shell.execute_reply":"2023-03-26T10:13:19.216668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-26T10:13:20.207209Z","iopub.execute_input":"2023-03-26T10:13:20.208085Z","iopub.status.idle":"2023-03-26T10:13:20.357506Z","shell.execute_reply.started":"2023-03-26T10:13:20.208027Z","shell.execute_reply":"2023-03-26T10:13:20.356025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog['Target'] = l1.fit_transform(defog['event'])\ntdcsfog['Target'] = l2.fit_transform(tdcsfog['event'])\n","metadata":{"execution":{"iopub.status.busy":"2023-03-26T10:13:24.259571Z","iopub.execute_input":"2023-03-26T10:13:24.260486Z","iopub.status.idle":"2023-03-26T10:13:29.573710Z","shell.execute_reply.started":"2023-03-26T10:13:24.260433Z","shell.execute_reply":"2023-03-26T10:13:29.572195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## MODELLING","metadata":{}},{"cell_type":"code","source":"X_defog = defog.drop(['Target','event', 'Id','Subject','StartHesitation','Turn','Walking','Time','Valid','Task'], axis = 1)\ny_defog = defog['Target']","metadata":{"execution":{"iopub.status.busy":"2023-03-26T10:33:45.214918Z","iopub.execute_input":"2023-03-26T10:33:45.215423Z","iopub.status.idle":"2023-03-26T10:33:45.596501Z","shell.execute_reply.started":"2023-03-26T10:33:45.215384Z","shell.execute_reply":"2023-03-26T10:33:45.595394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_model(X,y):\n    y_valid, gbm_val_probs, gini,ps_lis=[],[],[],[]\n    #ft_importance = pd.DataFrame(index=X.columns)\n    X.reset_index(drop = True,inplace = True)\n    y.reset_index(drop = True,inplace = True)\n    sk_fold = StratifiedKFold(n_splits=3, shuffle=True, random_state=21)\n    for fold, (train_idx, val_idx) in enumerate(tqdm(sk_fold.split(X, y))):\n        train_idx = pd.Series(train_idx).sample(n=2000000,random_state=100).values\n        print(\"\\nFold {}\".format(fold+1))\n        X_train, y_train = X.loc[train_idx,:], y[train_idx]\n        X_val, y_val = X.loc[val_idx,:], y[val_idx]\n        print(\"Subsetting finished\")\n       # print(\"Train shape: {}, {}, Valid shape: {}, {}\\n\".format(\n           # X_train.shape, y_train.shape, X_val.shape, y_val.shape))\n\n        params = {'boosting_type': 'gbdt',\n                  'n_estimators': 1000,\n                  'num_leaves': 50,\n                  'learning_rate': 0.05,\n                  'colsample_bytree': 0.9,\n                  'min_child_samples': 2000,\n                  'max_bins': 500,\n                  'reg_alpha': 2,\n                  'objective': 'multiclass',\n                  'random_state': 21}\n\n        gbm = LGBMClassifier(**params).fit(X_train, y_train, \n                                           eval_set=[(X_val, y_val)],\n                                           callbacks=[early_stopping(200), log_evaluation(500)],\n                                           eval_metric=['auc_mu','multi_logloss'])\n        gbm_prob = gbm.predict_proba(X_val)\n        gbm_val_probs.append(gbm_prob)\n        y_valid.append(y_val)\n\n        auc_score = roc_auc_score(y_val, gbm_prob,multi_class='ovo')\n        p_s = precision_score(y_val, np.argmax(gbm_prob, axis= -1), average='macro')\n        gini_score = 2*auc_score - 1\n        gini.append(gini_score)\n        ps_lis.append(p_s)\n\n        print(\"Validation Gini: {:.5f}, AUC: {:.4f}\".format(gini_score,auc_score))\n\n        del X_train, y_train, X_val, y_val\n        _ = gc.collect()\n    del X, y\n    return [gini,gbm,ps_lis]\n#plot_roc(y_valid, gbm_val_probs)","metadata":{"execution":{"iopub.status.busy":"2023-03-26T10:13:51.768056Z","iopub.execute_input":"2023-03-26T10:13:51.769256Z","iopub.status.idle":"2023-03-26T10:13:51.784127Z","shell.execute_reply.started":"2023-03-26T10:13:51.769193Z","shell.execute_reply":"2023-03-26T10:13:51.783009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#output_defog = train_model(X_defog,y_defog) saved the model and loading it\n","metadata":{"execution":{"iopub.status.busy":"2023-03-26T10:33:50.708161Z","iopub.execute_input":"2023-03-26T10:33:50.708565Z","iopub.status.idle":"2023-03-26T10:57:37.583687Z","shell.execute_reply.started":"2023-03-26T10:33:50.708530Z","shell.execute_reply":"2023-03-26T10:57:37.582453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_defog = pd.read_pickle('/kaggle/input/dfog-output/Output_defog.pkl')","metadata":{"execution":{"iopub.status.busy":"2023-03-26T16:10:26.110941Z","iopub.execute_input":"2023-03-26T16:10:26.112036Z","iopub.status.idle":"2023-03-26T16:10:26.144784Z","shell.execute_reply.started":"2023-03-26T16:10:26.111988Z","shell.execute_reply":"2023-03-26T16:10:26.143467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"MAP for defog train: \"+str(np.mean(output_defog[2])))","metadata":{"execution":{"iopub.status.busy":"2023-03-26T10:58:25.182096Z","iopub.execute_input":"2023-03-26T10:58:25.182520Z","iopub.status.idle":"2023-03-26T10:58:25.188392Z","shell.execute_reply.started":"2023-03-26T10:58:25.182484Z","shell.execute_reply":"2023-03-26T10:58:25.187079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\ndel X_defog,y_defog","metadata":{"execution":{"iopub.status.busy":"2023-03-26T10:59:49.045148Z","iopub.execute_input":"2023-03-26T10:59:49.045553Z","iopub.status.idle":"2023-03-26T10:59:49.230517Z","shell.execute_reply.started":"2023-03-26T10:59:49.045516Z","shell.execute_reply":"2023-03-26T10:59:49.228856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_tdcsfog = tdcsfog.drop(['Target','event', 'Id','Subject','StartHesitation','Turn','Walking','Time'], axis = 1)\ny_tdcsfog = tdcsfog['Target']","metadata":{"execution":{"iopub.status.busy":"2023-03-25T19:22:56.367100Z","iopub.execute_input":"2023-03-25T19:22:56.367520Z","iopub.status.idle":"2023-03-25T19:22:58.870245Z","shell.execute_reply.started":"2023-03-25T19:22:56.367486Z","shell.execute_reply":"2023-03-25T19:22:58.868981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#output_tdcsfog = train_model(X_tdcsfog,y_tdcsfog) saved the model and loading it","metadata":{"execution":{"iopub.status.busy":"2023-03-25T19:23:11.273108Z","iopub.execute_input":"2023-03-25T19:23:11.273588Z","iopub.status.idle":"2023-03-25T21:58:28.868405Z","shell.execute_reply.started":"2023-03-25T19:23:11.273549Z","shell.execute_reply":"2023-03-25T21:58:28.867050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_tdcsfog = pd.read_pickle('/kaggle/input/tfog-output/tdcs_output.pkl')","metadata":{"execution":{"iopub.status.busy":"2023-03-26T16:11:06.629562Z","iopub.execute_input":"2023-03-26T16:11:06.629984Z","iopub.status.idle":"2023-03-26T16:11:07.269816Z","shell.execute_reply.started":"2023-03-26T16:11:06.629948Z","shell.execute_reply":"2023-03-26T16:11:07.268651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"MAP for tdcsfog train: \"+str(np.mean(output_tdcsfog[2])))","metadata":{"execution":{"iopub.status.busy":"2023-03-26T11:38:20.947228Z","iopub.execute_input":"2023-03-26T11:38:20.947764Z","iopub.status.idle":"2023-03-26T11:38:20.956524Z","shell.execute_reply.started":"2023-03-26T11:38:20.947719Z","shell.execute_reply":"2023-03-26T11:38:20.954995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X_tdcsfog,y_tdcsfog","metadata":{"execution":{"iopub.status.busy":"2023-03-25T22:06:09.821268Z","iopub.execute_input":"2023-03-25T22:06:09.822460Z","iopub.status.idle":"2023-03-25T22:06:10.136519Z","shell.execute_reply.started":"2023-03-25T22:06:09.822400Z","shell.execute_reply":"2023-03-25T22:06:10.135136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## GENERATING PREDICTIONS","metadata":{"execution":{"iopub.status.busy":"2023-03-25T22:35:52.950961Z","iopub.execute_input":"2023-03-25T22:35:52.951475Z","iopub.status.idle":"2023-03-25T22:35:52.960058Z","shell.execute_reply.started":"2023-03-25T22:35:52.951434Z","shell.execute_reply":"2023-03-25T22:35:52.958543Z"}}},{"cell_type":"code","source":"DATA_ROOT_DEFOG = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog'\ndefog_test = pd.DataFrame()\ntry:\n    for root, dirs, files in os.walk(DATA_ROOT_DEFOG):\n        for name in files:       \n            f = os.path.join(root, name)\n            df_list= pd.read_csv(f)\n            words = name.split('.')[0]\n            df_list['Id']= words\n            df_list = pd.merge(df_list, tasks[['Id','t_kmeans']], how='left', on='Id').fillna(-1)\n            df_list = pd.merge(df_list, metadata_complex[['Id','Subject']+['Visit','Test','Medication','s_kmeans']], how='left', on='Id').fillna(-1)\n            tdcsfog_feats = fc.calculate(df_list, return_df=True, include_final_window=True, approve_sparsity=True, window_idx=\"begin\").astype(np.float32)\n            df_list = df_list.merge(tdcsfog_feats, how=\"left\", left_index=True, right_index=True)\n            df_list.fillna(method=\"ffill\", inplace=True)\n            df_list['Id'] = df_list['Id'].astype(str) + '_' + df_list['Time'].astype(str)\n            df_list.set_index('Id',inplace=True)\n            df_list.drop(['Subject','Time'],axis = 1,inplace = True)\n            defog_preds = output_defog[1].predict_proba(df_list)\n            defog_test = pd.concat([defog_test, df_list], axis=0)\n            defog_test['event'] = np.argmax(defog_preds, axis=-1)\n            defog_test['event'] = l1.inverse_transform(defog_test['event'])\n            defog_test['StartHesitation'] = np.where(defog_test['event']=='StartHesitation', 1, 0)\n            defog_test['Turn'] = np.where(defog_test['event']=='Turn', 1, 0)\n            defog_test['Walking'] = np.where(defog_test['event']=='Walking', 1, 0)\n            defog_test = defog_test[['StartHesitation','Turn','Walking']]\n            defog_test = defog_test.reset_index()\nexcept Exception as e:\n    raise e","metadata":{"execution":{"iopub.status.busy":"2023-03-26T11:17:17.225489Z","iopub.execute_input":"2023-03-26T11:17:17.226035Z","iopub.status.idle":"2023-03-26T11:17:19.459012Z","shell.execute_reply.started":"2023-03-26T11:17:17.225986Z","shell.execute_reply":"2023-03-26T11:17:19.457506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_ROOT_TDCSFOG = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog'\ntdcsfog_test = pd.DataFrame()\ntry:\n    for root, dirs, files in os.walk(DATA_ROOT_TDCSFOG):\n        for name in files:       \n            f = os.path.join(root, name)\n            df_list= pd.read_csv(f)\n            words = name.split('.')[0]\n            df_list['Id']= words\n            df_list = pd.merge(df_list, tasks[['Id','t_kmeans']], how='left', on='Id').fillna(-1)\n            df_list = pd.merge(df_list, metadata_complex[['Id','Subject']+['Visit','Test','Medication','s_kmeans']], how='left', on='Id').fillna(-1)\n            tdcsfog_feats = fc.calculate(df_list, return_df=True, include_final_window=True, approve_sparsity=True, window_idx=\"begin\").astype(np.float32)\n            df_list = df_list.merge(tdcsfog_feats, how=\"left\", left_index=True, right_index=True)\n            df_list.fillna(method=\"ffill\", inplace=True)\n            df_list['Id'] = df_list['Id'].astype(str) + '_' + df_list['Time'].astype(str)\n            df_list.set_index('Id',inplace=True)\n            df_list.drop(['Subject','Time'],axis = 1,inplace = True)\n            tfog_preds = output_tdcsfog[1].predict_proba(df_list)\n            tdcsfog_test = pd.concat([tdcsfog_test, df_list], axis=0)\n            tdcsfog_test['event'] = np.argmax(tfog_preds, axis=-1)\n            tdcsfog_test['event'] = l2.inverse_transform(tdcsfog_test['event'])\n            tdcsfog_test['StartHesitation'] = np.where(tdcsfog_test['event']=='StartHesitation', 1, 0)\n            tdcsfog_test['Turn'] = np.where(tdcsfog_test['event']=='Turn', 1, 0)\n            tdcsfog_test['Walking'] = np.where(tdcsfog_test['event']=='Walking', 1, 0)\n            tdcsfog_test = tdcsfog_test[['StartHesitation','Turn','Walking']]\n            tdcsfog_test = tdcsfog_test.reset_index()\nexcept Exception as e:\n    raise e","metadata":{"execution":{"iopub.status.busy":"2023-03-26T11:33:59.046110Z","iopub.execute_input":"2023-03-26T11:33:59.046556Z","iopub.status.idle":"2023-03-26T11:34:00.640740Z","shell.execute_reply.started":"2023-03-26T11:33:59.046518Z","shell.execute_reply":"2023-03-26T11:34:00.639365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-26T11:35:34.991399Z","iopub.execute_input":"2023-03-26T11:35:34.992038Z","iopub.status.idle":"2023-03-26T11:35:35.007111Z","shell.execute_reply.started":"2023-03-26T11:35:34.991977Z","shell.execute_reply":"2023-03-26T11:35:35.005324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-26T11:34:13.793344Z","iopub.execute_input":"2023-03-26T11:34:13.794100Z","iopub.status.idle":"2023-03-26T11:34:13.808125Z","shell.execute_reply.started":"2023-03-26T11:34:13.794052Z","shell.execute_reply":"2023-03-26T11:34:13.806570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.concat([tdcsfog_test,defog_test])\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-26T11:35:55.849158Z","iopub.execute_input":"2023-03-26T11:35:55.850691Z","iopub.status.idle":"2023-03-26T11:35:55.891370Z","shell.execute_reply.started":"2023-03-26T11:35:55.850631Z","shell.execute_reply":"2023-03-26T11:35:55.889805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv',index = False)","metadata":{"execution":{"iopub.status.busy":"2023-03-26T11:36:06.860465Z","iopub.execute_input":"2023-03-26T11:36:06.862445Z","iopub.status.idle":"2023-03-26T11:36:07.323469Z","shell.execute_reply.started":"2023-03-26T11:36:06.862374Z","shell.execute_reply":"2023-03-26T11:36:07.322171Z"},"trusted":true},"execution_count":null,"outputs":[]}]}