{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Importing basic python packages","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-11T10:20:37.423808Z","iopub.execute_input":"2023-03-11T10:20:37.424250Z","iopub.status.idle":"2023-03-11T10:20:37.430246Z","shell.execute_reply.started":"2023-03-11T10:20:37.424206Z","shell.execute_reply":"2023-03-11T10:20:37.428891Z"}}},{"cell_type":"code","source":"# Importing useful packages\nimport numpy as np \nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport glob\nfrom scipy.stats import describe\nimport plotly.express as px","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:42.698791Z","iopub.execute_input":"2023-03-12T22:32:42.699217Z","iopub.status.idle":"2023-03-12T22:32:46.695752Z","shell.execute_reply.started":"2023-03-12T22:32:42.699183Z","shell.execute_reply":"2023-03-12T22:32:46.694321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Data \n","metadata":{}},{"cell_type":"markdown","source":"## Creating defog dataframe using pandas","metadata":{}},{"cell_type":"code","source":"# Reading all the csv files in the dataframe\n#file_path = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog\"\nfile_path = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/139f60d29b.csv\"\ndf_defog = pd.read_csv(file_path,index_col = None, header = 0)\n\n#csv_files = glob.glob(os.path.join(file_path,\"*.csv\"))\n\n#file_list = []\n#for file_name in csv_files:\n #   df = pd.read_csv(file_name,index_col = None, header = 0)\n #   file_list.append(df)\n#df_defog = pd.concat(file_list, axis=0, ignore_index=True)  ","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:46.697843Z","iopub.execute_input":"2023-03-12T22:32:46.698179Z","iopub.status.idle":"2023-03-12T22:32:46.933392Z","shell.execute_reply.started":"2023-03-12T22:32:46.698148Z","shell.execute_reply":"2023-03-12T22:32:46.932145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Information about defog dataset\n","metadata":{}},{"cell_type":"code","source":"# Printing useful information about datasets\ndf_defog.info()","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:46.935017Z","iopub.execute_input":"2023-03-12T22:32:46.935402Z","iopub.status.idle":"2023-03-12T22:32:46.968454Z","shell.execute_reply.started":"2023-03-12T22:32:46.935352Z","shell.execute_reply":"2023-03-12T22:32:46.966909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Showing top 10 values for defog datasets\n","metadata":{}},{"cell_type":"code","source":"# Printing the top 10 values from defog datasets\ndf_defog.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:46.972073Z","iopub.execute_input":"2023-03-12T22:32:46.972884Z","iopub.status.idle":"2023-03-12T22:32:47.006512Z","shell.execute_reply.started":"2023-03-12T22:32:46.972826Z","shell.execute_reply":"2023-03-12T22:32:47.004959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Showing bottom 10 values for defog datasets","metadata":{}},{"cell_type":"code","source":"# Printing the bottom 10 values from defog datasets\ndf_defog.tail(10)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:47.007949Z","iopub.execute_input":"2023-03-12T22:32:47.008291Z","iopub.status.idle":"2023-03-12T22:32:47.026693Z","shell.execute_reply.started":"2023-03-12T22:32:47.008258Z","shell.execute_reply":"2023-03-12T22:32:47.025267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Shape of the defog dataframe\n","metadata":{}},{"cell_type":"code","source":"df_defog.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:47.028590Z","iopub.execute_input":"2023-03-12T22:32:47.029067Z","iopub.status.idle":"2023-03-12T22:32:47.037347Z","shell.execute_reply.started":"2023-03-12T22:32:47.029026Z","shell.execute_reply":"2023-03-12T22:32:47.036103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  Checking if there are any missing values in defog dataset ","metadata":{}},{"cell_type":"code","source":"# Checking if any of the dataset columns have missing values\ndf_defog.isnull().values.any().sum()","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:47.039390Z","iopub.execute_input":"2023-03-12T22:32:47.039836Z","iopub.status.idle":"2023-03-12T22:32:47.053755Z","shell.execute_reply.started":"2023-03-12T22:32:47.039798Z","shell.execute_reply":"2023-03-12T22:32:47.052077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Describing the defog datasets with skiwness and kurtosis","metadata":{}},{"cell_type":"code","source":"# Describing the datasets with variance, skewness and kurtosis\nstats_defog=df_defog.describe(include = 'all')\nstats_defog.loc['var'] = df_defog.var().tolist()\nstats_defog.loc['skew'] = df_defog.skew().tolist()\nstats_defog.loc['kurt'] = df_defog.kurtosis().tolist()\nnon_nan_df_defog=stats_defog.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:47.055381Z","iopub.execute_input":"2023-03-12T22:32:47.055750Z","iopub.status.idle":"2023-03-12T22:32:47.151070Z","shell.execute_reply.started":"2023-03-12T22:32:47.055715Z","shell.execute_reply":"2023-03-12T22:32:47.149743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Transposing the describe dataframe to convert rows to columns\nnon_nan_df_defog.transpose().style.background_gradient(subset = [\"count\",\"unique\",\"top\",\"freq\",\"mean\",\"std\",\"min\",\"25%\",\"50%\",\"75%\",\"max\",\"var\",\"kurt\",\"skew\"], \n                             cmap = \"seismic\", \n                             vmin = -1, \n                             vmax = 1)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:47.152751Z","iopub.execute_input":"2023-03-12T22:32:47.153157Z","iopub.status.idle":"2023-03-12T22:32:47.279071Z","shell.execute_reply.started":"2023-03-12T22:32:47.153119Z","shell.execute_reply":"2023-03-12T22:32:47.277325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating tdcsfog dataframe using pandas","metadata":{}},{"cell_type":"code","source":"## Creating tdcdfog dataframe using pandas\nfile_path = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/04e10e0797.csv\"\ndf_tdcdfog = pd.read_csv(file_path,index_col = None, header = 0)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:47.281421Z","iopub.execute_input":"2023-03-12T22:32:47.282049Z","iopub.status.idle":"2023-03-12T22:32:47.319741Z","shell.execute_reply.started":"2023-03-12T22:32:47.281986Z","shell.execute_reply":"2023-03-12T22:32:47.318367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Information about tdcdfog dataset","metadata":{}},{"cell_type":"code","source":"# Printing useful information about datasets\ndf_tdcdfog.info()","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:47.326548Z","iopub.execute_input":"2023-03-12T22:32:47.327308Z","iopub.status.idle":"2023-03-12T22:32:47.342721Z","shell.execute_reply.started":"2023-03-12T22:32:47.327251Z","shell.execute_reply":"2023-03-12T22:32:47.341244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Showing top 10 values for tdcdfog datasets","metadata":{}},{"cell_type":"code","source":"# Pringing top 10 values of the tdcdfog dataset\ndf_tdcdfog.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:47.345742Z","iopub.execute_input":"2023-03-12T22:32:47.346980Z","iopub.status.idle":"2023-03-12T22:32:47.363800Z","shell.execute_reply.started":"2023-03-12T22:32:47.346922Z","shell.execute_reply":"2023-03-12T22:32:47.361932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Showing bottom 10 values for tdcdfog datasets","metadata":{}},{"cell_type":"code","source":"# Pringing bottm 10 values of the tdcdfog dataset\ndf_tdcdfog.tail(10)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:47.366077Z","iopub.execute_input":"2023-03-12T22:32:47.367627Z","iopub.status.idle":"2023-03-12T22:32:47.389479Z","shell.execute_reply.started":"2023-03-12T22:32:47.367571Z","shell.execute_reply":"2023-03-12T22:32:47.388084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Shape of the tdcdfog dataframe","metadata":{}},{"cell_type":"code","source":"# Finding the shape of the dataset\ndf_tdcdfog.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:47.391560Z","iopub.execute_input":"2023-03-12T22:32:47.392074Z","iopub.status.idle":"2023-03-12T22:32:47.399977Z","shell.execute_reply.started":"2023-03-12T22:32:47.392023Z","shell.execute_reply":"2023-03-12T22:32:47.398402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  Checking if there are any missing values in tdcdfog dataset ","metadata":{}},{"cell_type":"code","source":"# Checking if any of fields have missing data\ndf_tdcdfog.isnull().values.any().sum()","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:47.401842Z","iopub.execute_input":"2023-03-12T22:32:47.403697Z","iopub.status.idle":"2023-03-12T22:32:47.415236Z","shell.execute_reply.started":"2023-03-12T22:32:47.403470Z","shell.execute_reply":"2023-03-12T22:32:47.413850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Describing the tdcdfog datasets with skiwness and kurtosis","metadata":{}},{"cell_type":"code","source":"# Describing the datasets with variance, skewness and kurtosis\nstats_tdcdfog=df_tdcdfog.describe(include = 'all')\nstats_tdcdfog.loc['var'] = df_tdcdfog.var().tolist()\nstats_tdcdfog.loc['skew'] = df_tdcdfog.skew().tolist()\nstats_tdcdfog.loc['kurt'] = df_tdcdfog.kurtosis().tolist()\nnon_nan_df_tdcdfog=stats_tdcdfog.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:47.417038Z","iopub.execute_input":"2023-03-12T22:32:47.417501Z","iopub.status.idle":"2023-03-12T22:32:47.455944Z","shell.execute_reply.started":"2023-03-12T22:32:47.417463Z","shell.execute_reply":"2023-03-12T22:32:47.454596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Transposing the describe dataframe to convert rows to columns\nnon_nan_df_tdcdfog.transpose().style.background_gradient(subset = [\"count\",\"mean\",\"std\",\"min\",\"25%\",\"50%\",\"75%\",\"max\",\"var\",\"kurt\",\"skew\"], \n                             cmap = \"seismic\", \n                             vmin = -1, \n                             vmax = 1)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:47.457328Z","iopub.execute_input":"2023-03-12T22:32:47.458443Z","iopub.status.idle":"2023-03-12T22:32:47.493598Z","shell.execute_reply.started":"2023-03-12T22:32:47.458392Z","shell.execute_reply":"2023-03-12T22:32:47.492185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating notype dataframe using pandas","metadata":{}},{"cell_type":"code","source":"file_path = \"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/notype/46cdfe23ea.csv\"\ndf_notype= pd.read_csv(file_path,index_col = None, header = 0)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:47.495276Z","iopub.execute_input":"2023-03-12T22:32:47.495671Z","iopub.status.idle":"2023-03-12T22:32:47.787687Z","shell.execute_reply.started":"2023-03-12T22:32:47.495633Z","shell.execute_reply":"2023-03-12T22:32:47.786702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Information about notype dataset\ndf_notype.info()","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:47.789121Z","iopub.execute_input":"2023-03-12T22:32:47.789475Z","iopub.status.idle":"2023-03-12T22:32:47.806141Z","shell.execute_reply.started":"2023-03-12T22:32:47.789440Z","shell.execute_reply":"2023-03-12T22:32:47.804763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Showing top 10 values for notype datasets","metadata":{}},{"cell_type":"code","source":"# Showing the top 10 values of the datasets\ndf_notype.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:47.807969Z","iopub.execute_input":"2023-03-12T22:32:47.808424Z","iopub.status.idle":"2023-03-12T22:32:47.829424Z","shell.execute_reply.started":"2023-03-12T22:32:47.808376Z","shell.execute_reply":"2023-03-12T22:32:47.828513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Showing 10 values for notype datasets","metadata":{}},{"cell_type":"code","source":"# Showing the top 10 values of the datasets\ndf_notype.tail(10)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:47.830660Z","iopub.execute_input":"2023-03-12T22:32:47.831520Z","iopub.status.idle":"2023-03-12T22:32:47.846945Z","shell.execute_reply.started":"2023-03-12T22:32:47.831481Z","shell.execute_reply":"2023-03-12T22:32:47.845426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Shape of the notpye dataframe","metadata":{}},{"cell_type":"code","source":"# Pring the shape of the notype dataset\ndf_notype.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:47.848788Z","iopub.execute_input":"2023-03-12T22:32:47.849489Z","iopub.status.idle":"2023-03-12T22:32:47.861037Z","shell.execute_reply.started":"2023-03-12T22:32:47.849442Z","shell.execute_reply":"2023-03-12T22:32:47.859627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  Checking if there are any missing values in notype dataset ","metadata":{}},{"cell_type":"code","source":"# Checking if the dataset has any missing values \ndf_notype.isnull().values.any().sum()","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:47.862461Z","iopub.execute_input":"2023-03-12T22:32:47.862801Z","iopub.status.idle":"2023-03-12T22:32:47.872726Z","shell.execute_reply.started":"2023-03-12T22:32:47.862768Z","shell.execute_reply":"2023-03-12T22:32:47.871385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Describing the notype datasets with skiwness and kurtosis","metadata":{}},{"cell_type":"code","source":"# Describing the datasets with variance, skewness and kurtosis\nstats_notype=df_notype.describe(include = 'all')\nstats_notype.loc['var'] = df_notype.var().tolist()\nstats_notype.loc['skew'] = df_notype.skew().tolist()\nstats_notype.loc['kurt'] = df_notype.kurtosis().tolist()\nnon_nan_df_notype=stats_notype.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:47.874827Z","iopub.execute_input":"2023-03-12T22:32:47.875351Z","iopub.status.idle":"2023-03-12T22:32:47.961495Z","shell.execute_reply.started":"2023-03-12T22:32:47.875301Z","shell.execute_reply":"2023-03-12T22:32:47.960358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Transposing the describe dataframe to convert rows to columns\nnon_nan_df_notype.transpose().style.background_gradient(subset = [\"count\",\"mean\",\"std\",\"min\",\"25%\",\"50%\",\"75%\",\"max\",\"var\",\"kurt\",\"skew\"], \n                             cmap = \"seismic\", \n                             vmin = -1, \n                             vmax = 1)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:47.962848Z","iopub.execute_input":"2023-03-12T22:32:47.963750Z","iopub.status.idle":"2023-03-12T22:32:48.006454Z","shell.execute_reply.started":"2023-03-12T22:32:47.963699Z","shell.execute_reply":"2023-03-12T22:32:48.004994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating training datasets from all the csv files\n","metadata":{}},{"cell_type":"code","source":"# Creating DEFOG datasets using all the csv files \n\nroot_file_path='/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/'\nDEFOG = pd.DataFrame()\nfor root, dirs, files in os.walk(root_file_path):\n    for name in files:       \n        f = os.path.join(root, name)\n        df_list= pd.read_csv(f)\n        words = name.split('.')[0]\n        df_list['file']= name.split('.')[0]\n        DEFOG = pd.concat([DEFOG, df_list], axis=0)\n        \n\nDEFOG\n","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:32:48.008159Z","iopub.execute_input":"2023-03-12T22:32:48.009336Z","iopub.status.idle":"2023-03-12T22:33:42.961125Z","shell.execute_reply.started":"2023-03-12T22:32:48.009293Z","shell.execute_reply":"2023-03-12T22:33:42.959173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating TDCSFOG datasets using all the csv files \n\nroot_file_path='/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog'\nTDCSFOG = pd.DataFrame()\nfor root, dirs, files in os.walk(root_file_path):\n    for name in files:       \n        f = os.path.join(root, name)\n        df_list= pd.read_csv(f)\n        words = name.split('.')[0]\n        df_list['file']= name.split('.')[0]\n        TDCSFOG = pd.concat([TDCSFOG, df_list], axis=0)\n        \n\nTDCSFOG\n","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:33:42.962709Z","iopub.execute_input":"2023-03-12T22:33:42.963099Z","iopub.status.idle":"2023-03-12T22:36:04.959170Z","shell.execute_reply.started":"2023-03-12T22:33:42.963063Z","shell.execute_reply":"2023-03-12T22:36:04.957773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using DEFOG valid data for further analysis\nDEFOG = DEFOG[(DEFOG['Task']==1) & (DEFOG['Valid']==1)]","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:36:04.960956Z","iopub.execute_input":"2023-03-12T22:36:04.961317Z","iopub.status.idle":"2023-03-12T22:36:05.626930Z","shell.execute_reply.started":"2023-03-12T22:36:04.961285Z","shell.execute_reply":"2023-03-12T22:36:05.625037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reading defog metadata file and combining it with DEFOG dataset\ndefog_metadata = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv\")\ndefog_metadata","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:36:05.629072Z","iopub.execute_input":"2023-03-12T22:36:05.629481Z","iopub.status.idle":"2023-03-12T22:36:05.655189Z","shell.execute_reply.started":"2023-03-12T22:36:05.629444Z","shell.execute_reply":"2023-03-12T22:36:05.652983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Merging the DEFOG and defog_metadata datasets into one dataset\ndefog_merged= defog_metadata.merge(DEFOG, how = 'inner', left_on = 'Id', right_on = 'file')\ndefog_merged.drop(['file','Valid','Task'], axis = 1, inplace = True)\ndefog_merged","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:36:05.657507Z","iopub.execute_input":"2023-03-12T22:36:05.658034Z","iopub.status.idle":"2023-03-12T22:36:08.056578Z","shell.execute_reply.started":"2023-03-12T22:36:05.657994Z","shell.execute_reply":"2023-03-12T22:36:08.055385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reading tdcsfog_metadata dataset \ntdcsfog_metadata = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv\")\ntdcsfog_metadata","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:36:08.058319Z","iopub.execute_input":"2023-03-12T22:36:08.058689Z","iopub.status.idle":"2023-03-12T22:36:08.082688Z","shell.execute_reply.started":"2023-03-12T22:36:08.058654Z","shell.execute_reply":"2023-03-12T22:36:08.081289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Merging the TDCSFOG and tdcsfog_metadata datasets into one dataset\ntdcsfog_merged= tdcsfog_metadata.merge(TDCSFOG, how = 'inner', left_on = 'Id', right_on = 'file')\ntdcsfog_merged.drop(['file'], axis = 1, inplace = True)\ntdcsfog_merged","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:36:08.088543Z","iopub.execute_input":"2023-03-12T22:36:08.088958Z","iopub.status.idle":"2023-03-12T22:36:12.194602Z","shell.execute_reply.started":"2023-03-12T22:36:08.088907Z","shell.execute_reply":"2023-03-12T22:36:12.193295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering of Time Series data","metadata":{}},{"cell_type":"code","source":"# Importing training models packages\nfrom sklearn.model_selection import KFold, StratifiedKFold, train_test_split, GridSearchCV\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:36:12.196131Z","iopub.execute_input":"2023-03-12T22:36:12.197102Z","iopub.status.idle":"2023-03-12T22:36:12.343382Z","shell.execute_reply.started":"2023-03-12T22:36:12.197061Z","shell.execute_reply":"2023-03-12T22:36:12.342402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conditions = [\n    (defog_merged['StartHesitation'] == 1),\n    (defog_merged['Turn'] == 1),\n    (defog_merged['Walking'] == 1)]\nchoices = ['StartHesitation', 'Turn', 'Walking']\ndefog_merged['event'] = np.select(conditions, choices, default='Normal')","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:36:12.344781Z","iopub.execute_input":"2023-03-12T22:36:12.345973Z","iopub.status.idle":"2023-03-12T22:36:13.249730Z","shell.execute_reply.started":"2023-03-12T22:36:12.345921Z","shell.execute_reply":"2023-03-12T22:36:13.248367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating the traning dataframe\ntrain_df = defog_merged[['AccV','AccML','AccAP','event']]","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:36:13.251304Z","iopub.execute_input":"2023-03-12T22:36:13.252420Z","iopub.status.idle":"2023-03-12T22:36:13.991534Z","shell.execute_reply.started":"2023-03-12T22:36:13.252378Z","shell.execute_reply":"2023-03-12T22:36:13.990215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()\n\ntrain_df['target'] = le.fit_transform(train_df['event'])","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:36:13.993047Z","iopub.execute_input":"2023-03-12T22:36:13.993395Z","iopub.status.idle":"2023-03-12T22:36:15.140589Z","shell.execute_reply.started":"2023-03-12T22:36:13.993361Z","shell.execute_reply":"2023-03-12T22:36:15.138767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_df.drop(['event','target'], axis=1)\ny = train_df['target']","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:36:15.143264Z","iopub.execute_input":"2023-03-12T22:36:15.144830Z","iopub.status.idle":"2023-03-12T22:36:15.194955Z","shell.execute_reply.started":"2023-03-12T22:36:15.144758Z","shell.execute_reply":"2023-03-12T22:36:15.193137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Traning the LIGHTBGM model","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb\n\n\n# split dataset into training and test set\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=1004)\n\n#Converting the dataset in proper LGB format\nd_train=lgb.Dataset(X_train, label=y_train)\n#setting up the parameters\nparams={}\nparams['learning_rate']=0.03\nparams['boosting_type']='gbdt' #GradientBoostingDecisionTree\nparams['objective']='multiclass' #Multi-class target feature\nparams['metric']='multi_logloss' #metric for multi-class\nparams['max_depth']=7\nparams['num_class']=4 #no.of unique values in the target class not inclusive of the end value\nparams['verbose']=-1\n#training the model\nclf=lgb.train(params,d_train,1000)  #training the model on 1,000 epocs\n#prediction on the test dataset\ny_pred_1=clf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:36:15.196516Z","iopub.execute_input":"2023-03-12T22:36:15.196909Z","iopub.status.idle":"2023-03-12T22:44:54.536233Z","shell.execute_reply.started":"2023-03-12T22:36:15.196860Z","shell.execute_reply":"2023-03-12T22:44:54.534973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import precision_score\nprecision_score(y_test, np.argmax(y_pred_1, axis=-1), average='macro')","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:44:54.537776Z","iopub.execute_input":"2023-03-12T22:44:54.538138Z","iopub.status.idle":"2023-03-12T22:44:54.785384Z","shell.execute_reply.started":"2023-03-12T22:44:54.538104Z","shell.execute_reply":"2023-03-12T22:44:54.783991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preparing the test dataset for inferences","metadata":{}},{"cell_type":"code","source":"test_defog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog/02ab235146.csv'\ntest_defog = pd.read_csv(test_defog_path)\nname = os.path.basename(test_defog_path)\nid_value = name.split('.')[0]\ntest_defog['Id_value'] = id_value\ntest_defog['Id'] = test_defog['Id_value'].astype(str) + '_' + test_defog['Time'].astype(str)\ntest_defog = test_defog[['Id','AccV','AccML','AccAP']]\ntest_defog.set_index('Id',inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:45:59.825919Z","iopub.execute_input":"2023-03-12T22:45:59.827459Z","iopub.status.idle":"2023-03-12T22:46:00.593264Z","shell.execute_reply.started":"2023-03-12T22:45:59.827394Z","shell.execute_reply":"2023-03-12T22:46:00.591971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predicting even probablity\ntest_defog_pred=clf.predict(test_defog)\ntest_defog['event'] = np.argmax(test_defog_pred, axis=-1)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:46:03.814523Z","iopub.execute_input":"2023-03-12T22:46:03.814953Z","iopub.status.idle":"2023-03-12T22:46:34.631949Z","shell.execute_reply.started":"2023-03-12T22:46:03.814903Z","shell.execute_reply":"2023-03-12T22:46:34.630414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Explode the column event into three different columns\ntest_defog['StartHesitation'] = np.where(test_defog['event']==1, 1, 0)\ntest_defog['Turn'] = np.where(test_defog['event']==2, 1, 0)\ntest_defog['Walking'] = np.where(test_defog['event']==3, 1, 0)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:46:40.726108Z","iopub.execute_input":"2023-03-12T22:46:40.726526Z","iopub.status.idle":"2023-03-12T22:46:40.740024Z","shell.execute_reply.started":"2023-03-12T22:46:40.726487Z","shell.execute_reply":"2023-03-12T22:46:40.738876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_defog.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:46:43.408738Z","iopub.execute_input":"2023-03-12T22:46:43.410368Z","iopub.status.idle":"2023-03-12T22:46:43.431047Z","shell.execute_reply.started":"2023-03-12T22:46:43.410283Z","shell.execute_reply":"2023-03-12T22:46:43.429157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Performing the similar activity with tdcsfog dataset\ntest_tdcsfog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog/003f117e14.csv'\ntest_tdcsfog = pd.read_csv(test_tdcsfog_path)\nname = os.path.basename(test_tdcsfog_path)\nid_value = name.split('.')[0]\ntest_tdcsfog['Id_value'] = id_value\ntest_tdcsfog['Id'] = test_tdcsfog['Id_value'].astype(str) + '_' + test_tdcsfog['Time'].astype(str)\ntest_tdcsfog = test_tdcsfog[['Id','AccV','AccML','AccAP']]\ntest_tdcsfog.set_index('Id',inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:47:41.815860Z","iopub.execute_input":"2023-03-12T22:47:41.817831Z","iopub.status.idle":"2023-03-12T22:47:41.852734Z","shell.execute_reply.started":"2023-03-12T22:47:41.817732Z","shell.execute_reply":"2023-03-12T22:47:41.851541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tdcsfog_pred=clf.predict(test_tdcsfog)\ntest_tdcsfog['event'] = np.argmax(test_tdcsfog_pred, axis=-1)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:47:52.761102Z","iopub.execute_input":"2023-03-12T22:47:52.761575Z","iopub.status.idle":"2023-03-12T22:47:53.256829Z","shell.execute_reply.started":"2023-03-12T22:47:52.761532Z","shell.execute_reply":"2023-03-12T22:47:53.255878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tdcsfog['StartHesitation'] = np.where(test_tdcsfog['event']==1, 1, 0)\ntest_tdcsfog['Turn'] = np.where(test_tdcsfog['event']==2, 1, 0)\ntest_tdcsfog['Walking'] = np.where(test_tdcsfog['event']==3, 1, 0)\ntest_tdcsfog.reset_index('Id', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:48:04.877393Z","iopub.execute_input":"2023-03-12T22:48:04.877794Z","iopub.status.idle":"2023-03-12T22:48:04.888558Z","shell.execute_reply.started":"2023-03-12T22:48:04.877760Z","shell.execute_reply":"2023-03-12T22:48:04.887601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tdcsfog.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:48:21.018107Z","iopub.execute_input":"2023-03-12T22:48:21.019115Z","iopub.status.idle":"2023-03-12T22:48:21.042072Z","shell.execute_reply.started":"2023-03-12T22:48:21.019044Z","shell.execute_reply":"2023-03-12T22:48:21.040612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating the similar dataset for  submission.csv\nsubmit = pd.concat([test_tdcsfog,test_defog])\nsubmit = submit[['Id', 'StartHesitation', 'Turn','Walking']]","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:48:42.332423Z","iopub.execute_input":"2023-03-12T22:48:42.332953Z","iopub.status.idle":"2023-03-12T22:48:42.382857Z","shell.execute_reply.started":"2023-03-12T22:48:42.332874Z","shell.execute_reply":"2023-03-12T22:48:42.380856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:48:53.945267Z","iopub.execute_input":"2023-03-12T22:48:53.945803Z","iopub.status.idle":"2023-03-12T22:48:53.960623Z","shell.execute_reply.started":"2023-03-12T22:48:53.945744Z","shell.execute_reply":"2023-03-12T22:48:53.959124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Exporting the summsion data \nsubmit.to_csv('submission.csv', index=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:51:54.221062Z","iopub.execute_input":"2023-03-12T22:51:54.222395Z","iopub.status.idle":"2023-03-12T22:51:54.562073Z","shell.execute_reply.started":"2023-03-12T22:51:54.222332Z","shell.execute_reply":"2023-03-12T22:51:54.560727Z"},"trusted":true},"execution_count":null,"outputs":[]}]}