{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os\nfrom os import listdir\nfrom os.path import isfile, join \nimport random\nimport cv2\n\nfrom sklearn.model_selection import train_test_split \nfrom sklearn.ensemble import BaggingClassifier\nfrom sklearn import tree\nfrom sklearn.metrics import confusion_matrix","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-08T15:20:28.048123Z","iopub.execute_input":"2023-06-08T15:20:28.048468Z","iopub.status.idle":"2023-06-08T15:20:29.085745Z","shell.execute_reply.started":"2023-06-08T15:20:28.048444Z","shell.execute_reply":"2023-06-08T15:20:29.084765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Problem Scope**\n\nParkinson's Disease (PD) is a neurodegenerative disease that is characterized by the loss of dopaminergic neurons in the midbrain structures. A common hallmark of this disease is Freezing of Gait (FOG) which is a brief episodic absence or marked reduction of forward progression of the fet despite the intention to walk. PD pathology at early stages starts at the dorsal motor nucleus of the vagus nerve, intermediate reticular zone and the olfactory bulb and progresses to affect the midbrain SNr dopamine cells and the cerebral cortex. It is hypothesized that FOG occurs because of the degeneration of the brainstem cholinergic locomotor centres including the Pedunculopontine Nucleus (PPN) which is found to be responsible for controlling \"well-coordinated locomotion\" in animal models. The brainstem cholinergic locmotor centrs including the mesencephalic locomotor region and projects to the reticulospinal neurons. \n\n**FOG Prediction Project**\n\nThis project involves using data collected by an acceleromter located in the lumbar spine and trying to predict not only instances of FOG vs. non-FOG walking but also to classify three different types of FOG within this class. The three different classes to identify are: Walking, Turning, and Start Hesitation. \n\nTo create my model I first filtered the accel. data using a butterworth filter and then created my own features derived from the raw accel. data. These features include the integrated Accel. data, and calculating the standard deviation and the mean of the time-series data using a rolling window of 250 frames.","metadata":{}},{"cell_type":"code","source":"#Directory\ndefog_file = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/'\ndefog = pd.DataFrame()\nfor root, dirs, files in os.walk(defog_file):\n    for name in files:       \n        f = os.path.join(root, name)\n        df_list= pd.read_csv(f)\n        words = name.split('.')[0]\n        df_list['file']= name.split('.')[0]\n        defog = pd.concat([defog, df_list], axis=0)\ndefog\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:20:30.950437Z","iopub.execute_input":"2023-06-08T15:20:30.950788Z","iopub.status.idle":"2023-06-08T15:20:59.740550Z","shell.execute_reply.started":"2023-06-08T15:20:30.950763Z","shell.execute_reply":"2023-06-08T15:20:59.739461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#setup defog\nfilt_defog = defog[(defog['Valid'] == True) & (defog['Task'] == True)]\nfilt_defog","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:21:03.018902Z","iopub.execute_input":"2023-06-08T15:21:03.019222Z","iopub.status.idle":"2023-06-08T15:21:03.347573Z","shell.execute_reply.started":"2023-06-08T15:21:03.019197Z","shell.execute_reply":"2023-06-08T15:21:03.346518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# copy filt_defog to create class\n\nfilt_defog_copy = filt_defog.copy()  # Create a copy of the DataFrame\n\n# Initialize 'Class' column with 0\nfilt_defog_copy['Class'] = 0\n\n# Set 'Class' values based on conditions\nfilt_defog_copy.loc[filt_defog_copy['Turn'] == 1, 'Class'] = 1\nfilt_defog_copy.loc[filt_defog_copy['Walking'] == 1, 'Class'] = 2\nfilt_defog_copy.loc[filt_defog_copy['StartHesitation'] == 1, 'Class'] = 3\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:21:06.475126Z","iopub.execute_input":"2023-06-08T15:21:06.475520Z","iopub.status.idle":"2023-06-08T15:21:06.549921Z","shell.execute_reply.started":"2023-06-08T15:21:06.475491Z","shell.execute_reply":"2023-06-08T15:21:06.549024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# finalize defog\ndf_defog = filt_defog_copy[['AccV','AccML','AccAP','Class','file']]\ndf_defog","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:21:08.884066Z","iopub.execute_input":"2023-06-08T15:21:08.884467Z","iopub.status.idle":"2023-06-08T15:21:09.016883Z","shell.execute_reply.started":"2023-06-08T15:21:08.884443Z","shell.execute_reply":"2023-06-08T15:21:09.015704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# butteworth filter 2nd order with 15hz cutoff \nfrom scipy.signal import butter, filtfilt\n\ncutoff_freq = 15 \nsampling_rate = 100  \n\n# Define the filter parameters\norder = 2\nnyquist_freq = 0.5 * sampling_rate\nnormalized_cutoff_freq = cutoff_freq / nyquist_freq\n\n# Create the Butterworth filter coefficients\nb, a = butter(order, normalized_cutoff_freq, btype='low', analog=False)\n\n# Apply the filter to each column in the dataframe\ndf_filtered = pd.DataFrame()\nfor column in df_defog[['AccV','AccML','AccAP']].columns:\n    filtered_data = filtfilt(b, a, df_defog[column])\n    df_filtered[column] = filtered_data","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:21:11.569986Z","iopub.execute_input":"2023-06-08T15:21:11.570287Z","iopub.status.idle":"2023-06-08T15:21:12.006571Z","shell.execute_reply.started":"2023-06-08T15:21:11.570264Z","shell.execute_reply":"2023-06-08T15:21:12.005618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculate more features\n# integrate over a 250 sample window\n# get stdev and means of the windows \n\nwindow_size = 250  # Number of rows in each rolling window\n\n# Calculate the integral using rolling sum\ndf_filtered['Integral_AccV'] = df_filtered['AccV'].rolling(window_size, min_periods=1).sum()\ndf_filtered['Integral_AccML'] = df_filtered['AccML'].rolling(window_size, min_periods=1).sum()\ndf_filtered['Integral_AccAP'] = df_filtered['AccAP'].rolling(window_size, min_periods=1).sum()\n\ndf_filtered['StdDev_AccV'] = df_filtered['AccV'].rolling(window_size, min_periods=1).std()\ndf_filtered['StdDev_AccML'] = df_filtered['AccML'].rolling(window_size, min_periods=1).std()\ndf_filtered['StdDev_AccAP'] = df_filtered['AccAP'].rolling(window_size, min_periods=1).std()\n\ndf_filtered['Mean_AccV'] = df_filtered['AccV'].rolling(window_size, min_periods=1).mean()\ndf_filtered['Mean_AccML'] = df_filtered['AccML'].rolling(window_size, min_periods=1).mean()\ndf_filtered['Mean_AccAP'] = df_filtered['AccAP'].rolling(window_size, min_periods=1).mean()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:21:14.121766Z","iopub.execute_input":"2023-06-08T15:21:14.122090Z","iopub.status.idle":"2023-06-08T15:21:14.830798Z","shell.execute_reply.started":"2023-06-08T15:21:14.122065Z","shell.execute_reply":"2023-06-08T15:21:14.829999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# append class and file ?\ndf_filtered = df_filtered.reset_index(drop=True)\ndf_defog = df_defog.reset_index(drop=True)\n\nappend_columns = df_defog[['Class', 'file']]\ndf_filtered_defog = pd.concat([df_filtered, append_columns], axis = 1)\n\ndf_filtered_defog = df_filtered_defog.dropna()\n\ndf_filtered_defog","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:21:17.413751Z","iopub.execute_input":"2023-06-08T15:21:17.414083Z","iopub.status.idle":"2023-06-08T15:21:18.596569Z","shell.execute_reply.started":"2023-06-08T15:21:17.414058Z","shell.execute_reply":"2023-06-08T15:21:18.595603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# setup tdcsfog \ntdcs_file = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/'\ntdcsfog = pd.DataFrame()\nfor root, dirs, files in os.walk(tdcs_file):\n    for name in files:       \n        f = os.path.join(root, name)\n        df_list= pd.read_csv(f)\n        words = name.split('.')[0]\n        df_list['file']= name.split('.')[0]\n        tdcsfog = pd.concat([tdcsfog, df_list], axis=0)\ntdcsfog","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:21:23.250273Z","iopub.execute_input":"2023-06-08T15:21:23.250643Z","iopub.status.idle":"2023-06-08T15:22:26.371876Z","shell.execute_reply.started":"2023-06-08T15:21:23.250617Z","shell.execute_reply":"2023-06-08T15:22:26.370749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcs_copy = tdcsfog.copy()  # Create a copy of the DataFrame\n\n# Initialize 'Class' column with 0\ntdcs_copy['Class'] = 0\n\n# Set 'Class' values based on conditions\ntdcs_copy.loc[tdcs_copy['Turn'] == 1, 'Class'] = 1\ntdcs_copy.loc[tdcs_copy['Walking'] == 1, 'Class'] = 2\ntdcs_copy.loc[tdcs_copy['StartHesitation'] == 1, 'Class'] = 3\n","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:22:34.652455Z","iopub.execute_input":"2023-06-08T15:22:34.653364Z","iopub.status.idle":"2023-06-08T15:22:34.858987Z","shell.execute_reply.started":"2023-06-08T15:22:34.653334Z","shell.execute_reply":"2023-06-08T15:22:34.857376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# finalize tdcs\ndf_tdcs = tdcs_copy[['AccV','AccML','AccAP','Class','file']]\nprint(len(df_tdcs))\ndf_tdcs.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:22:36.882855Z","iopub.execute_input":"2023-06-08T15:22:36.883162Z","iopub.status.idle":"2023-06-08T15:22:37.102182Z","shell.execute_reply.started":"2023-06-08T15:22:36.883139Z","shell.execute_reply":"2023-06-08T15:22:37.100839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# butteworth filter 2nd order with 15hz cutoff \nfrom scipy.signal import butter, filtfilt\n\ncutoff_freq = 15 \nsampling_rate = 128  \n\n# Define the filter parameters\norder = 2\nnyquist_freq = 0.5 * sampling_rate\nnormalized_cutoff_freq = cutoff_freq / nyquist_freq\n\n# Create the Butterworth filter coefficients\nb, a = butter(order, normalized_cutoff_freq, btype='low', analog=False)\n\n# Apply the filter to each column in the dataframe\ndf_filtered2 = pd.DataFrame()\nfor column in df_tdcs[['AccV','AccML','AccAP']].columns:\n    filtered_data = filtfilt(b, a, df_tdcs[column])\n    df_filtered2[column] = filtered_data","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:22:40.978520Z","iopub.execute_input":"2023-06-08T15:22:40.978828Z","iopub.status.idle":"2023-06-08T15:22:41.710426Z","shell.execute_reply.started":"2023-06-08T15:22:40.978806Z","shell.execute_reply":"2023-06-08T15:22:41.703389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculate more features\n# integrate over a 250 sample window\n# get stdev and means of the windows \n\nwindow_size = 250  # Number of rows in each rolling window\nstep_size = window_size // 2\n\n# Calculate the integral using rolling sum\ndf_filtered2['Integral_AccV'] = df_filtered2['AccV'].rolling(window_size, min_periods=1).sum()\ndf_filtered2['Integral_AccML'] = df_filtered2['AccML'].rolling(window_size, min_periods=1).sum()\ndf_filtered2['Integral_AccAP'] = df_filtered2['AccAP'].rolling(window_size, min_periods=1).sum()\n\ndf_filtered2['StdDev_AccV'] = df_filtered2['AccV'].rolling(window_size, min_periods=1).std()\ndf_filtered2['StdDev_AccML'] = df_filtered2['AccML'].rolling(window_size, min_periods=1).std()\ndf_filtered2['StdDev_AccAP'] = df_filtered2['AccAP'].rolling(window_size, min_periods=1).std()\n\ndf_filtered2['Mean_AccV'] = df_filtered2['AccV'].rolling(window_size, min_periods=1).mean()\ndf_filtered2['Mean_AccML'] = df_filtered2['AccML'].rolling(window_size, min_periods=1).mean()\ndf_filtered2['Mean_AccAP'] = df_filtered2['AccAP'].rolling(window_size, min_periods=1).mean()\n\nprint(len(df_filtered2))","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:22:45.179853Z","iopub.execute_input":"2023-06-08T15:22:45.180189Z","iopub.status.idle":"2023-06-08T15:22:46.469400Z","shell.execute_reply.started":"2023-06-08T15:22:45.180164Z","shell.execute_reply":"2023-06-08T15:22:46.468128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_filtered2 = df_filtered2.reset_index(drop=True)\ndf_tdcs = df_tdcs.reset_index(drop=True)\n\nappend_columns = df_tdcs[['Class', 'file']]\ndf_filtered_tdcs = pd.concat([df_filtered2, append_columns], axis = 1)\n\n\ndf_filtered_tdcs = df_filtered_tdcs.dropna()\n\ndf_filtered_tdcs","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:22:51.399722Z","iopub.execute_input":"2023-06-08T15:22:51.400033Z","iopub.status.idle":"2023-06-08T15:22:53.600349Z","shell.execute_reply.started":"2023-06-08T15:22:51.400010Z","shell.execute_reply":"2023-06-08T15:22:53.599378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_filtered_tdcs.iloc[0:20000, 3:6].plot.line()\ndf_filtered_tdcs.iloc[0:20000, 0:3].plot.line()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:23:03.173212Z","iopub.execute_input":"2023-06-08T15:23:03.173869Z","iopub.status.idle":"2023-06-08T15:23:04.233200Z","shell.execute_reply.started":"2023-06-08T15:23:03.173831Z","shell.execute_reply":"2023-06-08T15:23:04.232133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# merge tdcs and defog as df\n\ndata = [df_filtered_defog, df_filtered_tdcs]\n\ndf = pd.concat(data)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:23:09.102276Z","iopub.execute_input":"2023-06-08T15:23:09.102768Z","iopub.status.idle":"2023-06-08T15:23:09.454365Z","shell.execute_reply.started":"2023-06-08T15:23:09.102734Z","shell.execute_reply":"2023-06-08T15:23:09.453434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Total rows in combined df:\", len(df))\nprint(df.head())","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:23:12.332311Z","iopub.execute_input":"2023-06-08T15:23:12.332689Z","iopub.status.idle":"2023-06-08T15:23:12.343879Z","shell.execute_reply.started":"2023-06-08T15:23:12.332662Z","shell.execute_reply":"2023-06-08T15:23:12.343020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot class counts \npd.value_counts(df['Class']).plot.bar()\nplt.title('Freq histogram')\nplt.xlabel('Class')\nplt.ylabel('Frequency')\ndf['Class'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:23:15.393149Z","iopub.execute_input":"2023-06-08T15:23:15.393513Z","iopub.status.idle":"2023-06-08T15:23:15.886259Z","shell.execute_reply.started":"2023-06-08T15:23:15.393489Z","shell.execute_reply":"2023-06-08T15:23:15.885373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grouped_df = df.groupby('file')['Class'].sum()\n\n# Step 2: Get the list of participants with a sum of 'Class' greater than 0\nparticipants_with_sum_greater_than_zero = grouped_df[grouped_df > 0].index.tolist()\n\n# Step 3: Filter the original dataframe based on the selected participants\nfiltered_df = df[df['file'].isin(participants_with_sum_greater_than_zero)]\n\nfiltered_df","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:23:19.281569Z","iopub.execute_input":"2023-06-08T15:23:19.282023Z","iopub.status.idle":"2023-06-08T15:23:20.913071Z","shell.execute_reply.started":"2023-06-08T15:23:19.281984Z","shell.execute_reply":"2023-06-08T15:23:20.912078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# drop filename \ndf_final = filtered_df.drop('file', axis = 1, inplace = False)\ndf_final","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:23:24.522673Z","iopub.execute_input":"2023-06-08T15:23:24.523316Z","iopub.status.idle":"2023-06-08T15:23:24.700033Z","shell.execute_reply.started":"2023-06-08T15:23:24.523286Z","shell.execute_reply":"2023-06-08T15:23:24.699368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create X and y for model \nX = np.array(df_final.loc[:, df_final.columns != 'Class'])\ny = np.array(df_final.loc[:, df_final.columns == 'Class'])\nprint('Shape of X: {}'.format(X.shape))\nprint('Shape of y: {}'.format(y.shape))","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:23:27.745085Z","iopub.execute_input":"2023-06-08T15:23:27.745450Z","iopub.status.idle":"2023-06-08T15:23:28.038563Z","shell.execute_reply.started":"2023-06-08T15:23:27.745424Z","shell.execute_reply":"2023-06-08T15:23:28.037381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import SMOTE and Tomek packages \nfrom imblearn.combine import SMOTETomek\nfrom imblearn.under_sampling import TomekLinks","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:23:31.365257Z","iopub.execute_input":"2023-06-08T15:23:31.365653Z","iopub.status.idle":"2023-06-08T15:23:31.548956Z","shell.execute_reply.started":"2023-06-08T15:23:31.365624Z","shell.execute_reply":"2023-06-08T15:23:31.548001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# oversample minority class, undersample majority class \nresample = SMOTETomek(tomek=TomekLinks(sampling_strategy='majority'))\nX,y = resample.fit_resample(X,y)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:23:35.497735Z","iopub.execute_input":"2023-06-08T15:23:35.498095Z","iopub.status.idle":"2023-06-08T15:30:33.718836Z","shell.execute_reply.started":"2023-06-08T15:23:35.498065Z","shell.execute_reply":"2023-06-08T15:30:33.718112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split data into seq lengths of 200 each and then randomize them \nindices = np.arange(len(X))\nsequences = [indices[i:i+200] for i in range(0, len(X), 200)]\nnp.random.shuffle(sequences)\nshuffled_indices = np.concatenate(sequences)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:33:18.589587Z","iopub.execute_input":"2023-06-08T15:33:18.589925Z","iopub.status.idle":"2023-06-08T15:33:18.814291Z","shell.execute_reply.started":"2023-06-08T15:33:18.589901Z","shell.execute_reply":"2023-06-08T15:33:18.812775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_x, val_x, train_y, val_y = train_test_split(X[shuffled_indices], y[shuffled_indices], test_size=0.1)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:33:22.128041Z","iopub.execute_input":"2023-06-08T15:33:22.128409Z","iopub.status.idle":"2023-06-08T15:33:26.410551Z","shell.execute_reply.started":"2023-06-08T15:33:22.128384Z","shell.execute_reply":"2023-06-08T15:33:26.409567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# initialize train and test sets \n#train_x, val_x, train_y, val_y = train_test_split(X, y, test_size = 0.3)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# use bagging classifier as default for now \nmodel = BaggingClassifier(tree.DecisionTreeClassifier(random_state = 1))\nmodel.fit(train_x, train_y)\nprint(model.score(val_x, val_y))","metadata":{"execution":{"iopub.status.busy":"2023-06-08T15:33:32.045272Z","iopub.execute_input":"2023-06-08T15:33:32.045623Z","iopub.status.idle":"2023-06-08T18:14:50.288267Z","shell.execute_reply.started":"2023-06-08T15:33:32.045599Z","shell.execute_reply":"2023-06-08T18:14:50.285990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predict on train set and test values \ny_pred = model.predict(val_x)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:17:03.708392Z","iopub.execute_input":"2023-06-08T18:17:03.709630Z","iopub.status.idle":"2023-06-08T18:17:20.126364Z","shell.execute_reply.started":"2023-06-08T18:17:03.709589Z","shell.execute_reply":"2023-06-08T18:17:20.125579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conf_mat = confusion_matrix(y_true=val_y, y_pred=y_pred)\nprint('Confusion matrix with changes:' + ':\\n', conf_mat)\n\nlabels = ['Class 0', 'Class 1', 'Class 2', 'Class3']\nfig = plt.figure()\nax = fig.add_subplot(111)\ncax = ax.matshow(conf_mat, cmap=plt.cm.Blues)\nfig.colorbar(cax)\nax.set_xticklabels([''] + labels)\nax.set_yticklabels([''] + labels)\nplt.xlabel('Predicted')\nplt.ylabel('Expected')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:17:58.726939Z","iopub.execute_input":"2023-06-08T18:17:58.727292Z","iopub.status.idle":"2023-06-08T18:17:59.253387Z","shell.execute_reply.started":"2023-06-08T18:17:58.727264Z","shell.execute_reply":"2023-06-08T18:17:59.252673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"csvprint = pd.DataFrame(print(conf_mat))\ncsvprint.to_csv('/kaggle/working/result.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:18:07.213639Z","iopub.execute_input":"2023-06-08T18:18:07.214014Z","iopub.status.idle":"2023-06-08T18:18:07.228507Z","shell.execute_reply.started":"2023-06-08T18:18:07.213984Z","shell.execute_reply":"2023-06-08T18:18:07.227424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# initialize final test files \n\ntest_defog_file = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog/'\ntest_defog = pd.DataFrame()\nfor root, dirs, files in os.walk(test_defog_file):\n    for name in files:       \n        f = os.path.join(root, name)\n        df_list= pd.read_csv(f)\n        words = name.split('.')[0]\n        df_list['file']= name.split('.')[0]\n        test_defog = pd.concat([test_defog, df_list], axis=0)\ntest_defog\n\nprint(len(test_defog))\n\ntest_tdcs_file = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog/'\ntest_tdcs = pd.DataFrame()\nfor root, dirs, files in os.walk(test_tdcs_file):\n    for name in files:       \n        f = os.path.join(root, name)\n        df_list= pd.read_csv(f)\n        words = name.split('.')[0]\n        df_list['file']= name.split('.')[0]\n        test_tdcs = pd.concat([test_tdcs, df_list], axis=0)\ntest_tdcs\n\nprint(len(test_tdcs))","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:33:40.157215Z","iopub.execute_input":"2023-06-08T18:33:40.157584Z","iopub.status.idle":"2023-06-08T18:33:40.322586Z","shell.execute_reply.started":"2023-06-08T18:33:40.157558Z","shell.execute_reply":"2023-06-08T18:33:40.321693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#setup test_defog with butterworthfilter and new features\ncutoff_freq = 15 \nsampling_rate = 100  \n\n# Define the filter parameters\norder = 2\nnyquist_freq = 0.5 * sampling_rate\nnormalized_cutoff_freq = cutoff_freq / nyquist_freq\n\n# Create the Butterworth filter coefficients\nb, a = butter(order, normalized_cutoff_freq, btype='low', analog=False)\n\n# Apply the filter to each column in the dataframe\ntest_df_filtered = pd.DataFrame()\nfor column in test_defog[['AccV','AccML','AccAP']].columns:\n    filtered_data = filtfilt(b, a, test_defog[column])\n    test_df_filtered[column] = filtered_data\n\n# Create features\nwindow_size = 250  # Number of rows in each rolling window\n\n# Calculate the integral using rolling sum\ntest_df_filtered['Integral_AccV'] = test_df_filtered['AccV'].rolling(window_size, min_periods=1).sum()\ntest_df_filtered['Integral_AccML'] = test_df_filtered['AccML'].rolling(window_size, min_periods=1).sum()\ntest_df_filtered['Integral_AccAP'] = test_df_filtered['AccAP'].rolling(window_size, min_periods=1).sum()\n\ntest_df_filtered['StdDev_AccV'] = test_df_filtered['AccV'].rolling(window_size, min_periods=1).std()\ntest_df_filtered['StdDev_AccML'] = test_df_filtered['AccML'].rolling(window_size, min_periods=1).std()\ntest_df_filtered['StdDev_AccAP'] = test_df_filtered['AccAP'].rolling(window_size, min_periods=1).std()\n\ntest_df_filtered['Mean_AccV'] = test_df_filtered['AccV'].rolling(window_size, min_periods=1).mean()\ntest_df_filtered['Mean_AccML'] = test_df_filtered['AccML'].rolling(window_size, min_periods=1).mean()\ntest_df_filtered['Mean_AccAP'] = test_df_filtered['AccAP'].rolling(window_size, min_periods=1).mean()\n\n# appendback filename \ntest_df_filtered = test_df_filtered.reset_index(drop=True)\ntest_defog = test_defog.reset_index(drop=True)\n\nappend_columns = test_defog[['file']]\ntest_defog_final = pd.concat([test_df_filtered, append_columns], axis = 1)\n\n\ntest_defog_final = test_defog_final.fillna(method = 'backfill', inplace = False)\n\ntest_defog_final\n\nprint(len(test_defog), len(test_defog_final))","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:33:42.506300Z","iopub.execute_input":"2023-06-08T18:33:42.506742Z","iopub.status.idle":"2023-06-08T18:33:42.709685Z","shell.execute_reply.started":"2023-06-08T18:33:42.506712Z","shell.execute_reply":"2023-06-08T18:33:42.708521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#setup test_tdcs with filter and features\n#setup test_tdcs with butterworthfilter and new features\ncutoff_freq = 15 \nsampling_rate = 128 \n\n# Define the filter parameters\norder = 2\nnyquist_freq = 0.5 * sampling_rate\nnormalized_cutoff_freq = cutoff_freq / nyquist_freq\n\n# Create the Butterworth filter coefficients\nb, a = butter(order, normalized_cutoff_freq, btype='low', analog=False)\n\n# Apply the filter to each column in the dataframe\ntest_df_filtered2 = pd.DataFrame()\nfor column in test_tdcs[['AccV','AccML','AccAP']].columns:\n    filtered_data = filtfilt(b, a, test_tdcs[column])\n    test_df_filtered2[column] = filtered_data\n\n# Create features\nwindow_size = 250  # Number of rows in each rolling window\n\n# Calculate the integral using rolling sum\ntest_df_filtered2['Integral_AccV'] = test_df_filtered2['AccV'].rolling(window_size, min_periods=1).sum()\ntest_df_filtered2['Integral_AccML'] = test_df_filtered2['AccML'].rolling(window_size, min_periods=1).sum()\ntest_df_filtered2['Integral_AccAP'] = test_df_filtered2['AccAP'].rolling(window_size, min_periods=1).sum()\n\ntest_df_filtered2['StdDev_AccV'] = test_df_filtered2['AccV'].rolling(window_size, min_periods=1).std()\ntest_df_filtered2['StdDev_AccML'] = test_df_filtered2['AccML'].rolling(window_size, min_periods=1).std()\ntest_df_filtered2['StdDev_AccAP'] = test_df_filtered2['AccAP'].rolling(window_size, min_periods=1).std()\n\ntest_df_filtered2['Mean_AccV'] = test_df_filtered2['AccV'].rolling(window_size, min_periods=1).mean()\ntest_df_filtered2['Mean_AccML'] = test_df_filtered2['AccML'].rolling(window_size, min_periods=1).mean()\ntest_df_filtered2['Mean_AccAP'] = test_df_filtered2['AccAP'].rolling(window_size, min_periods=1).mean()\n\n# appendback filename \ntest_df_filtered2 = test_df_filtered2.reset_index(drop=True)\ntest_tdcs = test_tdcs.reset_index(drop=True)\n\nappend_columns = test_tdcs[['file']]\ntest_tdcs_final = pd.concat([test_df_filtered2, append_columns], axis = 1)\n\n\ntest_tdcs_final = test_tdcs_final.fillna(method = 'backfill', inplace = False)\n\ntest_tdcs_final\n\nprint(len(test_tdcs), len(test_tdcs_final))","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:33:45.461828Z","iopub.execute_input":"2023-06-08T18:33:45.463151Z","iopub.status.idle":"2023-06-08T18:33:45.487805Z","shell.execute_reply.started":"2023-06-08T18:33:45.463102Z","shell.execute_reply":"2023-06-08T18:33:45.486829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# finalize test set as test_final\ndata3 = [test_defog_final, test_tdcs_final]\n\ntest_final = pd.concat(data3)\n\nprint(len(test_final), test_final.shape)\ntest_final","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:33:48.237786Z","iopub.execute_input":"2023-06-08T18:33:48.238177Z","iopub.status.idle":"2023-06-08T18:33:48.265264Z","shell.execute_reply.started":"2023-06-08T18:33:48.238148Z","shell.execute_reply":"2023-06-08T18:33:48.263920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create an Id and time dataframe for submission \nId_test = test_final['file']\n\nId_test = pd.DataFrame(Id_test)\n\nId_test.head()\n\ntest_final2 = test_final.copy()\n\nTime_test = test_final2.groupby('file').cumcount()\nTime_test\n\nTime_test_df = pd.DataFrame(Time_test)\nTime_test_df\n\ndata_2 = [Id_test, Time_test_df]\nId_time_test = pd.concat(data_2, axis = 1)\n\nId_time_test.reset_index(drop=True, inplace=True)\nId_time_test","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:33:50.684858Z","iopub.execute_input":"2023-06-08T18:33:50.685237Z","iopub.status.idle":"2023-06-08T18:33:50.740687Z","shell.execute_reply.started":"2023-06-08T18:33:50.685208Z","shell.execute_reply":"2023-06-08T18:33:50.740031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#%%\ntest_final = test_final.drop('file', axis = 1, inplace = False)\n#test_final = test_final.drop('Time', axis = 1, inplace = False)\n\n#%%\ntest_final_x = np.array(test_final.loc[:, test_final.columns != 'Class'])","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:33:55.992811Z","iopub.execute_input":"2023-06-08T18:33:55.993204Z","iopub.status.idle":"2023-06-08T18:33:56.013413Z","shell.execute_reply.started":"2023-06-08T18:33:55.993174Z","shell.execute_reply":"2023-06-08T18:33:56.012072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_test_new = model.predict(test_final_x)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:33:58.147044Z","iopub.execute_input":"2023-06-08T18:33:58.148289Z","iopub.status.idle":"2023-06-08T18:33:58.435964Z","shell.execute_reply.started":"2023-06-08T18:33:58.148243Z","shell.execute_reply":"2023-06-08T18:33:58.434935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Id_time_test['Id'] = Id_time_test.iloc[:, 0].astype(str) + \"_\" + Id_time_test.iloc[:, 1].astype(str)\n\n# Display the updated dataframe\nprint(Id_time_test)\n\n# concat class predictions \n\n#%%\ndf_preds = pd.DataFrame(y_pred_test_new)\ndf_id = pd.DataFrame(Id_time_test.iloc[:,2])\n\nsubmission_prep = pd.concat([df_id, df_preds], axis = 1)\nsubmission_prep.columns = ['Id','Class']\n#%%\n\nsubmission_prep['StartHesitation'] = 0\nsubmission_prep['Turn'] = 0\nsubmission_prep['Walking'] = 0\n\nsubmission_prep.loc[submission_prep['Class'] == 3, 'StartHesitation'] = 1\nsubmission_prep.loc[submission_prep['Class'] == 1, 'Turn'] = 1\nsubmission_prep.loc[submission_prep['Class'] == 2, 'Walking'] = 1","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:34:00.296430Z","iopub.execute_input":"2023-06-08T18:34:00.296793Z","iopub.status.idle":"2023-06-08T18:34:00.476536Z","shell.execute_reply.started":"2023-06-08T18:34:00.296769Z","shell.execute_reply":"2023-06-08T18:34:00.475088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/sample_submission.csv\")\nsample_submission.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:34:03.663844Z","iopub.execute_input":"2023-06-08T18:34:03.664313Z","iopub.status.idle":"2023-06-08T18:34:03.807844Z","shell.execute_reply.started":"2023-06-08T18:34:03.664278Z","shell.execute_reply":"2023-06-08T18:34:03.806916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = submission_prep.drop('Class', axis = 1, inplace = False)\n\nprint(submission.shape)\n\nsubmission.to_csv('/kaggle/working/submission.csv', index = False)\nsubmission.to_csv(\"submission.csv\", index = False)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-08T20:19:40.639718Z","iopub.execute_input":"2023-06-08T20:19:40.640095Z","iopub.status.idle":"2023-06-08T20:19:41.054451Z","shell.execute_reply.started":"2023-06-08T20:19:40.640064Z","shell.execute_reply":"2023-06-08T20:19:41.053230Z"},"trusted":true},"execution_count":null,"outputs":[]}]}