{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true,"outputs_hidden":true},"execution":{"iopub.status.busy":"2023-06-04T15:03:05.409107Z","iopub.execute_input":"2023-06-04T15:03:05.409510Z","iopub.status.idle":"2023-06-04T15:03:05.600975Z","shell.execute_reply.started":"2023-06-04T15:03:05.409479Z","shell.execute_reply":"2023-06-04T15:03:05.599747Z"},"collapsed":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nfrom mpl_toolkits.mplot3d import Axes3D\nimport numpy as np\nimport os\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.svm import SVC\nfrom sklearn.metrics import f1_score\nfrom sklearn.metrics import roc_auc_score\nimport glob\nfrom sklearn.ensemble import RandomForestClassifier\nfrom scipy.stats import mannwhitneyu","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:05.603238Z","iopub.execute_input":"2023-06-04T15:03:05.603985Z","iopub.status.idle":"2023-06-04T15:03:07.509296Z","shell.execute_reply.started":"2023-06-04T15:03:05.603945Z","shell.execute_reply":"2023-06-04T15:03:07.507902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Intro","metadata":{}},{"cell_type":"markdown","source":"https://www.kaggle.com/code/konstantinsamolinov/to-concatenate-or-not-to-concatenate  \nwe explore data and know, that TdcsFog and DeFOG can be compared","metadata":{}},{"cell_type":"markdown","source":"# Import Data","metadata":{}},{"cell_type":"code","source":"# root directory\nroot = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction'","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-06-04T15:03:07.510630Z","iopub.execute_input":"2023-06-04T15:03:07.511653Z","iopub.status.idle":"2023-06-04T15:03:07.517987Z","shell.execute_reply.started":"2023-06-04T15:03:07.511618Z","shell.execute_reply":"2023-06-04T15:03:07.516035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n### copy it from own notebook where we explore data (link after libraries)\n## take 10% of all data\n\ndf_tdcs_meta = pd.read_csv(os.path.join(root, 'tdcsfog_metadata.csv'))\ndf_defog_meta = pd.read_csv(os.path.join(root, 'defog_metadata.csv'))\n###\ndf_subjects = pd.read_csv(os.path.join(root, 'subjects.csv'))\ntdcs_file_path = glob.glob(os.path.join(root, 'train', 'tdcsfog', '*.csv'), recursive=True)\ntdcs_file_path = tdcs_file_path[::int(len(tdcs_file_path)/10)]\ndf_tdcs = pd.DataFrame()\n###\nfor fp in tdcs_file_path:    \n    tmp = pd.read_csv(fp)\n    file_id = os.path.basename(fp).replace(\".csv\", \"\")\n    subject = df_tdcs_meta.loc[df_tdcs_meta['Id'] == file_id, 'Subject'].iloc[0]\n    tmp['Medication'] = df_tdcs_meta.loc[df_tdcs_meta['Id'] == file_id, 'Medication'].iloc[0]\n    tmp['Age'] = df_subjects.loc[df_subjects['Subject'] == subject, 'Age'].iloc[0]\n    tmp['Sex'] = df_subjects.loc[df_subjects['Subject'] == subject, 'Sex'].iloc[0]\n    tmp['YearsSinceDx'] = df_subjects.loc[df_subjects['Subject'] == subject, 'YearsSinceDx'].iloc[0]\n    tmp['NFOGQ'] =df_subjects.loc[df_subjects['Subject'] == subject, 'NFOGQ'].iloc[0]\n    df_tdcs = pd.concat([df_tdcs, tmp]).reset_index(drop=True)\n###\ndefog_file_path = glob.glob(os.path.join(root, 'train', 'defog', '*.csv'), recursive=True)\ndefog_file_path = defog_file_path[::int(len(defog_file_path)/10)]\n###\ndf_defog = pd.DataFrame()\nfor fp in defog_file_path:\n    tmp = pd.read_csv(fp)\n    file_id = os.path.basename(fp).replace(\".csv\", \"\")\n    subject = df_defog_meta.loc[df_defog_meta['Id'] == file_id, 'Subject'].iloc[0]\n    tmp['Medication'] = df_defog_meta.loc[df_defog_meta['Id'] == file_id, 'Medication'].iloc[0]\n    tmp['Age'] = df_subjects.loc[df_subjects['Subject'] == subject, 'Age'].iloc[0]\n    tmp['Sex'] = df_subjects.loc[df_subjects['Subject'] == subject, 'Sex'].iloc[0]\n    tmp['YearsSinceDx'] = df_subjects.loc[df_subjects['Subject'] == subject, 'YearsSinceDx'].iloc[0]\n    tmp['NFOGQ'] =df_subjects.loc[df_subjects['Subject'] == subject, 'NFOGQ'].iloc[0]\n    tmp = tmp[(tmp['Valid'] == True) & (tmp['Task']==True)]\n    tmp = tmp.drop(['Valid', 'Task'], axis=1)\n    df_defog = pd.concat([df_defog, tmp]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:07.521120Z","iopub.execute_input":"2023-06-04T15:03:07.521530Z","iopub.status.idle":"2023-06-04T15:03:12.232401Z","shell.execute_reply.started":"2023-06-04T15:03:07.521494Z","shell.execute_reply":"2023-06-04T15:03:12.231118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dict_col = ['AccV', 'AccML', 'AccAP']\nCONST = 0.10197\nfor col in dict_col:\n    df_tdcs[col] = df_tdcs[col] * CONST","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:12.233724Z","iopub.execute_input":"2023-06-04T15:03:12.234117Z","iopub.status.idle":"2023-06-04T15:03:12.243977Z","shell.execute_reply.started":"2023-06-04T15:03:12.234087Z","shell.execute_reply":"2023-06-04T15:03:12.243036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.concat([df_tdcs, df_defog]).reset_index(drop=True)\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:12.245498Z","iopub.execute_input":"2023-06-04T15:03:12.246104Z","iopub.status.idle":"2023-06-04T15:03:12.400485Z","shell.execute_reply.started":"2023-06-04T15:03:12.246072Z","shell.execute_reply":"2023-06-04T15:03:12.399340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare data","metadata":{}},{"cell_type":"markdown","source":"## Rework type of 'Sex' and 'Medication'","metadata":{}},{"cell_type":"code","source":"## prapare column Sex and Medication for regression model\ndf_train['Medication'] = np.where(df_train['Medication']=='on', 1, 0)\ndf_train['Sex'] = np.where(df_train['Sex']=='M', 1, 0)\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:12.402048Z","iopub.execute_input":"2023-06-04T15:03:12.402486Z","iopub.status.idle":"2023-06-04T15:03:12.614519Z","shell.execute_reply.started":"2023-06-04T15:03:12.402438Z","shell.execute_reply":"2023-06-04T15:03:12.613703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, we dont need so big DataFrame, lets sample it","metadata":{}},{"cell_type":"code","source":"# df_train_sampled = df_train.sample(frac=0.01) #take 1% of all data","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:12.615920Z","iopub.execute_input":"2023-06-04T15:03:12.616245Z","iopub.status.idle":"2023-06-04T15:03:12.620746Z","shell.execute_reply.started":"2023-06-04T15:03:12.616217Z","shell.execute_reply":"2023-06-04T15:03:12.619688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pd.DataFrame(data=[df_train.shape, df_train_sampled.shape], columns=['Columns', 'Rows'], index=['Before sampling', 'After Sampling']).transpose()","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:12.622452Z","iopub.execute_input":"2023-06-04T15:03:12.622982Z","iopub.status.idle":"2023-06-04T15:03:12.633492Z","shell.execute_reply.started":"2023-06-04T15:03:12.622953Z","shell.execute_reply":"2023-06-04T15:03:12.632137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Split Data","metadata":{}},{"cell_type":"code","source":"# split data into features and target.\ny = df_train[['StartHesitation', 'Turn', 'Walking']]                       # target\nX = df_train.drop(['StartHesitation', 'Turn', 'Walking', 'Time'], axis=1)  # feature","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:12.638113Z","iopub.execute_input":"2023-06-04T15:03:12.638509Z","iopub.status.idle":"2023-06-04T15:03:12.673137Z","shell.execute_reply.started":"2023-06-04T15:03:12.638470Z","shell.execute_reply":"2023-06-04T15:03:12.671932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame([df_train.StartHesitation.value_counts(), df_train.Walking.value_counts(), df_train.Turn.value_counts()]).transpose()","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:12.674500Z","iopub.execute_input":"2023-06-04T15:03:12.674809Z","iopub.status.idle":"2023-06-04T15:03:12.709482Z","shell.execute_reply.started":"2023-06-04T15:03:12.674784Z","shell.execute_reply":"2023-06-04T15:03:12.708244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = train_test_split(X, y, random_state=160891)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:12.710750Z","iopub.execute_input":"2023-06-04T15:03:12.711069Z","iopub.status.idle":"2023-06-04T15:03:12.831182Z","shell.execute_reply.started":"2023-06-04T15:03:12.711044Z","shell.execute_reply":"2023-06-04T15:03:12.830002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame([X_train.shape, X_valid.shape, y_train.shape, y_valid.shape], columns=['rows', 'columns'], \n             index=['X_train', 'X_valid', 'y_train', 'y_valid']).transpose()","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:12.832633Z","iopub.execute_input":"2023-06-04T15:03:12.832953Z","iopub.status.idle":"2023-06-04T15:03:12.845415Z","shell.execute_reply.started":"2023-06-04T15:03:12.832927Z","shell.execute_reply":"2023-06-04T15:03:12.844156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Split for targets and features","metadata":{}},{"cell_type":"code","source":"y_train_sh = y_train['StartHesitation']\ny_valid_sh = y_valid['StartHesitation']\ny_train_t = y_train['Turn']\ny_valid_t = y_valid['Turn']\ny_train_w = y_train['Walking']\ny_valid_w = y_valid['Walking']","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:12.846947Z","iopub.execute_input":"2023-06-04T15:03:12.847296Z","iopub.status.idle":"2023-06-04T15:03:12.857356Z","shell.execute_reply.started":"2023-06-04T15:03:12.847268Z","shell.execute_reply":"2023-06-04T15:03:12.856012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model","metadata":{}},{"cell_type":"markdown","source":"At this work we have movement vectors, so let's try Support Vector Machines for Classification goal.","metadata":{}},{"cell_type":"markdown","source":"### Little bit of theory of Support Vector Machines","metadata":{}},{"cell_type":"markdown","source":"The basics of Support Vector Machines and how it works are best understood with a simple example. Let’s imagine we have two tags: red and blue, and our data has two features: x and y. We want a classifier that, given a pair of (x,y) coordinates, outputs if it’s either red or blue. We plot our already labeled training data on a plane:","metadata":{}},{"cell_type":"markdown","source":"![alternatvie text](https://monkeylearn.com/static/52081a1b625e8ba22c00210d547b4f1a/d8712/plot_original.webp)","metadata":{}},{"cell_type":"markdown","source":"A support vector machine takes these data points and outputs the hyperplane (which in two dimensions it’s simply a line) that best separates the tags. This line is the decision boundary: anything that falls to one side of it we will classify as blue, and anything that falls to the other as red.","metadata":{}},{"cell_type":"markdown","source":"![alternatvie text](https://monkeylearn.com/static/57fd2448dfb67cfff990f32191463e80/d8712/plot_hyperplanes_2.webp)","metadata":{}},{"cell_type":"markdown","source":"For SVM, it’s the one that maximizes the margins from both tags. In other words: the hyperplane (remember it's a line in this case) whose distance to the nearest element of each tag is the largest.","metadata":{}},{"cell_type":"markdown","source":"![alternatvie text](https://monkeylearn.com/static/7002b9ebbacb0e878edbf30e8ff5b01c/d8712/plot_hyperplanes_annotated.webp)","metadata":{}},{"cell_type":"markdown","source":"SVC parameters:\n- C: float, default=1.0\n- kernel: {‘linear’, ‘poly’, ‘rbf’, ‘sigmoid’, ‘precomputed’} or callable, default=’rbf’\n- degreeint: default = 3\n- gamma: {‘scale’, ‘auto’} or float, default=’scale’\n- coef0: float, default=0.0\n- shrinking: bool, default=True\n- tolfloat: default=1e-3\n- cache_size: float, default=200\n- class_weight: dict or ‘balanced’, default=None\n- verbose: bool, default=False\n- max_iter: int, default=-1\n- decision_function_shape: {‘ovo’, ‘ovr’}, default=’ovr’\n- break_ties: bool, default=False\n- random_state: int, RandomState instance or None, default=None\n\n    ","metadata":{}},{"cell_type":"markdown","source":"**Choosing a kernel function**\n\nNow that we have the feature vectors, the only thing left to do is choosing a kernel function for our model. Every problem is different, and the kernel function depends on what the data looks like. In our example, our data was arranged in concentric circles, so we chose a kernel that matched those data points.\n\nTaking that into account, what’s best for natural language processing? Do we need a nonlinear classifier? Or is the data linearly separable? It turns out that it’s best to stick to a linear kernel. Why?\n\nBack in our example, we had two features. Some real uses of SVM in other fields may use tens or even hundreds of features. Meanwhile, NLP classifiers use thousands of features, since they can have up to one for every word that appears in the training data. This changes the problem a little bit: while using nonlinear kernels may be a good idea in other cases, having this many features will end up making nonlinear kernels overfit the data. Therefore, it’s best to just stick to a good old linear kernel, which actually results in the best performance in these cases.","metadata":{}},{"cell_type":"markdown","source":"Formula for vector in kernel:\n- polynomial (poly): ![](https://habrastorage.org/storage/habraeffect/8d/73/8d734dc93985270c76bfc0e059556264.png)\n- linear: ![](https://data-flair.training/blogs/wp-content/uploads/sites/2/2017/08/linear-splines-kernel-in-one-dimension.png)\n- rbf: ![](https://habrastorage.org/storage/habraeffect/df/c6/dfc66ea3b4a833ef4033ba07362b31d3.png)\n- sigmoid: ![](https://habrastorage.org/storage/habraeffect/35/37/353764f90f9af20442bcc97c6a9b4a08.png)","metadata":{}},{"cell_type":"code","source":"# model_svc_sh = SVC(class_weight='balanced')\n# model_svc_t = SVC(class_weight='balanced')\n# model_svc_w = SVC(class_weight='balanced')","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:12.858364Z","iopub.execute_input":"2023-06-04T15:03:12.858788Z","iopub.status.idle":"2023-06-04T15:03:12.871836Z","shell.execute_reply.started":"2023-06-04T15:03:12.858758Z","shell.execute_reply":"2023-06-04T15:03:12.870659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model for StartHesitation","metadata":{}},{"cell_type":"code","source":"# %%time\n# model_svc_sh.fit(X_train, y_train_sh)\n# predicted_valid_sh = model_svc_sh.predict(X_valid)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:12.873582Z","iopub.execute_input":"2023-06-04T15:03:12.873938Z","iopub.status.idle":"2023-06-04T15:03:12.886184Z","shell.execute_reply.started":"2023-06-04T15:03:12.873910Z","shell.execute_reply":"2023-06-04T15:03:12.885065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pd.DataFrame(data=[f1_score(y_valid_sh, predicted_valid_sh), \\\n#                    roc_auc_score(y_valid_sh, predicted_valid_sh)], index=['f1', 'Roc auc'], columns=['score'])","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:12.887949Z","iopub.execute_input":"2023-06-04T15:03:12.888421Z","iopub.status.idle":"2023-06-04T15:03:12.899445Z","shell.execute_reply.started":"2023-06-04T15:03:12.888389Z","shell.execute_reply":"2023-06-04T15:03:12.898039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pd.DataFrame([pd.Series(predicted_valid_sh).value_counts()], index=['Count']).transpose()","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:12.901442Z","iopub.execute_input":"2023-06-04T15:03:12.901899Z","iopub.status.idle":"2023-06-04T15:03:12.913149Z","shell.execute_reply.started":"2023-06-04T15:03:12.901859Z","shell.execute_reply":"2023-06-04T15:03:12.911696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model for Turn","metadata":{}},{"cell_type":"code","source":"# %%time\n# model_svc_t.fit(X_train, y_train_t)\n# predicted_valid_t = model_svc_t.predict(X_valid)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:12.914974Z","iopub.execute_input":"2023-06-04T15:03:12.915429Z","iopub.status.idle":"2023-06-04T15:03:12.926285Z","shell.execute_reply.started":"2023-06-04T15:03:12.915388Z","shell.execute_reply":"2023-06-04T15:03:12.925387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pd.DataFrame(data=[f1_score(y_valid_t, predicted_valid_t), \\\n#                    roc_auc_score(y_valid_t, predicted_valid_t)], index=['f1', 'Roc auc'], columns=['score'])","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:12.929963Z","iopub.execute_input":"2023-06-04T15:03:12.930309Z","iopub.status.idle":"2023-06-04T15:03:12.938279Z","shell.execute_reply.started":"2023-06-04T15:03:12.930281Z","shell.execute_reply":"2023-06-04T15:03:12.937358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pd.DataFrame([pd.Series(predicted_valid_t).value_counts()], index=['Count']).transpose()","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:12.939721Z","iopub.execute_input":"2023-06-04T15:03:12.940089Z","iopub.status.idle":"2023-06-04T15:03:12.953938Z","shell.execute_reply.started":"2023-06-04T15:03:12.940059Z","shell.execute_reply":"2023-06-04T15:03:12.952696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model for Walking","metadata":{}},{"cell_type":"code","source":"# %%time\n# model_svc_w.fit(X_train, y_train_w)\n# predicted_valid_w = model_svc_w.predict(X_valid)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:12.955900Z","iopub.execute_input":"2023-06-04T15:03:12.956356Z","iopub.status.idle":"2023-06-04T15:03:12.966551Z","shell.execute_reply.started":"2023-06-04T15:03:12.956294Z","shell.execute_reply":"2023-06-04T15:03:12.965364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pd.DataFrame(data=[f1_score(y_valid_w, predicted_valid_w), \\\n#                    roc_auc_score(y_valid_w, predicted_valid_w)], index=['f1', 'Roc auc'], columns=['score'])","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:12.968246Z","iopub.execute_input":"2023-06-04T15:03:12.968704Z","iopub.status.idle":"2023-06-04T15:03:12.978520Z","shell.execute_reply.started":"2023-06-04T15:03:12.968667Z","shell.execute_reply":"2023-06-04T15:03:12.977488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pd.DataFrame([pd.Series(predicted_valid_w).value_counts()], index=['Count']).transpose()","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:12.980183Z","iopub.execute_input":"2023-06-04T15:03:12.980559Z","iopub.status.idle":"2023-06-04T15:03:12.990797Z","shell.execute_reply.started":"2023-06-04T15:03:12.980531Z","shell.execute_reply":"2023-06-04T15:03:12.989566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Disbalanced classes greatly slow down the training of the model.","metadata":{}},{"cell_type":"markdown","source":"# Test data and subimt","metadata":{}},{"cell_type":"code","source":"# list of all tdcsfog csv file path\ntdcs_test_file_path = glob.glob(os.path.join(root, 'test', 'tdcsfog', '*.csv'), recursive=True)\nprint(f'the number of files to be read: {len(tdcs_test_file_path)}')","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:12.992031Z","iopub.execute_input":"2023-06-04T15:03:12.992384Z","iopub.status.idle":"2023-06-04T15:03:13.006447Z","shell.execute_reply.started":"2023-06-04T15:03:12.992352Z","shell.execute_reply":"2023-06-04T15:03:13.005260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\n# Initialize a DataFrame to combine data from multiple CSV files.\ndf_tdcs_test = pd.DataFrame()\n\nfor fp in tdcs_test_file_path:\n    \n    # load data into a variable 'tmp'.\n    tmp = pd.read_csv(fp)\n    \n    # get file Id from csv file name.\n    file_id = os.path.basename(fp).replace(\".csv\", \"\")\n    \n    # get subject Id.\n    subject = df_tdcs_meta.loc[df_tdcs_meta['Id'] == file_id, 'Subject'].iloc[0]\n    \n    # add metadata.\n    tmp['Medication'] = df_tdcs_meta.loc[df_tdcs_meta['Id'] == file_id, 'Medication'].iloc[0]\n    tmp['Age'] = df_subjects.loc[df_subjects['Subject'] == subject, 'Age'].iloc[0]\n    tmp['Sex'] = df_subjects.loc[df_subjects['Subject'] == subject, 'Sex'].iloc[0]\n    tmp['YearsSinceDx'] = df_subjects.loc[df_subjects['Subject'] == subject, 'YearsSinceDx'].iloc[0]\n    tmp['NFOGQ'] =df_subjects.loc[df_subjects['Subject'] == subject, 'NFOGQ'].iloc[0]\n    \n    # add Id data to submit.\n    tmp['Id'] = file_id + '_' + tmp['Time'].astype(str)\n    \n    # concat the data\n    df_tdcs_test = pd.concat([df_tdcs_test, tmp]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:13.009373Z","iopub.execute_input":"2023-06-04T15:03:13.009766Z","iopub.status.idle":"2023-06-04T15:03:13.048924Z","shell.execute_reply.started":"2023-06-04T15:03:13.009732Z","shell.execute_reply":"2023-06-04T15:03:13.047808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check the contents of the df_tdcs_test\ndf_tdcs_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:13.050116Z","iopub.execute_input":"2023-06-04T15:03:13.050506Z","iopub.status.idle":"2023-06-04T15:03:13.069607Z","shell.execute_reply.started":"2023-06-04T15:03:13.050477Z","shell.execute_reply":"2023-06-04T15:03:13.068329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# list of all tdcsfog csv file path\ndefog_test_file_path = glob.glob(os.path.join(root, 'test', 'defog', '*.csv'), recursive=True)\nprint(f'the number of files to be read: {len(defog_test_file_path)}')","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:13.076009Z","iopub.execute_input":"2023-06-04T15:03:13.076427Z","iopub.status.idle":"2023-06-04T15:03:13.085824Z","shell.execute_reply.started":"2023-06-04T15:03:13.076395Z","shell.execute_reply":"2023-06-04T15:03:13.084606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize a DataFrame to combine data from multiple CSV files.\ndf_defog_test = pd.DataFrame()\n\nfor fp in defog_test_file_path:\n    # load data into a variable 'tmp'.\n    tmp = pd.read_csv(fp)\n    \n    # get file Id from csv file name.\n    file_id = os.path.basename(fp).replace(\".csv\", \"\")\n    \n    # get subject Id.\n    subject = df_defog_meta.loc[df_defog_meta['Id'] == file_id, 'Subject'].iloc[0]\n    \n    # add metadata.\n    tmp['Medication'] = df_defog_meta.loc[df_defog_meta['Id'] == file_id, 'Medication'].iloc[0]\n    tmp['Age'] = df_subjects.loc[df_subjects['Subject'] == subject, 'Age'].iloc[0]\n    tmp['Sex'] = df_subjects.loc[df_subjects['Subject'] == subject, 'Sex'].iloc[0]\n    tmp['YearsSinceDx'] = df_subjects.loc[df_subjects['Subject'] == subject, 'YearsSinceDx'].iloc[0]\n    tmp['NFOGQ'] =df_subjects.loc[df_subjects['Subject'] == subject, 'NFOGQ'].iloc[0]\n    \n    # add Id data to submit.\n    tmp['Id'] = file_id + '_' + tmp['Time'].astype(str)\n    \n    # concat the data\n    df_defog_test = pd.concat([df_defog_test, tmp]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:13.087246Z","iopub.execute_input":"2023-06-04T15:03:13.087634Z","iopub.status.idle":"2023-06-04T15:03:13.863131Z","shell.execute_reply.started":"2023-06-04T15:03:13.087606Z","shell.execute_reply":"2023-06-04T15:03:13.861854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check the contents of the df_defog_test\ndf_defog_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:13.866051Z","iopub.execute_input":"2023-06-04T15:03:13.866515Z","iopub.status.idle":"2023-06-04T15:03:13.884841Z","shell.execute_reply.started":"2023-06-04T15:03:13.866474Z","shell.execute_reply":"2023-06-04T15:03:13.883701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in dict_col:\n    df_tdcs_test[col] = df_tdcs_test[col] * CONST","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:13.886720Z","iopub.execute_input":"2023-06-04T15:03:13.887169Z","iopub.status.idle":"2023-06-04T15:03:13.894176Z","shell.execute_reply.started":"2023-06-04T15:03:13.887130Z","shell.execute_reply":"2023-06-04T15:03:13.893276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# concat tdcs and defog data.\ndf_test = pd.concat([df_tdcs_test, df_defog_test]).reset_index(drop=True)\n\n# encode string columns into 0/1 format\ndf_test['Medication'] = np.where(df_test['Medication']=='on', 1, 0)\ndf_test['Sex'] = np.where(df_test['Sex']=='M', 1, 0)\ndisplay(df_test)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:13.895740Z","iopub.execute_input":"2023-06-04T15:03:13.896099Z","iopub.status.idle":"2023-06-04T15:03:14.117480Z","shell.execute_reply.started":"2023-06-04T15:03:13.896072Z","shell.execute_reply":"2023-06-04T15:03:14.116264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split data into submission Id and feature.\nId = df_test['Id']                             # Id for submission data\nX_test = df_test.drop(['Time', 'Id'], axis=1)  # feature of test data\nX_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:14.118776Z","iopub.execute_input":"2023-06-04T15:03:14.119074Z","iopub.status.idle":"2023-06-04T15:03:14.140988Z","shell.execute_reply.started":"2023-06-04T15:03:14.119049Z","shell.execute_reply":"2023-06-04T15:03:14.139884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time \n\n# # calculate prediction using trained models.\n# predicted_test_sh = model_svc_sh.predict(X_test)\n# predicted_test_w = model_svc_w.predict(X_test)\n# predicted_test_t = model_svc_t.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:14.142190Z","iopub.execute_input":"2023-06-04T15:03:14.145679Z","iopub.status.idle":"2023-06-04T15:03:14.150168Z","shell.execute_reply.started":"2023-06-04T15:03:14.145634Z","shell.execute_reply":"2023-06-04T15:03:14.149296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = pd.DataFrame(Id, columns=['Id'])\nsubmit['StartHesitation'] = 0\nsubmit['Turn'] = 0\nsubmit['Walking'] = 0","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:09:48.240635Z","iopub.execute_input":"2023-06-04T15:09:48.241044Z","iopub.status.idle":"2023-06-04T15:09:48.256625Z","shell.execute_reply.started":"2023-06-04T15:09:48.241015Z","shell.execute_reply":"2023-06-04T15:09:48.255312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(submit)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:09:48.518715Z","iopub.execute_input":"2023-06-04T15:09:48.519103Z","iopub.status.idle":"2023-06-04T15:09:48.537489Z","shell.execute_reply.started":"2023-06-04T15:09:48.519075Z","shell.execute_reply":"2023-06-04T15:09:48.536276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T15:03:24.916785Z","iopub.execute_input":"2023-06-04T15:03:24.918102Z","iopub.status.idle":"2023-06-04T15:03:25.936135Z","shell.execute_reply.started":"2023-06-04T15:03:24.918012Z","shell.execute_reply":"2023-06-04T15:03:25.935035Z"},"trusted":true},"execution_count":null,"outputs":[]}]}