{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\n\n# Merge all of the tdcsfog data into 1 file and send it in to the model\ntdcsfog_df = pd.DataFrame()\n\nfor dirname, _, filenames in os.walk('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/'):\n    for filename in filenames:\n        \n        # Add a column with the id from the file\n        curr_df = pd.read_csv(os.path.join(dirname, filename))\n        curr_df[\"Id\"] = filename.split('.')[0]\n        tdcsfog_df = pd.concat([tdcsfog_df, curr_df], axis = 0)\n\n# Combine the Id column with the time column so our output is in the correct format\ntdcsfog_df[\"Id\"] = tdcsfog_df[\"Id\"] + \"_\" + tdcsfog_df[\"Time\"].astype(str)\ntdcsfog_df.set_index(\"Id\", inplace = True)\n\n# Merge all of the defog data into 1 file and send it in to the model\ndefog_df = pd.DataFrame()\n\nfor dirname, _, filenames in os.walk('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog'):\n    for filename in filenames:\n        \n        # Add a column with the id from the file\n        curr_df = pd.read_csv(os.path.join(dirname, filename))\n        curr_df[\"Id\"] = filename.split('.')[0]\n        defog_df = pd.concat([defog_df, curr_df], axis = 0)\n\n# Combine the Id column with the time column so our output is in the correct format\ndefog_df[\"Id\"] = defog_df[\"Id\"] + \"_\" + defog_df[\"Time\"].astype(str)\ndefog_df.set_index(\"Id\", inplace = True)\n\n# Drop all data where the \"Valid\" And \"Task\" are not true (These have not been verified)\ndefog_df = defog_df[(defog_df[\"Valid\"] == 1) & (defog_df[\"Task\"] == 1)]\n\n# Drop Valid and Task columns now that we do not need them\ndefog_df = defog_df.drop([\"Valid\", \"Task\"], axis=1)\n\ntdcsfog_df\ndefog_df","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-16T08:45:52.780892Z","iopub.execute_input":"2023-04-16T08:45:52.781362Z","iopub.status.idle":"2023-04-16T08:49:07.316192Z","shell.execute_reply.started":"2023-04-16T08:45:52.781322Z","shell.execute_reply":"2023-04-16T08:49:07.314978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\n\n# Train our model on the tdcsfog dataset\nx_tdcs = tdcsfog_df[[\"AccV\", \"AccML\", \"AccAP\"]]\ny_tdcs = tdcsfog_df[[\"StartHesitation\", \"Turn\", \"Walking\"]]\n\nx_train_tdcs, x_test_tdcs, y_train_tdcs, y_test_tdcs = train_test_split(x_tdcs, y_tdcs, test_size = 0.1, random_state = 12)\n\ntdcsfog_model = DecisionTreeClassifier()\ntdcsfog_model.fit(x_train_tdcs, y_train_tdcs)\n\n# Train another model on the defog dataset\nx_defog = defog_df[[\"AccV\", \"AccML\", \"AccAP\"]]\ny_defog = defog_df[[\"StartHesitation\", \"Turn\", \"Walking\"]]\n\nx_train_defog, x_test_defog, y_train_defog, y_test_defog = train_test_split(x_defog, y_defog, test_size = 0.1, random_state = 12)\n\ndefog_model = DecisionTreeClassifier()\ndefog_model.fit(x_train_defog, y_train_defog)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T08:49:07.318109Z","iopub.execute_input":"2023-04-16T08:49:07.318450Z","iopub.status.idle":"2023-04-16T08:52:57.984607Z","shell.execute_reply.started":"2023-04-16T08:49:07.318417Z","shell.execute_reply":"2023-04-16T08:52:57.983357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Double check the accuracy of our models\nacc_tdcs = round(tdcsfog_model.score(x_test_tdcs, y_test_tdcs) * 100, 2)\nprint(acc_tdcs)\n\nacc_defog = round(defog_model.score(x_test_defog, y_test_defog) * 100, 2)\nprint(acc_defog)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T08:52:57.985937Z","iopub.execute_input":"2023-04-16T08:52:57.986271Z","iopub.status.idle":"2023-04-16T08:52:59.768466Z","shell.execute_reply.started":"2023-04-16T08:52:57.986238Z","shell.execute_reply":"2023-04-16T08:52:59.767197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare our test data for predictions\ntdcsfog_test_df = pd.DataFrame()\ndefog_test_df = pd.DataFrame()\n\n# Prepare the tdcsfog test data\nfor dirname, _, filenames in os.walk('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog'):\n    for filename in filenames:\n        \n        # Add a column with the id from the file\n        curr_df = pd.read_csv(os.path.join(dirname, filename))\n        curr_df[\"Id\"] = filename.split('.')[0]\n        tdcsfog_test_df = pd.concat([tdcsfog_test_df, curr_df], axis = 0)\n        \n# Prepare the defog test data\nfor dirname, _, filenames in os.walk('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog'):\n    for filename in filenames:\n        \n        # Add a column with the id from the file\n        curr_df = pd.read_csv(os.path.join(dirname, filename))\n        curr_df[\"Id\"] = filename.split('.')[0]\n        defog_test_df = pd.concat([defog_test_df, curr_df], axis = 0)\n\n# Combine the Id column with the time column so our output is in the correct format\ntdcsfog_test_df[\"Id\"] = tdcsfog_test_df[\"Id\"] + \"_\" + tdcsfog_test_df[\"Time\"].astype(str)\ntdcsfog_test_df.set_index(\"Id\", inplace = True)\ntdcsfog_test_df = tdcsfog_test_df.drop([\"Time\"], axis = 1)\n\n# Do the same, but for the defog set\ndefog_test_df[\"Id\"] = defog_test_df[\"Id\"] + \"_\" + defog_test_df[\"Time\"].astype(str)\ndefog_test_df.set_index(\"Id\", inplace = True)\ndefog_test_df = defog_test_df.drop([\"Time\"], axis = 1)\n\ntdcsfog_test_df\ndefog_test_df","metadata":{"execution":{"iopub.status.busy":"2023-04-16T08:52:59.770734Z","iopub.execute_input":"2023-04-16T08:52:59.771114Z","iopub.status.idle":"2023-04-16T08:53:00.201277Z","shell.execute_reply.started":"2023-04-16T08:52:59.771070Z","shell.execute_reply":"2023-04-16T08:53:00.200156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get predictions on the tdcs model\ny_pred_tdcs = tdcsfog_model.predict(tdcsfog_test_df)\ny_tdcs_df = pd.DataFrame(y_pred_tdcs)\n\n# Join our predictions with our ids, and reformat data for submission\nsubmission_tdcs = tdcsfog_test_df.reset_index('Id')\nsubmission_tdcs = submission_tdcs.join(y_tdcs_df)\nsubmission_tdcs = submission_tdcs.rename(columns={0: \"StartHesitation\", 1: \"Turn\", 2: \"Walking\"})\nsubmission_tdcs = submission_tdcs[[\"Id\", \"StartHesitation\", \"Turn\", \"Walking\"]]\n\n# Get predictions on the defog model\ny_pred_defog = defog_model.predict(defog_test_df)\ny_defog_df = pd.DataFrame(y_pred_defog)\n\n# Join our predictions with our ids, and reformat data for submission\nsubmission_defog = defog_test_df.reset_index('Id')\nsubmission_defog = submission_defog.join(y_defog_df)\nsubmission_defog = submission_defog.rename(columns={0: \"StartHesitation\", 1: \"Turn\", 2: \"Walking\"})\nsubmission_defog = submission_defog[[\"Id\", \"StartHesitation\", \"Turn\", \"Walking\"]]\n\nsubmission = pd.concat([submission_tdcs, submission_defog], axis = 0)\n\nsubmission.to_csv(\"/kaggle/working/submission.csv\", index=False)\n\nsubmission","metadata":{"execution":{"iopub.status.busy":"2023-04-16T08:53:00.203046Z","iopub.execute_input":"2023-04-16T08:53:00.203583Z","iopub.status.idle":"2023-04-16T08:53:00.847479Z","shell.execute_reply.started":"2023-04-16T08:53:00.203537Z","shell.execute_reply":"2023-04-16T08:53:00.846227Z"},"trusted":true},"execution_count":null,"outputs":[]}]}