{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!wget http://bit.ly/3ZLyF82 -O CSS.css -q\n    \nfrom IPython.core.display import HTML\nwith open('./CSS.css', 'r') as file:\n    custom_css = file.read()\n\nHTML(custom_css)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-10T21:37:36.542976Z","iopub.execute_input":"2023-04-10T21:37:36.543533Z","iopub.status.idle":"2023-04-10T21:37:57.724892Z","shell.execute_reply.started":"2023-04-10T21:37:36.543492Z","shell.execute_reply":"2023-04-10T21:37:57.723520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Importing the necessary libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nimport warnings\nwarnings.filterwarnings('ignore')\n","metadata":{"execution":{"iopub.status.busy":"2023-04-10T21:37:57.727171Z","iopub.execute_input":"2023-04-10T21:37:57.727518Z","iopub.status.idle":"2023-04-10T21:37:58.438360Z","shell.execute_reply.started":"2023-04-10T21:37:57.727484Z","shell.execute_reply":"2023-04-10T21:37:58.436885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_path='/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog'","metadata":{"execution":{"iopub.status.busy":"2023-04-10T21:37:58.440201Z","iopub.execute_input":"2023-04-10T21:37:58.440542Z","iopub.status.idle":"2023-04-10T21:37:58.445893Z","shell.execute_reply.started":"2023-04-10T21:37:58.440508Z","shell.execute_reply":"2023-04-10T21:37:58.444472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Defining the memory reduction function","metadata":{}},{"cell_type":"code","source":"## https://www.kaggle.com/code/arjanso/reducing-dataframe-memory-size-by-65\ndef reduce_memory_usage(df):\n    \n    start_mem = df.memory_usage().sum() / 1024 ** 2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024 ** 2 \n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-04-10T21:37:58.448807Z","iopub.execute_input":"2023-04-10T21:37:58.449174Z","iopub.status.idle":"2023-04-10T21:37:58.463966Z","shell.execute_reply.started":"2023-04-10T21:37:58.449141Z","shell.execute_reply":"2023-04-10T21:37:58.462490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reading all the files under tdcsfog dataset","metadata":{}},{"cell_type":"code","source":"tdcsfog_list =[]\nfor file_name in os.listdir(tdcsfog_path):\n    if file_name.endswith('.csv'):\n        file_path = os.path.join(tdcsfog_path, file_name)\n        file = pd.read_csv(file_path)\n        tdcsfog_list.append(file)","metadata":{"execution":{"iopub.status.busy":"2023-04-10T21:37:58.465956Z","iopub.execute_input":"2023-04-10T21:37:58.466474Z","iopub.status.idle":"2023-04-10T21:38:16.644282Z","shell.execute_reply.started":"2023-04-10T21:37:58.466422Z","shell.execute_reply":"2023-04-10T21:38:16.642941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog = pd.concat(tdcsfog_list, axis = 0)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-10T21:38:16.645755Z","iopub.execute_input":"2023-04-10T21:38:16.646395Z","iopub.status.idle":"2023-04-10T21:38:17.250909Z","shell.execute_reply.started":"2023-04-10T21:38:16.646304Z","shell.execute_reply":"2023-04-10T21:38:17.249595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog=reduce_memory_usage(tdcsfog)","metadata":{"execution":{"iopub.status.busy":"2023-04-10T21:38:17.253336Z","iopub.execute_input":"2023-04-10T21:38:17.254182Z","iopub.status.idle":"2023-04-10T21:38:17.798316Z","shell.execute_reply.started":"2023-04-10T21:38:17.254128Z","shell.execute_reply":"2023-04-10T21:38:17.796953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Defining the summary function","metadata":{}},{"cell_type":"code","source":"def summary(text, df):\n    print(f'{text} shape: {df.shape}')\n    summ = pd.DataFrame(df.dtypes, columns=['dtypes'])\n    summ['null'] = df.isnull().sum()\n    summ['unique'] = df.nunique()\n    summ['min'] = df.min()\n    summ['median'] = df.median()\n    summ['max'] = df.max()\n    summ['mean'] = df.mean()\n    summ['std'] = df.std()\n    return summ","metadata":{"execution":{"iopub.status.busy":"2023-04-10T21:38:17.799896Z","iopub.execute_input":"2023-04-10T21:38:17.800290Z","iopub.status.idle":"2023-04-10T21:38:17.807626Z","shell.execute_reply.started":"2023-04-10T21:38:17.800253Z","shell.execute_reply":"2023-04-10T21:38:17.806612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summary('tdcsfog',tdcsfog)","metadata":{"execution":{"iopub.status.busy":"2023-04-10T21:38:17.809079Z","iopub.execute_input":"2023-04-10T21:38:17.809697Z","iopub.status.idle":"2023-04-10T21:38:20.490092Z","shell.execute_reply.started":"2023-04-10T21:38:17.809658Z","shell.execute_reply":"2023-04-10T21:38:20.488660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ## EDA by PairGrid plot","metadata":{}},{"cell_type":"code","source":"g = sns.PairGrid(tdcsfog[['AccV', 'AccML', 'AccAP']])\ng.map_lower(plt.scatter, alpha = 0.6)\ng.map_diag(plt.hist, alpha = 0.7)","metadata":{"execution":{"iopub.status.busy":"2023-04-10T21:38:20.494059Z","iopub.execute_input":"2023-04-10T21:38:20.494900Z","iopub.status.idle":"2023-04-10T21:40:17.159929Z","shell.execute_reply.started":"2023-04-10T21:38:20.494856Z","shell.execute_reply":"2023-04-10T21:40:17.158481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Defining the features and target columns","metadata":{}},{"cell_type":"code","source":"X = tdcsfog.iloc[:, 1:4]  \ny1 = tdcsfog['StartHesitation']  \ny2 = tdcsfog['Turn']  \ny3 = tdcsfog['Walking'] ","metadata":{"execution":{"iopub.status.busy":"2023-04-10T21:40:17.161708Z","iopub.execute_input":"2023-04-10T21:40:17.162174Z","iopub.status.idle":"2023-04-10T21:40:17.203781Z","shell.execute_reply.started":"2023-04-10T21:40:17.162129Z","shell.execute_reply":"2023-04-10T21:40:17.202800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Spliting the train data into test and validation data","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report,mean_squared_log_error\n\nX_train, X_val, y1_train, y1_val = train_test_split(X, y1, test_size = 0.2, random_state = 52)\nX_train, X_val, y2_train, y2_val  = train_test_split(X, y2, test_size = 0.2, random_state = 52)\nX_train, X_val, y3_train, y3_val = train_test_split(X, y3, test_size = 0.2, random_state = 52)","metadata":{"execution":{"iopub.status.busy":"2023-04-10T21:40:17.205180Z","iopub.execute_input":"2023-04-10T21:40:17.205529Z","iopub.status.idle":"2023-04-10T21:40:20.275144Z","shell.execute_reply.started":"2023-04-10T21:40:17.205496Z","shell.execute_reply":"2023-04-10T21:40:20.273833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Traning the model for three target varaibles","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\n# Create three separate logistic regression models.\nmodel1 = LogisticRegression()\nmodel2 = LogisticRegression()\nmodel3 = LogisticRegression()\n\n# Train the models on the training data.\nmodel1.fit(X_train, y1_train)\nmodel2.fit(X_train, y2_train)\nmodel3.fit(X_train, y3_train)\n\n# Evaluate the models on the test data.\nprint('Accuracy for StartHesitation:', model1.score(X_val, y1_val))\nprint('Accuracy for Turn:', model2.score(X_val, y2_val))\nprint('Accuracy for Walking:', model3.score(X_val, y3_val))","metadata":{"execution":{"iopub.status.busy":"2023-04-10T22:28:42.255081Z","iopub.execute_input":"2023-04-10T22:28:42.255500Z","iopub.status.idle":"2023-04-10T22:29:19.772161Z","shell.execute_reply.started":"2023-04-10T22:28:42.255465Z","shell.execute_reply":"2023-04-10T22:29:19.770896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generating the classification reports for each of the predictions","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import classification_report\n\ny1_pred = model1.predict(X_val)\ny2_pred = model2.predict(X_val)\ny3_pred = model3.predict(X_val)\n\nprint(f'Classification Report for StartHesitation:{classification_report(y1_val, y1_pred)}')\n\nprint(f'Classification Report for Turn:{classification_report(y2_val, y2_pred)}')\n\nprint(f'Classification Report for Walking: {classification_report(y3_val, y3_pred)}')\n","metadata":{"execution":{"iopub.status.busy":"2023-04-10T22:29:48.162818Z","iopub.execute_input":"2023-04-10T22:29:48.163394Z","iopub.status.idle":"2023-04-10T22:29:54.788772Z","shell.execute_reply.started":"2023-04-10T22:29:48.163341Z","shell.execute_reply":"2023-04-10T22:29:54.787658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reading the test data","metadata":{}},{"cell_type":"code","source":"tdcsfog_test_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog'\ntdcsfog_test_list = []\nfor file_name in os.listdir(tdcsfog_test_path):\n    if file_name.endswith('.csv'):\n        file_path = os.path.join(tdcsfog_test_path, file_name)\n        file = pd.read_csv(file_path)\n        file['Id'] = file_name[:-4] + '_' + file['Time'].apply(str)\n        tdcsfog_test_list.append(file)\ntdcsfog_test = pd.concat(tdcsfog_test_list, axis = 0)\ntdcsfog_test","metadata":{"execution":{"iopub.status.busy":"2023-04-10T22:30:00.332090Z","iopub.execute_input":"2023-04-10T22:30:00.332487Z","iopub.status.idle":"2023-04-10T22:30:00.374843Z","shell.execute_reply.started":"2023-04-10T22:30:00.332451Z","shell.execute_reply":"2023-04-10T22:30:00.373270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_test_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog'\ndefog_test_list = []\nfor file_name in os.listdir(defog_test_path):\n    if file_name.endswith('.csv'):\n        file_path = os.path.join(defog_test_path, file_name)\n        file = pd.read_csv(file_path)\n        file['Id'] = file_name[:-4] + '_' + file['Time'].apply(str)\n        defog_test_list.append(file)\n\ndefog_test = pd.concat(defog_test_list, axis = 0)\ndefog_test","metadata":{"execution":{"iopub.status.busy":"2023-04-10T22:30:03.931237Z","iopub.execute_input":"2023-04-10T22:30:03.931634Z","iopub.status.idle":"2023-04-10T22:30:04.488329Z","shell.execute_reply.started":"2023-04-10T22:30:03.931599Z","shell.execute_reply":"2023-04-10T22:30:04.487094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_test = reduce_memory_usage(tdcsfog_test)","metadata":{"execution":{"iopub.status.busy":"2023-04-10T22:30:08.644664Z","iopub.execute_input":"2023-04-10T22:30:08.645089Z","iopub.status.idle":"2023-04-10T22:30:08.666494Z","shell.execute_reply.started":"2023-04-10T22:30:08.645052Z","shell.execute_reply":"2023-04-10T22:30:08.665103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_test = reduce_memory_usage(defog_test)","metadata":{"execution":{"iopub.status.busy":"2023-04-10T22:30:11.307725Z","iopub.execute_input":"2023-04-10T22:30:11.308166Z","iopub.status.idle":"2023-04-10T22:30:11.715773Z","shell.execute_reply.started":"2023-04-10T22:30:11.308127Z","shell.execute_reply":"2023-04-10T22:30:11.714396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Merging two test datasets","metadata":{}},{"cell_type":"code","source":"test = pd.concat([tdcsfog_test, defog_test], axis = 0).reset_index(drop = True)\ntest","metadata":{"execution":{"iopub.status.busy":"2023-04-10T22:30:14.897824Z","iopub.execute_input":"2023-04-10T22:30:14.899199Z","iopub.status.idle":"2023-04-10T22:30:15.060095Z","shell.execute_reply.started":"2023-04-10T22:30:14.899138Z","shell.execute_reply":"2023-04-10T22:30:15.058763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predicting the model on test datasets","metadata":{}},{"cell_type":"code","source":"\nX_test = test.iloc[:, 1:4]\n\ny1_pred = model1.predict(X_test)\ny2_pred = model2.predict(X_test)\ny3_pred = model3.predict(X_test)\n\ntest['StartHesitation'] = y1_pred \ntest['Turn'] = y2_pred \ntest['Walking'] = y3_pred\n\ntest","metadata":{"execution":{"iopub.status.busy":"2023-04-10T22:30:49.880362Z","iopub.execute_input":"2023-04-10T22:30:49.880778Z","iopub.status.idle":"2023-04-10T22:30:49.981430Z","shell.execute_reply.started":"2023-04-10T22:30:49.880739Z","shell.execute_reply":"2023-04-10T22:30:49.980036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generating the submission file","metadata":{}},{"cell_type":"code","source":"submission = test.iloc[:, 4:].fillna(0.0)\nsubmission.to_csv(\"submission.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2023-04-10T22:31:19.304141Z","iopub.execute_input":"2023-04-10T22:31:19.304594Z","iopub.status.idle":"2023-04-10T22:31:19.756659Z","shell.execute_reply.started":"2023-04-10T22:31:19.304555Z","shell.execute_reply":"2023-04-10T22:31:19.755512Z"},"trusted":true},"execution_count":null,"outputs":[]}]}