{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport os\nimport glob\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport tqdm\nfrom sklearn.ensemble import RandomForestClassifier\n\n# # base directory\nddir = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction'\n\n# get fog train data files\nfog_train = os.listdir(os.path.join(ddir, 'train/tdcsfog'))\n# print(f'number of files: {len(fog_train)}')\n\n# explore detail of fog train data\nfname = fog_train[0]\nfog_df = pd.read_csv(os.path.join(ddir, 'train/tdcsfog', fname))\nprint(fog_df.head(10))\n\n# descriptive details\n# print(fog_df.describe())\n\n# plot AccV, AccML, AccAP, StartHesitation, Turn, Walking\nrows = 3\ncolumns = 2\npnames = ['AccV', 'AccML', 'AccAP', 'StartHesitation', 'Turn', 'Walking', 'end']\n\ni = 0\nfig, x = plt.subplots(3, 2, figsize=(15, 8))\nfor c in range(columns):\n    for r in range(rows):\n        x[r, c].plot(fog_df[pnames[i]])\n        x[r, c].set_title(pnames[i])\n        i += 1\nfig.suptitle(fname, fontsize=20)\nplt.tight_layout(rect=[99, 5, 99, 5])\n# plt.show()\n\npnames = ['AccV', 'AccML', 'AccAP']\n\nfig, x = plt.subplots(3, 2, figsize=(15, 8))\n\n# plot acceleration while Turn is False\nfor r in range(rows):\n    turn_false = fog_df[fog_df['Turn']==False].reset_index(drop=True)\n    x[r, 0].plot(turn_false[pnames[r]])\n    x[r, 0].set_title(f'Turn is False: {pnames[r]}')\n    # Match Y display range between True and False\n    ymin, ymax = min(fog_df[pnames[r]]), max(fog_df[pnames[r]])\n    x[r, 0].set_ylim(ymin-0.1*abs(ymin), ymax+0.1*ymax)\n\n#plot acceleration while Turn is True\nfor r in range(rows):\n    turn_true = fog_df[fog_df['Turn']==True].reset_index(drop=True)\n    x[r, 1].plot(turn_true[pnames[r]])\n    x[r, 1].set_title(f'Turn is True: {pnames[r]}')\n    # Match Y display range between True and False\n    ymin, ymax = min(fog_df[pnames[r]]), max(fog_df[pnames[r]])\n    x[r, 0].set_ylim(ymin-0.1*abs(ymin), ymax+0.1*ymax)\n\nfig.suptitle(f'Turn is False vs. True @{fname}', fontsize=20)\nplt.tight_layout(rect=[99, 5, 99, 5])\n# plt.show()\n\n# explore tdcsfog metadata\nfog_meta = pd.read_csv(os.path.join(ddir, 'tdcsfog_metadata.csv'))\nprint(fog_meta.head(10))\n\n# number of data corresponding to each subject\nsubject_count = fog_meta[['Id', 'Subject']].groupby('Subject').count()\n\n# print('small number of data', subject_count.sort_values(by='Id').head(2))\n# print('large number of data', subject_count.sort_values(by='Id').tail(2))\n\n# display unique subject data\nsubject_id = 'f5586f'\n# print(fog_meta[fog_meta['Subject']==subject_id])\n\n# visit count\nvisit_counts = fog_meta['Visit'].value_counts()\n\nplt.figure()\nplt.bar(x=visit_counts.index, height=visit_counts.values)\nplt.title('Visit counts')\nplt.xlabel('Visit')\nplt.ylabel('counts')\n# plt.show()\n\n# test count\ntest_counts = fog_meta['Test'].value_counts()\n\nplt.figure()\nplt.bar(x=test_counts.index, height=test_counts.values)\nplt.title('Test counts')\nplt.xlabel('Test')\nplt.ylabel('counts')\n# plt.show()\n\n# medicaiton count\nmed_counts = fog_meta['Medication'].value_counts()\n\nplt.figure()\nplt.bar(x=med_counts.index, height=med_counts.values)\nplt.title('Medication counts')\nplt.xlabel('Medication')\nplt.ylabel('counts')\n# plt.show()\n\n\n# get defog train data files\ndefog_train = os.listdir(os.path.join(ddir, 'train/defog'))\n# print(f'number of files: {len(defog_train)}')\n\n# explore details of first csv file in defog\nfname = defog_train[5]\ndefog_df = pd.read_csv(os.path.join(ddir, 'train/defog', fname))\nprint(defog_df.head(10))\n\n# descriptive details\n# print(defog_df.describe())\n\n# show data for when FOG occurs during Turn\n\n# # plot AccV, AccML, AccAP, StartHesitation, Turn, Walking\n# rows = 3\n# columns = 2\npnames = ['AccV', 'AccML', 'AccAP', 'StartHesitation', 'Turn', 'Walking', 'end']\n\ni = 0\n\nfig, x = plt.subplots(rows, columns, figsize = (15, 8))\nfor c in range(columns):\n    for r in range(rows):\n        if pnames[i] == 'end':\n            break\n        else:\n            x[r, c].plot(defog_df[pnames[i]])\n            x[r, c].set_title(pnames[i])\n        i += 1\nfig.suptitle(fname, fontsize=20)\nplt.tight_layout(rect=[99, 5, 99, 5])\n# plt.show()\n\n# data for when Valid and Task are True\ni = 0\nis_valid = defog_df[(defog_df['Valid']==True) & (defog_df['Task']==True)].reset_index(drop=True)\nfig, x = plt.subplots(3, 2, figsize=(15, 8))\nfor c in range(columns):\n    for r in range(rows):\n        if pnames == 'end':\n            break\n        else:\n            x[r, c].plot(is_valid[pnames[i]])\n            x[r, c].set_title(pnames[i])\n        i += 1\n\nfig.suptitle(f'Valid and Task are True @{fname}', fontsize=20)\nplt.tight_layout(rect=[99, 5, 99, 5])\n# plt.show()\n\n# Turn is False vs True\npnames = ['AccV', 'AccML', 'AccAP']\n\nfig, x = plt.subplots(3, 2, figsize=(15, 8))\n\n# acceleration Turn is False\nfor r in range(rows):\n    turn_false = is_valid[is_valid['Turn']==False].reset_index(drop=True)\n    x[r, 0].plot(turn_false[pnames[r]])\n    x[r, 0].set_title(f'Turn is False: {pnames[r]}')\n    #  Match Y display range\n    ymin, ymax = min(is_valid[pnames[r]]), max(is_valid[pnames[r]])\n    x[r, 0].set_ylim(ymin-0.1*abs(ymin), ymax+0.1*ymax)\n\n# acceleration Turn is True\nfor r in range(rows):\n    turn_true = is_valid[is_valid['Turn']==True].reset_index(drop=True)\n    x[r, 1].plot(turn_true[pnames[r]])\n    x[r, 1].set_title(f'Turn is True: {pnames[r]}')\n    # Match Y display range\n    ymin, ymax = min(is_valid[pnames[r]]), max(is_valid[pnames[r]])\n    x[r, 1].set_ylim(ymin-0.1*abs(ymin), ymax+0.1*ymax)\n\nfig.suptitle(f'Turn is False vs True @{fname}', fontsize=20)\nplt.tight_layout(rect=[99, 5, 99, 5])\n# plt.show()\n\n# explore defog metadata\ndefog_meta = pd.read_csv(os.path.join(ddir, 'defog_metadata.csv'))\nprint(defog_meta.head(10))\n# print(defog_meta.describe(include='all'))\n\n# number of data corresponding to each subject\nsubject_count = defog_meta[['Id', 'Subject']].groupby('Subject').count()\n\n# print('small number of data', subject_count.sort_values(by='Id').head(2))\n# print('large number of data', subject_count.sort_values(by='Id').tail(2))\n\n# display unique subject data\nsubject_id = 'bf608b'\n# print(defog_meta[defog_meta['Subject']==subject_id])\n\n# visit count\nvisit_counts = defog_meta['Visit'].value_counts()\n\nplt.figure()\nplt.bar(x=visit_counts.index, height=visit_counts.values)\nplt.title('Visit counts')\nplt.xlabel('Visit')\nplt.ylabel('counts')\n# plt.show()\n\n# medicaiton count\nmed_counts = defog_meta['Medication'].value_counts()\n\nplt.figure()\nplt.bar(x=med_counts.index, height=med_counts.values)\nplt.title('Medication counts')\nplt.xlabel('Medication')\nplt.ylabel('counts')\n# plt.show()\n\n\n\n# explore subject metadata\nsubject_meta = pd.read_csv(os.path.join(ddir, 'subjects.csv'))\nprint(subject_meta.head(10))\n\n# train the model\n\n# get all tdcsfog files\nfog_path = glob.glob(os.path.join(ddir, '/train/tdcsfog', '*.csv'), recursive=True)\nfog_path = fog_path[::100]\n# print(len(fog_path))\n\n# unified dataframe for fog\nfog = pd.DataFrame()\n\nfor fp in tqdm.tqdm(fog_path):\n    tmp = pd.read_csv(fp)\n    file_id = os.path.basename(fp).replace(\".csv\",\"\")\n    subject = fog_meta.loc[fog_meta['Id']==file_id, 'Subject'].iloc[0]\n    tmp['Medication'] = fog_meta.loc[fog_meta['Id']== file_id, 'Medication'].iloc[0]\n    tmp['Age'] = subject_meta.loc[subject_meta['Subject']==subject, 'Age'].iloc[0]\n    tmp['Sex'] = subject_meta.loc[subject_meta['Subject']==subject, 'Sex'].iloc[0]\n    tmp['YearsSinceDx'] = subject_meta.loc[subject_meta['Subject']==subject, 'YearsSinceDx'].iloc[0]\n    tmp['NFOGQ'] = subject_meta.loc[subject_meta['Subject']==subject, 'NFOGQ'].iloc[0]\n    fog = pd.concat([fog, tmp]).reset_index(drop=True)\n\nprint(fog.head(20))\n\n# get all tdcsfog files\ndefog_path = glob.glob(os.path.join(ddir, 'train/defog', '*.csv'), recursive=True)\ndefog_path = defog_path[::100]\n\n# unified dataframe for defog\ndefog = pd.DataFrame()\n\nfor fp in tqdm.tqdm(defog_path):\n    tmp = pd.read_csv(fp)\n    file_id = os.path.basename(fp).replace(\".csv\",\"\")\n    subject = defog_meta.loc[defog_meta['Id']==file_id, 'Subject'].iloc[0]\n    tmp['Medication'] = defog_meta.loc[defog_meta['Id']== file_id, 'Medication'].iloc[0]\n    tmp['Age'] = subject_meta.loc[subject_meta['Subject']==subject, 'Age'].iloc[0]\n    tmp['Sex'] = subject_meta.loc[subject_meta['Subject']==subject, 'Sex'].iloc[0]\n    tmp['YearsSinceDx'] = subject_meta.loc[subject_meta['Subject']==subject, 'YearsSinceDx'].iloc[0]\n    tmp['NFOGQ'] = subject_meta.loc[subject_meta['Subject']==subject, 'NFOGQ'].iloc[0]\n\n    tmp = tmp[(tmp['Valid']==True) & (tmp['Task']==True)]\n    tmp = tmp.drop(['Valid', 'Task'], axis=1)\n\n    defog = pd.concat([defog, tmp]).reset_index(drop=True)\n\nprint(defog.head(20))\n\n# unify fog and defog\ntrain_df = pd.concat([fog, defog]).reset_index(drop=True)\nprint(train_df.head(20))\n\n# encode M/F to 1/0 and on/off to 1/0\ntrain_df['Medication'] = np.where(train_df['Medication']=='on',1,0)\ntrain_df['Sex'] = np.where(train_df['Sex']=='M',1,0)\nprint(train_df.head(20))\n\n# split data into target and features\n# target\nb = train_df[['StartHesitation', 'Turn', 'Walking']]\n# features\na = train_df.drop(['StartHesitation','Turn','Walking','Time'],axis=1)\n\nprint(b.head())\nprint(a.head())\n\nmodel = RandomForestClassifier(random_state=0)\nmodel.fit(a,b)\n\n\n\n# prepare test data and predict\n\n# get all tdcsfog files\nfog_test_path = glob.glob(os.path.join(ddir, 'test/tdcsfog', '*.csv'), recursive=True)\nfog_test_path = fog_test_path[::100]\n# print(len(fog_path))\n\n# unified dataframe for fog\nfog_test = pd.DataFrame()\n\nfor fp in tqdm.tqdm(fog_test_path):\n    tmp = pd.read_csv(fp)\n    file_id = os.path.basename(fp).replace(\".csv\",\"\")\n    subject = fog_meta.loc[fog_meta['Id']==file_id, 'Subject'].iloc[0]\n    tmp['Medication'] = fog_meta.loc[fog_meta['Id']== file_id, 'Medication'].iloc[0]\n    tmp['Age'] = subject_meta.loc[subject_meta['Subject']==subject, 'Age'].iloc[0]\n    tmp['Sex'] = subject_meta.loc[subject_meta['Subject']==subject, 'Sex'].iloc[0]\n    tmp['YearsSinceDx'] = subject_meta.loc[subject_meta['Subject']==subject, 'YearsSinceDx'].iloc[0]\n    tmp['NFOGQ'] = subject_meta.loc[subject_meta['Subject']==subject, 'NFOGQ'].iloc[0]\n    # add Id data\n    tmp['Id'] = file_id + '_' + tmp['Time'].astype(str)\n    fog_test = pd.concat([fog_test, tmp]).reset_index(drop=True)\n\nprint(fog.head(20))\n\n# get all tdcsfog files\ndefog_test_path = glob.glob(os.path.join(ddir, 'test/defog', '*.csv'), recursive=True)\ndefog_test_path = defog_test_path[::100]\n\n# unified dataframe for defog\ndefog_test = pd.DataFrame()\n\nfor fp in tqdm.tqdm(defog_test_path):\n    tmp = pd.read_csv(fp)\n    file_id = os.path.basename(fp).replace(\".csv\",\"\")\n    subject = defog_meta.loc[defog_meta['Id']==file_id, 'Subject'].iloc[0]\n    tmp['Medication'] = defog_meta.loc[defog_meta['Id']== file_id, 'Medication'].iloc[0]\n    tmp['Age'] = subject_meta.loc[subject_meta['Subject']==subject, 'Age'].iloc[0]\n    tmp['Sex'] = subject_meta.loc[subject_meta['Subject']==subject, 'Sex'].iloc[0]\n    tmp['YearsSinceDx'] = subject_meta.loc[subject_meta['Subject']==subject, 'YearsSinceDx'].iloc[0]\n    tmp['NFOGQ'] = subject_meta.loc[subject_meta['Subject']==subject, 'NFOGQ'].iloc[0]\n\n\n    # add Id data\n    tmp['Id'] = file_id + '_' + tmp['Time'].astype(str)\n    defog_test = pd.concat([defog_test, tmp]).reset_index(drop=True)\n\nprint(defog.head(20))\n\n# unify fog and defog\ntest_df = pd.concat([fog_test, defog_test]).reset_index(drop=True)\n# print(train_df.head(20))\n\n# encode M/F to 1/0 and on/off to 1/0\ntest_df['Medication'] = np.where(test_df['Medication']=='on',1,0)\ntest_df['Sex'] = np.where(test_df['Sex']=='M',1,0)\nprint(test_df.head(20))\n\n# id for submission\nid = test_df['Id']\n# feature of test data\na_test = test_df.drop(['Time', 'Id'],axis=1)\nprint(a_test.head())\n\n\nprediction = model.predict(a_test)\n\n# submission data\nsub = pd.DataFrame(id, columns=['Id'])\nsub['StartHesitation'] = prediction[:, 0]\nsub['Turn'] = prediction[:, 1]\nsub['Walking'] = prediction[:, 2]\n\nprint(sub)\n\nsub.to_csv(\"submission.csv\", index=\"False\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-26T23:34:18.277022Z","iopub.execute_input":"2023-04-26T23:34:18.277504Z","iopub.status.idle":"2023-04-26T23:34:40.557159Z","shell.execute_reply.started":"2023-04-26T23:34:18.277466Z","shell.execute_reply":"2023-04-26T23:34:40.555721Z"},"trusted":true},"execution_count":null,"outputs":[]}]}