{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# -*- coding: utf-8 -*-\n\"\"\"FOG_Challenge.ipynb\n\nAutomatically generated by Colaboratory.\n\nOriginal file is located at\n    https://colab.research.google.com/drive/1RikNEB_f4TSnYhgYjaoO198uVW-4uUqu\n\"\"\"\n\nimport pandas as pd\nimport numpy as np\nimport os\n\ntrain_defog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog'\ntrain_tdcsfog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog'\n\ntest_defog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog'\ntest_tdcsfog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog'\n\ntrain_defog_dfs = []\ntrain_tdcsfog_dfs = []\n\ntest_defog_dfs = []\ntest_tdcsfog_dfs = []\n\ndef create_df(root, file, data_type, training):\n  path = os.path.join(root, file)\n  id = str(file.split('.')[0])\n  df = pd.read_csv(path)\n  df['Id'] = id\n  if(data_type == 'defog' and training):\n    df = df[df['Valid'] == True]\n    df = df[df['Task'] == True]\n    df = df.drop(['Valid', 'Task'], axis = 1)\n  return df\n\n\nfor root, subdir, files in os.walk(train_defog_path):\n  train_defog_dfs = [create_df(root, file, root.split('/')[-1], True) for file in files]\n\nfor root, subdir, files in os.walk(train_tdcsfog_path):\n  train_tdcsfog_dfs = [create_df(root, file, root.split('/')[-1], True) for file in files]\n\nfor root, subdir, files in os.walk(test_defog_path):\n  test_defog_dfs = [create_df(root, file, root.split('/')[-1], False) for file in files]\n\nfor root, subdir, files in os.walk(test_tdcsfog_path):\n  test_tdcsfog_dfs = [create_df(root, file, root.split('/')[-1], False) for file in files]","metadata":{"_uuid":"015c5872-fb85-4cf8-aca5-22f68b7bb2c2","_cell_guid":"d73dcf9b-7d16-4d07-8a66-3a55303a3342","execution":{"iopub.status.busy":"2023-04-20T21:41:28.864398Z","iopub.execute_input":"2023-04-20T21:41:28.864893Z","iopub.status.idle":"2023-04-20T21:42:16.094248Z","shell.execute_reply.started":"2023-04-20T21:41:28.864851Z","shell.execute_reply":"2023-04-20T21:42:16.093185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_defog_df = pd.concat(train_defog_dfs)\n\ntrain_tdcsfog_df = pd.concat(train_tdcsfog_dfs)\n\ntest_defog_df = pd.concat(test_defog_dfs)\ntest_tdcsfog_df = pd.concat(test_tdcsfog_dfs)\n\nsubjects_df = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/subjects.csv')\ndefog_metadata_df = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv')\ntdcsfog_metadata_df = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv')","metadata":{"execution":{"iopub.status.busy":"2023-04-20T21:44:10.014269Z","iopub.execute_input":"2023-04-20T21:44:10.014692Z","iopub.status.idle":"2023-04-20T21:44:10.962701Z","shell.execute_reply.started":"2023-04-20T21:44:10.014653Z","shell.execute_reply":"2023-04-20T21:44:10.961440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#making numerical data categorical\ndefog_metadata_df['Medication'] = [0 if x == 'off' else 1 for x in defog_metadata_df['Medication'].to_numpy()]\ntdcsfog_metadata_df['Medication'] = [0 if x == 'off' else 1 for x in tdcsfog_metadata_df['Medication'].to_numpy()]\nbinned_ages = pd.cut(subjects_df['Age'], [28, 47, 54, 61, 67, 74, 80, 87, 94], labels = ['0', '1', '2', '3', '4', '5', '6', '7'], include_lowest = True)\nsubjects_df['Age'] = pd.to_numeric(binned_ages, downcast = 'integer')\nsubjects_df['Sex'] = [0 if x == 'M' else 1 for x in subjects_df['Sex'].to_numpy()]\nsubjects_df.drop('UPDRSIII_Off', axis = 1, inplace = True)\nbinned_ratings =  pd.cut(subjects_df['UPDRSIII_On'], [5, 12, 20, 27, 34, 42, 49, 56, 64, 71, 79], labels = ['0', '1', '2', '3', '4', '5', '6', '7', '8', '9'], include_lowest = True)\nbinned_ratings = np.where(binned_ratings.isnull(), '-1', binned_ratings)\nbinned_questionnaire_scores = pd.cut(subjects_df['NFOGQ'], [0, 5, 8, 11, 14, 17, 20, 23, 26, 29], labels = ['0', '1', '2', '3', '4', '5', '6', '7', '8'], include_lowest = True)\nbinned_questionnaire_scores = np.where(binned_questionnaire_scores.isnull(), '-1', binned_questionnaire_scores)\nbinned_years_since = pd.cut(subjects_df['YearsSinceDx'], [0, 3, 6, 9, 12, 15, 18, 21, 24, 27, 30], labels = ['0', '1', '2', '3', '4', '5', '6', '7', '8', '9'], include_lowest = True)\nsubjects_df['NFOGQ'] = pd.to_numeric(binned_questionnaire_scores, downcast = 'integer')\nsubjects_df['YearsSinceDx'] = pd.to_numeric(binned_years_since, downcast = 'integer')\nsubjects_df['UPDRSIII_On'] = pd.to_numeric(binned_ratings, downcast = 'integer')\n\n\n#Since the metadata columns have better data for visits, I don't really need it here; i'll keep one of each subject record and then when I join with the metadata columns i'll get the other record back\n#I'm doing this instead of leaving this column alone because either A. I don't do this, join by subjects, and get duplicates or B. don't do this, join by subjects and visits, and maintain the NaN entries\nsubjects_df.sort_values(by='Visit', inplace = True)\nsubjects_df.drop_duplicates(subset=['Subject'], keep='first', inplace = True)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-04-20T21:44:13.465699Z","iopub.execute_input":"2023-04-20T21:44:13.466237Z","iopub.status.idle":"2023-04-20T21:44:13.506751Z","shell.execute_reply.started":"2023-04-20T21:44:13.466186Z","shell.execute_reply":"2023-04-20T21:44:13.505243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_subjects_metadata_df = pd.merge(subjects_df, defog_metadata_df, on=['Subject'])\n\ndefog_subjects_metadata_df.drop(['Visit_x', 'Visit_y'], axis = 1, inplace = True)\ntdcsfog_subjects_metadata_df = subjects_df.merge(tdcsfog_metadata_df, on=['Subject'])\n\ntdcsfog_subjects_metadata_df.drop(['Visit_x', 'Visit_y'], axis = 1, inplace = True)\ntdcsfog_subjects_metadata_df.drop('Test', axis = 1, inplace = True)\n\ntrain_defog_df = train_defog_df.merge(defog_subjects_metadata_df, on='Id')\n\ntest_defog_df = test_defog_df.merge(defog_subjects_metadata_df, on='Id')\n\ntrain_tdcsfog_df = train_tdcsfog_df.merge(tdcsfog_subjects_metadata_df, on='Id')\ntest_tdcsfog_df = test_tdcsfog_df.merge(tdcsfog_subjects_metadata_df, on='Id')\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-04-20T21:44:16.058342Z","iopub.execute_input":"2023-04-20T21:44:16.058812Z","iopub.status.idle":"2023-04-20T21:44:19.110256Z","shell.execute_reply.started":"2023-04-20T21:44:16.058771Z","shell.execute_reply":"2023-04-20T21:44:19.108712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_defog_df.isna().values.sum())\nprint(test_defog_df.isna().values.sum())\nprint(train_tdcsfog_df.isna().values.sum())\nprint(test_tdcsfog_df.isna().values.sum())","metadata":{"execution":{"iopub.status.busy":"2023-04-20T21:46:46.433158Z","iopub.execute_input":"2023-04-20T21:46:46.433651Z","iopub.status.idle":"2023-04-20T21:46:47.920853Z","shell.execute_reply.started":"2023-04-20T21:46:46.433607Z","shell.execute_reply":"2023-04-20T21:46:47.919558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\ntrain_defog_labels = train_defog_df[['StartHesitation', 'Turn', 'Walking']]\ntrain_defog_features = train_defog_df.drop(['StartHesitation', 'Turn', 'Walking', 'Subject', 'Id'], axis = 1)\ntrain_tdcsfog_labels = train_tdcsfog_df[['StartHesitation', 'Turn', 'Walking']]\ntrain_tdcsfog_features = train_tdcsfog_df.drop(['StartHesitation', 'Turn', 'Walking', 'Subject', 'Id'], axis = 1)\n\ndefog_rf = RandomForestClassifier(n_estimators = 120, n_jobs = -1, max_samples = 1000000, class_weight = \"balanced\")\ntdcsfog_rf = RandomForestClassifier(n_estimators = 120, n_jobs = -1, max_samples = 1000000, class_weight = \"balanced\")\n\ndefog_rf.fit(train_defog_features, train_defog_labels)\ntdcsfog_rf.fit(train_tdcsfog_features, train_tdcsfog_labels)","metadata":{"execution":{"iopub.status.busy":"2023-04-20T21:46:54.327245Z","iopub.execute_input":"2023-04-20T21:46:54.327743Z","iopub.status.idle":"2023-04-20T21:58:37.015547Z","shell.execute_reply.started":"2023-04-20T21:46:54.327699Z","shell.execute_reply":"2023-04-20T21:58:37.014014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import csv\ndefog_predict = defog_rf.predict(test_defog_df.drop(['Subject', 'Id'], axis = 1))\ntdcsfog_predict = tdcsfog_rf.predict(test_tdcsfog_df.drop(['Subject', 'Id'], axis = 1))\n\nsubmission = []\n\nfor idx, row in test_tdcsfog_df.iterrows():\n    row['Id'] = row['Id']+'_'+str(row['Time'])\n    prediction = np.append(row['Id'], [tdcsfog_predict[idx]])\n    submission.append(prediction)\n\nfor idx, row in test_defog_df.iterrows():\n    row['Id'] = row['Id']+'_'+str(row['Time'])\n    prediction = np.append(row['Id'], [defog_predict[idx]])\n    submission.append(prediction)\n\nsubmission_df = pd.DataFrame(submission)\nsubmission_df.columns = ['Id', 'StartHesitation', 'Turn', 'Walking']\n","metadata":{"execution":{"iopub.status.busy":"2023-04-20T23:02:11.462553Z","iopub.execute_input":"2023-04-20T23:02:11.463575Z","iopub.status.idle":"2023-04-20T23:02:44.096371Z","shell.execute_reply.started":"2023-04-20T23:02:11.463475Z","shell.execute_reply":"2023-04-20T23:02:44.095034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('/kaggle/working/submission.csv', 'w') as f:\n    # create the csv writer\n    writer = csv.writer(f)\n\n    # write a row to the csv file\n    writer.writerow(['Id', 'StartHesitation', 'Turn', 'Walking'])\n    writer.writerows(submission)\nprint(submission_df.head())","metadata":{"execution":{"iopub.status.busy":"2023-04-20T23:03:12.860608Z","iopub.execute_input":"2023-04-20T23:03:12.861049Z","iopub.status.idle":"2023-04-20T23:03:13.603564Z","shell.execute_reply.started":"2023-04-20T23:03:12.861012Z","shell.execute_reply":"2023-04-20T23:03:13.602587Z"},"trusted":true},"execution_count":null,"outputs":[]}]}