{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Compiled subject information\n\nThis notebook goes through the metadata files and compiles all the information in 'subjects.csv', 'tdcsfog_metadata.csv', 'defog_metadata.csv', 'daily_metadata.csv' files into 1 dictionary, which is in turn saved as a .pkl file.","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"## Import","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport pickle\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-27T12:19:11.659099Z","iopub.execute_input":"2023-03-27T12:19:11.659524Z","iopub.status.idle":"2023-03-27T12:19:11.665151Z","shell.execute_reply.started":"2023-03-27T12:19:11.659485Z","shell.execute_reply":"2023-03-27T12:19:11.663711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load files","metadata":{}},{"cell_type":"code","source":"datapath = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction'","metadata":{"execution":{"iopub.status.busy":"2023-03-27T11:46:35.062869Z","iopub.execute_input":"2023-03-27T11:46:35.063266Z","iopub.status.idle":"2023-03-27T11:46:35.067926Z","shell.execute_reply.started":"2023-03-27T11:46:35.063232Z","shell.execute_reply":"2023-03-27T11:46:35.066865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"daily_m = pd.read_csv(os.path.join(datapath, 'daily_metadata.csv'))\ndefog_m = pd.read_csv(os.path.join(datapath, 'defog_metadata.csv'))\ntdcsfog_m = pd.read_csv(os.path.join(datapath, 'tdcsfog_metadata.csv'))\nevents = pd.read_csv(os.path.join(datapath, 'events.csv'))\ntasks = pd.read_csv(os.path.join(datapath, 'tasks.csv'))\nsubjects = pd.read_csv(os.path.join(datapath, 'subjects.csv'))","metadata":{"execution":{"iopub.status.busy":"2023-03-27T11:46:35.793576Z","iopub.execute_input":"2023-03-27T11:46:35.794062Z","iopub.status.idle":"2023-03-27T11:46:35.855121Z","shell.execute_reply.started":"2023-03-27T11:46:35.794017Z","shell.execute_reply":"2023-03-27T11:46:35.854089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### tdcsfog","metadata":{}},{"cell_type":"code","source":"tdcsfog_path = os.path.join(datapath, 'train','tdcsfog')","metadata":{"execution":{"iopub.status.busy":"2023-03-27T11:47:33.424373Z","iopub.execute_input":"2023-03-27T11:47:33.425343Z","iopub.status.idle":"2023-03-27T11:47:33.430938Z","shell.execute_reply.started":"2023-03-27T11:47:33.425287Z","shell.execute_reply":"2023-03-27T11:47:33.429644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_ids = list(tdcsfog_m['Id'])\ntdcsfog_ids.sort()","metadata":{"execution":{"iopub.status.busy":"2023-03-27T11:48:57.954621Z","iopub.execute_input":"2023-03-27T11:48:57.955108Z","iopub.status.idle":"2023-03-27T11:48:57.960876Z","shell.execute_reply.started":"2023-03-27T11:48:57.955068Z","shell.execute_reply":"2023-03-27T11:48:57.959305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_csvs = os.listdir(tdcsfog_path)\ntdcsfog_csv_ids = [x.split('.')[0] for x in tdcsfog_csvs]\ntdcsfog_csv_ids.sort()","metadata":{"execution":{"iopub.status.busy":"2023-03-27T11:48:59.549748Z","iopub.execute_input":"2023-03-27T11:48:59.550180Z","iopub.status.idle":"2023-03-27T11:48:59.559614Z","shell.execute_reply.started":"2023-03-27T11:48:59.550138Z","shell.execute_reply":"2023-03-27T11:48:59.558602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_ids == tdcsfog_csv_ids","metadata":{"execution":{"iopub.status.busy":"2023-03-27T11:49:01.277907Z","iopub.execute_input":"2023-03-27T11:49:01.278387Z","iopub.status.idle":"2023-03-27T11:49:01.286308Z","shell.execute_reply.started":"2023-03-27T11:49:01.278342Z","shell.execute_reply":"2023-03-27T11:49:01.284920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* All ids in the tdcsfog_metadata.csv file are present in the tdcsfog train folder.","metadata":{}},{"cell_type":"markdown","source":"### defog","metadata":{}},{"cell_type":"code","source":"defog_path = os.path.join(datapath, 'train', 'defog')","metadata":{"execution":{"iopub.status.busy":"2023-03-27T11:49:35.800673Z","iopub.execute_input":"2023-03-27T11:49:35.801138Z","iopub.status.idle":"2023-03-27T11:49:35.807589Z","shell.execute_reply.started":"2023-03-27T11:49:35.801098Z","shell.execute_reply":"2023-03-27T11:49:35.806109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_ids = list(defog_m['Id'])\ndefog_ids.sort()","metadata":{"execution":{"iopub.status.busy":"2023-03-27T11:49:46.105041Z","iopub.execute_input":"2023-03-27T11:49:46.105670Z","iopub.status.idle":"2023-03-27T11:49:46.113230Z","shell.execute_reply.started":"2023-03-27T11:49:46.105606Z","shell.execute_reply":"2023-03-27T11:49:46.111784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_csvs = os.listdir(defog_path)\ndefog_csv_ids = [x.split('.')[0] for x in defog_csvs]","metadata":{"execution":{"iopub.status.busy":"2023-03-27T11:50:47.128301Z","iopub.execute_input":"2023-03-27T11:50:47.129402Z","iopub.status.idle":"2023-03-27T11:50:47.137936Z","shell.execute_reply.started":"2023-03-27T11:50:47.129342Z","shell.execute_reply":"2023-03-27T11:50:47.136266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_ids == defog_csv_ids","metadata":{"execution":{"iopub.status.busy":"2023-03-27T11:50:53.404323Z","iopub.execute_input":"2023-03-27T11:50:53.404785Z","iopub.status.idle":"2023-03-27T11:50:53.412425Z","shell.execute_reply.started":"2023-03-27T11:50:53.404748Z","shell.execute_reply":"2023-03-27T11:50:53.411152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(defog_ids), len(defog_csv_ids)","metadata":{"execution":{"iopub.status.busy":"2023-03-27T11:51:03.982551Z","iopub.execute_input":"2023-03-27T11:51:03.983017Z","iopub.status.idle":"2023-03-27T11:51:03.990918Z","shell.execute_reply.started":"2023-03-27T11:51:03.982978Z","shell.execute_reply":"2023-03-27T11:51:03.989485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* The defog_metadata.csv file contains more ids than the defog train folder has files.","metadata":{}},{"cell_type":"code","source":"notype_path = os.path.join(datapath, 'train', 'notype')\nnotype_csvs = os.listdir(notype_path)\nnotype_csv_ids = [x.split('.')[0] for x in notype_csvs]","metadata":{"execution":{"iopub.status.busy":"2023-03-27T11:53:00.443429Z","iopub.execute_input":"2023-03-27T11:53:00.443859Z","iopub.status.idle":"2023-03-27T11:53:00.452106Z","shell.execute_reply.started":"2023-03-27T11:53:00.443823Z","shell.execute_reply":"2023-03-27T11:53:00.450974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_defog_csvs = defog_csv_ids + notype_csv_ids\nall_defog_csvs.sort()","metadata":{"execution":{"iopub.status.busy":"2023-03-27T11:53:04.521229Z","iopub.execute_input":"2023-03-27T11:53:04.521967Z","iopub.status.idle":"2023-03-27T11:53:04.526819Z","shell.execute_reply.started":"2023-03-27T11:53:04.521924Z","shell.execute_reply":"2023-03-27T11:53:04.525612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_defog_csvs == defog_ids","metadata":{"execution":{"iopub.status.busy":"2023-03-27T11:53:10.979711Z","iopub.execute_input":"2023-03-27T11:53:10.980124Z","iopub.status.idle":"2023-03-27T11:53:10.987716Z","shell.execute_reply.started":"2023-03-27T11:53:10.980085Z","shell.execute_reply":"2023-03-27T11:53:10.986417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Some of the defog files are in the 'notype' folder.","metadata":{}},{"cell_type":"markdown","source":"## Compile subject information","metadata":{}},{"cell_type":"code","source":"all_subjects = list(set(subjects['Subject']))","metadata":{"execution":{"iopub.status.busy":"2023-03-27T11:55:00.021653Z","iopub.execute_input":"2023-03-27T11:55:00.022057Z","iopub.status.idle":"2023-03-27T11:55:00.027634Z","shell.execute_reply.started":"2023-03-27T11:55:00.022021Z","shell.execute_reply":"2023-03-27T11:55:00.026387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subject_dict = {}\nfor s in all_subjects:\n    subject_dict[s] = {}\n    visits = []\n    sdf = subjects[subjects['Subject'] == s].reset_index(drop=True)\n    # subject.csv\n    for i in range(len(sdf)):\n        row = sdf.iloc[i]\n        visit_dict = {}\n        visit_dict['Visit'] = row['Visit']\n        visit_dict['Age'] = row['Age']\n        visit_dict['Sex'] = row['Sex']\n        visit_dict['YearsSinceDx'] = row['YearsSinceDx']\n        visit_dict['UPDRSIII_On'] = row['UPDRSIII_On']\n        visit_dict['UPDRSIII_Off'] = row['UPDRSIII_Off']\n        visit_dict['NFOGQ'] = row['NFOGQ']\n        visits.append(visit_dict)\n    subject_dict[s]['visits'] = visits\n    # has recording\n    has_rec = {}\n    has_rec['tdcsfog'] = s in tdcsfog_m['Subject'].values\n    has_rec['defog'] = s in defog_m['Subject'].values\n    has_rec['daily'] = s in daily_m['Subject'].values\n    subject_dict[s]['has_rec'] = has_rec\n    # tdcsfog\n    tdcsfog_recs = []\n    if has_rec['tdcsfog']:\n        sdf = tdcsfog_m[tdcsfog_m['Subject'] == s].reset_index(drop=True)\n        for i in range(len(sdf)):\n            rec_dict = {}\n            row = sdf.iloc[i]\n            rec_dict['Id'] = row['Id']\n            rec_dict['Visit'] = row['Visit']\n            rec_dict['Test'] = row['Test']\n            rec_dict['Medication'] = row['Medication']\n            tdcsfog_recs.append(rec_dict)\n    else:\n        tdcsfog_recs.append(None)\n    # defog\n    defog_recs = []\n    if has_rec['defog']:\n        sdf = defog_m[defog_m['Subject'] == s].reset_index(drop=True)\n        for i in range(len(sdf)):\n            rec_dict = {}\n            row = sdf.iloc[i]\n            rec_dict['Id'] = row['Id']\n            rec_dict['Visit'] = row['Visit']\n            rec_dict['Medication'] = row['Medication']\n            defog_recs.append(rec_dict)\n    else:\n        defog_recs.append(None)\n    subject_dict[s]['defog'] = defog_recs\n    # daily\n    daily_recs = []\n    if has_rec['daily']:\n        sdf = daily_m[daily_m['Subject'] == s].reset_index(drop=True)\n        for i in range(len(sdf)):\n            rec_dict = {}\n            row = sdf.iloc[i]\n            rec_dict['Id'] = row['Id']\n            rec_dict['Visit'] = row['Visit']\n            rec_dict['Rec_Begin'] = row[\"Beginning of recording [00:00-23:59]\"]\n            daily_recs.append(rec_dict)\n    else:\n        daily_recs.append(None)\n    subject_dict[s]['daily'] = daily_recs","metadata":{"execution":{"iopub.status.busy":"2023-03-27T11:58:56.826667Z","iopub.execute_input":"2023-03-27T11:58:56.827074Z","iopub.status.idle":"2023-03-27T11:58:57.210918Z","shell.execute_reply.started":"2023-03-27T11:58:56.827037Z","shell.execute_reply":"2023-03-27T11:58:57.209608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subject_dict","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:18:50.771795Z","iopub.execute_input":"2023-03-27T12:18:50.772632Z","iopub.status.idle":"2023-03-27T12:18:50.886920Z","shell.execute_reply.started":"2023-03-27T12:18:50.772582Z","shell.execute_reply":"2023-03-27T12:18:50.885699Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('/kaggle/working/subject_dict.pkl','wb') as f:\n    pickle.dump(subject_dict, f)","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:20:25.438951Z","iopub.execute_input":"2023-03-27T12:20:25.439411Z","iopub.status.idle":"2023-03-27T12:20:25.449627Z","shell.execute_reply.started":"2023-03-27T12:20:25.439370Z","shell.execute_reply":"2023-03-27T12:20:25.448069Z"},"trusted":true},"execution_count":null,"outputs":[]}]}