{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"- Firstly we need to create a rectangular data set including both the train and the test set information. After that we can keep the train only to do the trainning model and the test data file now also include the information of the meta data set.\n\nStep to write the notebook:\n1. read the data\n2. Create rectangular files one per data set (train and test) (that is for each observation points, there should be the time lag and time lead accelaration of 3 interested series)\n Can we save it (TRY TO ACCESS IT FROM ANOTHER NOTEBOOK)\n3. Compute the classifier\n4. Run the classifier on the test set to generate the submission file\n","metadata":{}},{"cell_type":"markdown","source":"# 1. Read The data","metadata":{}},{"cell_type":"markdown","source":"## 1.1 Train data","metadata":{}},{"cell_type":"code","source":"import pandas as pd\npd.set_option('display.max_columns', None)\npd.set_option('display.max_rows', None)\n\nimport numpy as np\nimport glob # For importing datasets\nfrom tqdm.auto import tqdm # For progress bar\nfrom sklearn import *\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\np = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/'\n\n#train = glob.glob(p+'train/**/**') # Grabs all three training datasets\ntrain_tdcsfog_csv_list = glob.glob(p+'train/tdcsfog/**') # Grabs only the tdcsfog train dataset\ntrain_defog_csv_list = glob.glob(p+'train/defog/**') # Grabs only the tdcsfog train dataset\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-25T18:53:46.858516Z","iopub.execute_input":"2023-04-25T18:53:46.859025Z","iopub.status.idle":"2023-04-25T18:53:46.876016Z","shell.execute_reply.started":"2023-04-25T18:53:46.858968Z","shell.execute_reply":"2023-04-25T18:53:46.874539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = glob.glob(p+'test/**/**')\nsubjects = pd.read_csv(p+'subjects.csv')\ntasks = pd.read_csv(p+'tasks.csv')\nsub = pd.read_csv(p+'sample_submission.csv')\nevents=pd.read_csv(p+'events.csv')\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-25T18:53:46.879152Z","iopub.execute_input":"2023-04-25T18:53:46.879593Z","iopub.status.idle":"2023-04-25T18:53:47.105957Z","shell.execute_reply.started":"2023-04-25T18:53:46.879546Z","shell.execute_reply":"2023-04-25T18:53:47.104236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#subjects['NFOGQ'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T18:53:47.108000Z","iopub.execute_input":"2023-04-25T18:53:47.108533Z","iopub.status.idle":"2023-04-25T18:53:47.115103Z","shell.execute_reply.started":"2023-04-25T18:53:47.108478Z","shell.execute_reply":"2023-04-25T18:53:47.113293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_metadata=pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv')\ndefog_metadata=pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv')\n# daily_metadata=pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/daily_metadata.csv')\n\n# subjects.drop(['Visit'], axis=1, inplace=True)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-25T18:53:47.117820Z","iopub.execute_input":"2023-04-25T18:53:47.118332Z","iopub.status.idle":"2023-04-25T18:53:47.135703Z","shell.execute_reply.started":"2023-04-25T18:53:47.118279Z","shell.execute_reply":"2023-04-25T18:53:47.134034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_complex_tdcs = tdcsfog_metadata.merge(subjects,how='left',on='Subject').copy()\nmetadata_complex_tdcs['Medication']=metadata_complex_tdcs['Medication'].factorize()[0]\nmetadata_complex_defog = defog_metadata.merge(subjects,how='left',on='Subject').copy()\nmetadata_complex_defog['Medication'] = metadata_complex_defog['Medication'].factorize()[0]","metadata":{"execution":{"iopub.status.busy":"2023-04-25T18:53:47.139084Z","iopub.execute_input":"2023-04-25T18:53:47.139533Z","iopub.status.idle":"2023-04-25T18:53:47.163169Z","shell.execute_reply.started":"2023-04-25T18:53:47.139494Z","shell.execute_reply":"2023-04-25T18:53:47.161836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pathlib\ndef reader(f, meta):\n    try:\n        df = pd.read_csv(f,\n            usecols=['Time', 'AccV', 'AccML', 'AccAP', 'StartHesitation', 'Turn' , 'Walking']\n        )        \n        df['Id'] = f.split('/')[-1].split('.')[0]\n        df['Dataset'] = pathlib.Path(f).parts[-2]\n        df = pd.merge(df, meta, how='left', on='Id')\n        return df\n    except: pass    \n    \n    \n    \n    \n    \n#import pathlib\n#def reader(f):\n#    try:\n#        df = pd.read_csv(f, index_col=\"Time\", usecols=['Time', 'AccV', 'AccML', 'AccAP', 'StartHesitation', 'Turn' , 'Walking'])\n#        \n#        df['Id'] = f.split('/')[-1].split('.')[0]\n#        df['Module'] = pathlib.Path(f).parts[-2]       \n#        df = pd.merge(df, tasks[['Id','t_kmeans']], how='left', on='Id').fillna(-1)\n##         df = pd.merge(df, subjects[['Id','s_kmeans']], how='left', on='Id').fillna(-1)\n#        df = pd.merge(df, metadata_complex[['Id','Subject']+['Visit','Test','Medication','s_kmeans']], how='left', on='Id').fillna(-1)\n#        df_feats = fc.calculate(df, return_df=True, include_final_window=True, approve_sparsity=True, window_idx=\"begin\").astype(np.float32)\n#        df = df.merge(df_feats, how=\"left\", left_index=True, right_index=True)\n#        df.fillna(method=\"ffill\", inplace=True)\n#        return df\n#    except: pass\n\n#train = pd.concat([reader(f) for f in tqdm(train)]).fillna(0); print(train.shape)\n#cols = [c for c in train.columns if c not in ['Id','Subject','Module', 'Time', 'StartHesitation', 'Turn' , 'Walking', 'Valid', 'Task','Event']]\n#pcols = ['StartHesitation', 'Turn' , 'Walking']\n#scols = ['Id', 'StartHesitation', 'Turn' , 'Walking']\n\n\n\n\n\n# Concatenates the tcds train rows\ntrain_tcds = pd.concat([reader(f, metadata_complex_tdcs) for f in tqdm(train_tdcsfog_csv_list)])\ntrain_tcds = train_tcds.reset_index(drop=True)\nprint(train_tcds.shape)\n\n# Concatenates the defog train rows\ntrain_defog = pd.concat([reader(f, metadata_complex_defog) for f in tqdm(train_defog_csv_list)])\ntrain_defog = train_defog.reset_index(drop=True)\nprint(train_defog.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-25T18:53:47.165382Z","iopub.execute_input":"2023-04-25T18:53:47.166421Z","iopub.status.idle":"2023-04-25T18:54:45.848443Z","shell.execute_reply.started":"2023-04-25T18:53:47.166361Z","shell.execute_reply":"2023-04-25T18:54:45.846299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_tcds['Activities']=train_tcds['StartHesitation']*1 + train_tcds['Turn']*2 + train_tcds['Walking']*3\nfrom tabulate import tabulate\ntrain_tcds['Activities'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T18:54:45.850459Z","iopub.execute_input":"2023-04-25T18:54:45.850907Z","iopub.status.idle":"2023-04-25T18:54:45.976035Z","shell.execute_reply.started":"2023-04-25T18:54:45.850868Z","shell.execute_reply":"2023-04-25T18:54:45.974359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.2 Test data","metadata":{}},{"cell_type":"code","source":"#ADDING BY TRANG\ntest_tdcsfog_csv_list = glob.glob(p+'test/tdcsfog/**') # Grabs only the tdcsfog test dataset\ntest_defog_csv_list = glob.glob(p+'test/defog/**') # Grabs only the tdcsfog test dataset\n\n\ndef readernew(f, meta):\n    try:\n        df = pd.read_csv(f,\n            usecols=['Time', 'AccV', 'AccML', 'AccAP']\n        )        \n        df['Id'] = f.split('/')[-1].split('.')[0]\n        df['Dataset'] = pathlib.Path(f).parts[-2]\n        df = pd.merge(df, meta, how='left', on='Id')\n        return df\n    except: pass\n    \n# Concatenates the defog test rows\ntest_defog = pd.concat([readernew(f, metadata_complex_defog) for f in tqdm(test_defog_csv_list)])\ntest_defog = test_defog.reset_index(drop=True)\nprint(test_defog.shape)\n\n# Concatenates the tcds test rows\ntest_tcds = pd.concat([readernew(f, metadata_complex_tdcs) for f in tqdm(test_tdcsfog_csv_list)])\ntest_tcds = test_tcds.reset_index(drop=True)\n#print(test_tcds.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T18:54:45.978008Z","iopub.execute_input":"2023-04-25T18:54:45.978537Z","iopub.status.idle":"2023-04-25T18:54:46.568963Z","shell.execute_reply.started":"2023-04-25T18:54:45.978491Z","shell.execute_reply":"2023-04-25T18:54:46.567036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.3 Explore file Event","metadata":{}},{"cell_type":"code","source":"#events['Count']=1\n#events['Count'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T18:54:46.570793Z","iopub.execute_input":"2023-04-25T18:54:46.571210Z","iopub.status.idle":"2023-04-25T18:54:46.577932Z","shell.execute_reply.started":"2023-04-25T18:54:46.571174Z","shell.execute_reply":"2023-04-25T18:54:46.576080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#events['Kinetic'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T18:54:46.580431Z","iopub.execute_input":"2023-04-25T18:54:46.580821Z","iopub.status.idle":"2023-04-25T18:54:46.592399Z","shell.execute_reply.started":"2023-04-25T18:54:46.580786Z","shell.execute_reply":"2023-04-25T18:54:46.590802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#events['Type'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T18:54:46.595064Z","iopub.execute_input":"2023-04-25T18:54:46.595568Z","iopub.status.idle":"2023-04-25T18:54:46.605268Z","shell.execute_reply.started":"2023-04-25T18:54:46.595517Z","shell.execute_reply":"2023-04-25T18:54:46.603547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(events.head(1))","metadata":{"execution":{"iopub.status.busy":"2023-04-25T18:54:46.607314Z","iopub.execute_input":"2023-04-25T18:54:46.607669Z","iopub.status.idle":"2023-04-25T18:54:46.621951Z","shell.execute_reply.started":"2023-04-25T18:54:46.607636Z","shell.execute_reply":"2023-04-25T18:54:46.620610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print(events)\n#events['Init']=round(events['Init']*100)\n#events['Completion']=round(events['Completion']*100)\n#events['Index']=events.index\n#events['Duration']=events['Completion']-events['Init']\n#print(events.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T18:54:46.623808Z","iopub.execute_input":"2023-04-25T18:54:46.624164Z","iopub.status.idle":"2023-04-25T18:54:46.634325Z","shell.execute_reply.started":"2023-04-25T18:54:46.624132Z","shell.execute_reply":"2023-04-25T18:54:46.633199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#eventsnew=pd.DataFrame({'Id':[1],'Time':[1],'Task':[\"new\"]})\n#for i in range(0, 3711):\n#    df = pd.DataFrame({'Id': np.repeat(events['Id'].iloc[i], round(events['Duration'].iloc[i])),\n#                       'Time': np.repeat(events['Begin'].iloc[i], round(events['Duration'].iloc[i]))+range(0, round(events['Duration'].iloc[i])),\n#                       'Task': np.repeat(events['Task'].iloc[i], round(events['Duration'].iloc[i]))})\n#    print(eventsnew.tail(2))\n#    eventsnew=pd.concat([eventsnew, df], axis=0)    ","metadata":{"execution":{"iopub.status.busy":"2023-04-25T18:54:46.638946Z","iopub.execute_input":"2023-04-25T18:54:46.640051Z","iopub.status.idle":"2023-04-25T18:54:46.651772Z","shell.execute_reply.started":"2023-04-25T18:54:46.640009Z","shell.execute_reply":"2023-04-25T18:54:46.650816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.Explore file Task","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#tasks['Task'].value_counts()\n#tasks['New']=1\n#tasks['New'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T18:54:46.652775Z","iopub.execute_input":"2023-04-25T18:54:46.653151Z","iopub.status.idle":"2023-04-25T18:54:46.666113Z","shell.execute_reply.started":"2023-04-25T18:54:46.653115Z","shell.execute_reply":"2023-04-25T18:54:46.664843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tasks['Begin']=round(tasks['Begin']*100)\ntasks['End']=round(tasks['End']*100)\ntasks['Index']=tasks.index\n","metadata":{"execution":{"iopub.status.busy":"2023-04-25T18:54:46.668853Z","iopub.execute_input":"2023-04-25T18:54:46.669539Z","iopub.status.idle":"2023-04-25T18:54:46.683345Z","shell.execute_reply.started":"2023-04-25T18:54:46.669477Z","shell.execute_reply":"2023-04-25T18:54:46.682144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tasks['Duration']=tasks['End']-tasks['Begin']\nprint(tasks.head(10))\nprint(tasks.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-25T18:54:46.684627Z","iopub.execute_input":"2023-04-25T18:54:46.685014Z","iopub.status.idle":"2023-04-25T18:54:46.705555Z","shell.execute_reply.started":"2023-04-25T18:54:46.684977Z","shell.execute_reply":"2023-04-25T18:54:46.703912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tasksnew=pd.DataFrame({'Id':[1],'Time':[1],'Task':[\"new\"]})\nfor i in range(0, 2816):\n    #print(tasks['Duration'].iloc[i])\n    df = pd.DataFrame({'Id': np.repeat(tasks['Id'].iloc[i], round(tasks['Duration'].iloc[i])),\n                       'Time': np.repeat(tasks['Begin'].iloc[i], round(tasks['Duration'].iloc[i]))+range(0, round(tasks['Duration'].iloc[i])),\n                       'Task': np.repeat(tasks['Task'].iloc[i], round(tasks['Duration'].iloc[i]))})\n    tasksnew=pd.concat([tasksnew, df], axis=0)     \nprint(tasksnew.head(10))\n    ","metadata":{"execution":{"iopub.status.busy":"2023-04-25T19:45:39.548962Z","iopub.execute_input":"2023-04-25T19:45:39.549597Z","iopub.status.idle":"2023-04-25T19:54:16.577765Z","shell.execute_reply.started":"2023-04-25T19:45:39.549551Z","shell.execute_reply":"2023-04-25T19:54:16.575853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tasksmore=tasknew[-1,:]","metadata":{"execution":{"iopub.status.busy":"2023-04-25T19:18:05.561651Z","iopub.execute_input":"2023-04-25T19:18:05.562202Z","iopub.status.idle":"2023-04-25T19:18:05.575574Z","shell.execute_reply.started":"2023-04-25T19:18:05.562157Z","shell.execute_reply":"2023-04-25T19:18:05.573696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.5 Explore data daily","metadata":{}},{"cell_type":"code","source":"daily_metadata=pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/daily_metadata.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_subject=defog_metadata['Subject'].value_counts().rename_axis('Subject').reset_index(name='counts')\ndefog_subject['Subject']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_subject=tdcsfog_metadata['Subject'].value_counts().rename_axis('Subject').reset_index(name='counts')\ntdcsfog_subject['Subject']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects_subject=subjects['Subject'].value_counts().rename_axis('Subject').reset_index(name='counts')\nsubjects_subject['Subject']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"daily_subject=daily_metadata['Subject'].value_counts().rename_axis('Subject').reset_index(name='counts')\ndaily_subject['Subject']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.intersect1d(daily_subject['Subject'],defog_subject['Subject'],assume_unique=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.intersect1d(daily_subject['Subject'],subjects_subject['Subject'],assume_unique=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.intersect1d(tdcsfog_subject['Subject'],defog_subject['Subject'],assume_unique=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are only 36 subjects in the defog series that has information in the daily series. 9 other subjects in defog should be in the test set. we dont know who has FOG and who has not from those who are not in defog","metadata":{}},{"cell_type":"code","source":"daily_parquet_list = glob.glob(p+'unlabeled/**')\ndaily_parquet_list","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n# Read the Parquet file into a DataFrame\ndf = pd.read_parquet('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/unlabeled/48b636e0f5.parquet')\n# Print the first 10 rows of the DataFrame\nprint(df.head(10))","metadata":{"execution":{"iopub.status.busy":"2023-04-25T19:02:10.437188Z","iopub.status.idle":"2023-04-25T19:02:10.437990Z","shell.execute_reply.started":"2023-04-25T19:02:10.437569Z","shell.execute_reply":"2023-04-25T19:02:10.437608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pathlib\ndef readerparquet(f):\n    try:\n        df = pd.read_csv(f, index_col=\"Time\", usecols=['Time', 'AccV', 'AccML', 'AccAP'])\n        return df\n    except: pass\ndaily = pd.concat([reader(f) for f in tqdm(daily_parquet_list)]).fillna(0); print(daily.shape)","metadata":{},"execution_count":null,"outputs":[]}]}