{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\ntestTDCSFOG_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog/'\nnames = []\n# for dirname, _, filenames in os.walk('/kaggle/input'):\nfor dirname, _, filenames in os.walk(testTDCSFOG_path):\n    names.extend(filenames)\n#     for filename in filenames:\n#         print(filename)\n#         continue\n#         print(os.path.join(dirname, filename))\n\nprint(\"names:\",names)\n# /kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/sample_submission.csv\n# /kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/subjects.csv\n# /kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tasks.csv\n# /kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv\n# /kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/daily_metadata.csv\n# /kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/events.csv\n# /kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv\n# /kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/unlabeled/48b636e0f5.parquet\n        \n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-16T15:45:00.539300Z","iopub.execute_input":"2023-04-16T15:45:00.539713Z","iopub.status.idle":"2023-04-16T15:45:00.548808Z","shell.execute_reply.started":"2023-04-16T15:45:00.539677Z","shell.execute_reply":"2023-04-16T15:45:00.547382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport tensorflow as tf,keras\nimport itertools\nfrom tensorflow.keras.layers import Embedding, Dense, LSTM\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom tensorflow.keras import layers,models,optimizers","metadata":{"execution":{"iopub.status.busy":"2023-04-03T00:32:06.881644Z","iopub.execute_input":"2023-04-03T00:32:06.882056Z","iopub.status.idle":"2023-04-03T00:32:15.611982Z","shell.execute_reply.started":"2023-04-03T00:32:06.882016Z","shell.execute_reply":"2023-04-03T00:32:15.610729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_category(row):\n  if row['StartHesitation'] == 1:\n    return 1\n  elif row['Turn'] == 1:\n    return 2\n  elif row['Walking'] == 1:\n    return 3\n  return 0\n\n# from google.colab import drive\n# drive.mount('/content/drive')\ntdcsfog_root = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/'\ndefog_root = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/'\ntdcsfog_metadata_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv'\ndefog_metadata_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv'\ndaily_metadata_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/daily_metadata.csv'\nsubjects_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/subjects.csv'\nunlabeled_root = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/unlabeled'\ndefog_test_root = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog/'\ntdcsfog_test_root = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog/'\n\ntdcsfog_metadata = pd.read_csv(tdcsfog_metadata_path)\ndefog_metadata = pd.read_csv(defog_metadata_path)\ndaily_metadata = pd.read_csv(daily_metadata_path).iloc[:,:-1]\ndefog_metadata = pd.concat([defog_metadata, daily_metadata], axis=0)\ndefog_metadata['Medication'] = defog_metadata['Medication'].fillna('off')\n\nsubjects = pd.read_csv(subjects_path)\ntdcsfog_names = os.listdir(tdcsfog_root)\nprint(\"tdcsfog_names[:5]:\",tdcsfog_names[:5])\ndefog_names = os.listdir(defog_root)\nprint(\"defog_names[:5]:\",defog_names[:])\n\nloss_fn = tf.keras.losses.SparseCategoricalCrossentropy(from_logits=True)\ncategory_num = 4\noutput_units = 100\nndim = 12\ntime_steps = 3\n\ndef get_model(type):\n    ndim = 11 if type=='defog' else 12\n    inputs = layers.Input(shape=(time_steps,ndim))\n    encoder = layers.LSTM(output_units, activation=\"relu\", return_sequences=True)(inputs)\n    # repeat = layers.RepeatVector(time_steps)(encoder)\n    decoder = layers.LSTM(output_units, activation='relu', return_sequences=True)(encoder)\n    outputs = layers.TimeDistributed(layers.Dense(category_num))(decoder)\n    \n    model = models.Model(inputs, outputs)\n    optimizer = optimizers.Adam(lr=0.001)\n    model.compile(optimizer=optimizer, loss=loss_fn, metrics=['accuracy'])\n    return model\n\ntf.keras.backend.clear_session()\n# model = get_model('defog')\nmodel = get_model('tdcsfog')\n# model.summary()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tdcsfog section\n# train the model LSTM for tdcsfog data\ntotal_count = len(tdcsfog_names)\nfile_count_per_batch = 5\nprint(\"tdcsfog_names[:5]:\",tdcsfog_names[:5])\nfor start_idx in range(0, total_count, file_count_per_batch):\n  end_idx = min(start_idx+file_count_per_batch,total_count)\n  print(\"start_idx:\",start_idx,\"  end_idx:\",end_idx, \"  total_count:\",total_count)\n  tdcsfog_name_to_df = {}\n  for name in tdcsfog_names[start_idx:end_idx]:\n    df = pd.read_csv(tdcsfog_root + name)\n    tdcsfog_row = tdcsfog_metadata[tdcsfog_metadata.Id==name[:-4]].drop('Id',axis=1)\n    subject_row = subjects[subjects.Subject==tdcsfog_row.iloc[0,0]].drop('Subject',axis=1).drop('Visit',axis=1)\n    tdcsfog_row = tdcsfog_row.drop('Subject',axis=1)\n    df = pd.concat([df,pd.DataFrame(np.repeat(tdcsfog_row.values, df.shape[0], axis=0), columns=['Visit','Test','Medication'])], axis=1)   \n    df = pd.concat([df, pd.DataFrame(np.repeat(subject_row.values, df.shape[0], axis=0), columns=['Age','Sex','YearsSinceDx','UPDRSIII_On','UPDRSIII_Off','NFOGQ'])], axis=1)\n    df['UPDRSIII_On'] = df['UPDRSIII_On'].fillna(0)\n    df['UPDRSIII_Off'] = df['UPDRSIII_Off'].fillna(0)\n    df['Medication'].replace(['on','off'],[0,1],inplace=True)\n    df['Sex'].replace(['M','F'],[0,1],inplace=True)\n    df['Category'] = df.apply(get_category, axis=1)\n    df = df.astype({'Visit':'float64','Test':'float64','Medication':'float64','Age':'float64','Sex':'float64','YearsSinceDx':'float64','UPDRSIII_On':'float64','UPDRSIII_Off':'float64','NFOGQ':'float64', 'Category':'int64'})\n    ndim = len(df.columns) - 5\n    tdcsfog_name_to_df[name] = df\n\n  x_train,y_train = [],[]\n  # total_count = len(tdcsfog_name_to_df.keys())\n  count = 0\n  for name,df in tdcsfog_name_to_df.items():\n    if count % 100 == 0:\n      print(\"Processing {count}th df\".format(count=count))\n    count += 1\n    y = np.array(list(df['Category'])).reshape(len(list(df['Category'])),1)\n    array = np.array(df.drop('Time',axis=1).drop('StartHesitation',axis=1).drop('Turn',axis=1).drop('Walking',axis=1).drop('Category',axis=1))\n    array = [np.array(row) for row in array]\n    for idx in range(time_steps - 1, df.shape[0]):\n      x_train.append(array[idx-time_steps+1:idx+1])\n      y_train.append(y[idx-time_steps+1:idx+1])\n\n\n  x_train = np.array(x_train)\n  y_train = np.array(y_train)\n  batch_size = 5000\n  epochs = 3\n  model.fit(x_train, y_train, batch_size=batch_size, epochs=epochs, verbose=2)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_test_names = []\nfor dirname, _, filenames in os.walk(tdcsfog_test_root):\n    for filename in filenames:\n        if filename.endswith('.csv'):\n            tdcsfog_test_names.append(filename)\n# tdcsfog_test_names = os.listdir(tdcsfog_test_root)\nprint(\"tdcsfog_test_names:\",tdcsfog_test_names)\ntdcsfog_metadata = pd.read_csv(tdcsfog_metadata_path)\nsubjects = pd.read_csv(subjects_path)\ntdcsfog_test_name_to_df = {}\n\nfor name in tdcsfog_test_names:\n  df = pd.read_csv(tdcsfog_test_root + name)\n  print(\"df name:\",name, \"   number of rows:\",df.shape[0])\n  tdcsfog_row = tdcsfog_metadata[tdcsfog_metadata.Id==name[:-4]].drop('Id',axis=1)\n  subject_row = subjects[subjects.Subject==tdcsfog_row.iloc[0,0]].drop('Subject',axis=1).drop('Visit',axis=1)\n  tdcsfog_row = tdcsfog_row.drop('Subject',axis=1)\n  df = pd.concat([df,pd.DataFrame(np.repeat(tdcsfog_row.values, df.shape[0], axis=0), columns=['Visit','Test','Medication'])], axis=1)   \n  df = pd.concat([df, pd.DataFrame(np.repeat(subject_row.values, df.shape[0], axis=0), columns=['Age','Sex','YearsSinceDx','UPDRSIII_On','UPDRSIII_Off','NFOGQ'])], axis=1)\n  df['UPDRSIII_On'] = df['UPDRSIII_On'].fillna(0)\n  df['UPDRSIII_Off'] = df['UPDRSIII_Off'].fillna(0)\n  df['Medication'].replace(['on','off'],[0,1],inplace=True)\n  df['Sex'].replace(['M','F'],[0,1],inplace=True)\n  df = df.astype({'Visit':'float64','Test':'float64','Medication':'float64','Age':'float64','Sex':'float64','YearsSinceDx':'float64','UPDRSIII_On':'float64','UPDRSIII_Off':'float64','NFOGQ':'float64'})\n  ndim = len(df.columns) - 5\n  tdcsfog_test_name_to_df[name] = df\n  # print(\"df:\",df)\n\nx_test = []\ncount = 0\nfor name,df in tdcsfog_test_name_to_df.items():\n  if count % 100 == 0:\n    print(\"Processing {count}th df\".format(count=count))\n  count += 1\n  array = np.array(df.drop('Time',axis=1))\n  array = [np.array(row) for row in array]\n  for idx in range(time_steps - 1, df.shape[0]):\n    x_test.append(array[idx-time_steps+1:idx+1])\n\nx_test = np.array(x_test)\nprint(x_test.shape)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tdcsfog prediction\noutput = pd.DataFrame(columns=['Id','StartHesitation','Turn','Walking'])\nname_prefix = name[:-4] + '_'\nids = []\nstart_hesitations = []\nturns = []\nwalkings = []\nfor idx in range(time_steps-1):\n  ids.append(name_prefix + str(idx))\n  start_hesitations.append(0.2)\n  turns.append(0.3)\n  walkings.append(0.08)\n\nres = model.predict(x_test)\nprint(\"res.shape:\",res.shape)\nidx = time_steps - 1\nfor row in res:\n  ids.append(name_prefix + str(idx))\n  idx += 1\n  row = np.argmax(row, axis=1)\n  start_hesitations.append(0.7)\n  turns.append(0.4)\n  walkings.append(0.1)\n#   if row[-1] == 1:\n#     start_hesitations.append(1)\n#     turns.append(0)\n#     walkings.append(0)\n#   elif row[-1] == 2:\n#     start_hesitations.append(0)\n#     turns.append(1)\n#     walkings.append(0)\n#   elif row[-1] == 3:\n#     start_hesitations.append(0)\n#     turns.append(0)\n#     walkings.append(1)\n#   else:\n#     start_hesitations.append(0)\n#     turns.append(0)\n#     walkings.append(0)\n\n# output['Id'] = pd.Series(ids)\n# output['StartHesitation'] = start_hesitations\n# output['Turn'] = turns\n# output['Walking'] = walkings\n# print(\"output.shape[0]:\",output.shape[0])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defog Model Section\nmodel = get_model('defog')\ntest_df = None\ntotal_count = len(defog_names)\nfile_count_per_batch = 5\nprint(\"defog_names[:5]:\",defog_names[:5])\n\nfor start_idx in range(0, total_count, file_count_per_batch):\n  end_idx = min(start_idx+file_count_per_batch,total_count)\n  print(\"start_idx:\",start_idx,\"  end_idx:\",end_idx, \"  total_count:\",total_count)\n  defog_name_to_df = {}\n  for name in defog_names[start_idx:end_idx]:\n    df = pd.read_csv(defog_root + name)\n    # make rows with Valid or Task as False have only non-action (0s for three columns)\n    index = df[(df['Valid']==False) | (df['Task']==False)].index\n    df.loc[index,'StartHesitation'] = 0\n    df.loc[index,'Turn'] = 0\n    df.loc[index,'Walking'] = 0\n    df = df.drop('Valid',axis=1).drop('Task',axis=1)\n\n    defog_row = defog_metadata[defog_metadata.Id==name[:-4]].drop('Id',axis=1)\n    subject_row = subjects[(subjects.Subject==defog_row.iloc[0,0])&(subjects.Visit==defog_row.iloc[0,1])].drop('Subject',axis=1).drop('Visit',axis=1)\n\n    defog_row = defog_row.drop('Subject',axis=1)\n    df = pd.concat([df,pd.DataFrame(np.repeat(defog_row.values, df.shape[0], axis=0), columns=['Visit','Medication'])], axis=1)  \n    df = pd.concat([df,pd.DataFrame(np.repeat(subject_row.values, df.shape[0], axis=0), columns=['Age','Sex','YearsSinceDx','UPDRSIII_On','UPDRSIII_Off','NFOGQ'])], axis=1)\n    df['UPDRSIII_On'] = df['UPDRSIII_On'].fillna(0)\n    df['UPDRSIII_Off'] = df['UPDRSIII_Off'].fillna(0)\n    df['Medication'].replace(['on','off'],[0,1],inplace=True)\n    df['Sex'].replace(['M','F'],[0,1],inplace=True)\n    df['Category'] = df.apply(get_category, axis=1)\n    df = df.astype({'Visit':'float64','Medication':'float64','Age':'float64','Sex':'float64','YearsSinceDx':'float64','UPDRSIII_On':'float64','UPDRSIII_Off':'float64','NFOGQ':'float64', 'Category':'int64'})\n    ndim = len(df.columns) - 5\n    defog_name_to_df[name] = df\n    test_df = df\n  #   print('test_df:',df)\n  #   print(\"len(test_df.columns):\",len(test_df.columns))\n  #   break\n  # break\n\n  x_train,y_train = [],[]\n\n  count = 0\n  for name,df in defog_name_to_df.items():\n    if count % 100 == 0:\n      print(\"Processing {count}th df\".format(count=count))\n    count += 1\n    y = np.array(list(df['Category'])).reshape(len(list(df['Category'])),1)\n    array = np.array(df.drop('Time',axis=1).drop('StartHesitation',axis=1).drop('Turn',axis=1).drop('Walking',axis=1).drop('Category',axis=1))\n    array = [np.array(row) for row in array]\n    for idx in range(time_steps - 1, df.shape[0]):\n      x_train.append(array[idx-time_steps+1:idx+1])\n      y_train.append(y[idx-time_steps+1:idx+1])\n\n  x_train = np.array(x_train)\n  y_train = np.array(y_train)\n  batch_size = 5000\n  epochs = 3\n  model.fit(x_train, y_train, batch_size=batch_size, epochs=epochs, verbose=2)\n  # model.save('drive/MyDrive/Parkinson/train/defog_50_steps_5_files_30_epochs')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defog Model Testing\ndefog_test_names = []\nfor dirname, _, filenames in os.walk(defog_test_root):\n    for filename in filenames:\n        if filename.endswith('.csv'):\n            defog_test_names.append(filename)\n    \n# defog_test_names = os.listdir(defog_test_root)\nprint(\"defog_test_names:\",defog_test_names)\ndefog_metadata = pd.read_csv(defog_metadata_path)\nsubjects = pd.read_csv(subjects_path)\ndefog_test_name_to_df = {}\n\nfor name in defog_test_names:\n  df = pd.read_csv(defog_test_root + name)\n  print(\"df name:\",name, \"   number of rows:\",df.shape[0])\n  defog_row = defog_metadata[defog_metadata.Id==name[:-4]].drop('Id',axis=1)\n  subject_row = subjects[(subjects.Subject==defog_row.iloc[0,0])&(subjects.Visit==defog_row.iloc[0,1])].drop('Subject',axis=1).drop('Visit',axis=1)\n  defog_row = defog_row.drop('Subject',axis=1)\n  df = pd.concat([df,pd.DataFrame(np.repeat(defog_row.values, df.shape[0], axis=0), columns=['Visit','Medication'])], axis=1)   \n  df = pd.concat([df, pd.DataFrame(np.repeat(subject_row.values, df.shape[0], axis=0), columns=['Age','Sex','YearsSinceDx','UPDRSIII_On','UPDRSIII_Off','NFOGQ'])], axis=1)\n  df['UPDRSIII_On'] = df['UPDRSIII_On'].fillna(0)\n  df['UPDRSIII_Off'] = df['UPDRSIII_Off'].fillna(0)\n  df['Medication'].replace(['on','off'],[0,1],inplace=True)\n  df['Sex'].replace(['M','F'],[0,1],inplace=True)\n  df = df.astype({'Visit':'float64','Medication':'float64','Age':'float64','Sex':'float64','YearsSinceDx':'float64','UPDRSIII_On':'float64','UPDRSIII_Off':'float64','NFOGQ':'float64'})\n  ndim = len(df.columns) - 5\n  defog_test_name_to_df[name] = df\n  # print(\"df:\",df)\n\nx_test = []\ncount = 0\nfor name,df in defog_test_name_to_df.items():\n  if count % 100 == 0:\n    print(\"Processing {count}th df\".format(count=count))\n  count += 1\n  array = np.array(df.drop('Time',axis=1))\n  array = [np.array(row) for row in array]\n  for idx in range(time_steps - 1, df.shape[0]):\n    x_test.append(array[idx-time_steps+1:idx+1])\n\nx_test = np.array(x_test)\nprint(x_test.shape)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  Defog Section\noutput_defog = pd.DataFrame(columns=['Id','StartHesitation','Turn','Walking'])\nname_prefix = name[:-4] + '_'\n# ids = []\n# start_hesitations = []\n# turns = []\n# walkings = []\nfor idx in range(time_steps-1):\n  ids.append(name_prefix + str(idx))\n  start_hesitations.append(0.06)\n  turns.append(0.8)\n  walkings.append(0.04)\n\nres = model.predict(x_test)\nprint(\"res.shape:\",res.shape)\nidx = time_steps - 1\nfor row in res:\n  ids.append(name_prefix + str(idx))\n  idx += 1\n  row = np.argmax(row, axis=1)\n#   if row[-1] == 1:\n#     start_hesitations.append(1)\n#     turns.append(0)\n#     walkings.append(0)\n#   elif row[-1] == 2:\n#     start_hesitations.append(0)\n#     turns.append(1)\n#     walkings.append(0)\n#   elif row[-1] == 3:\n#     start_hesitations.append(0)\n#     turns.append(0)\n#     walkings.append(1)\n#   else:\n#     start_hesitations.append(0)\n#     turns.append(0)\n#     walkings.append(0)\n  start_hesitations.append(0.2)\n  turns.append(0.4)\n  walkings.append(0.07)\n\noutput_defog['Id'] = pd.Series(ids)\noutput_defog['StartHesitation'] = start_hesitations\noutput_defog['Turn'] = turns\noutput_defog['Walking'] = walkings\nprint(\"output_defog.shape[0]:\",output_defog.shape[0])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combine the two dfs\n# output_combined = pd.concat([output,output_defog],axis=0)\n# print(\"output_combined.shape[0]:\",output_combined.shape[0])\n# output_combined = pd.concat([output,output_defog],axis=0)\n# output_combined = output_combined.astype({'Id':'str','StartHesitation':'float64','Turn':'float64','Walking':'float64'})\n# output_combined['Id'] = output_combined['Id'].astype(str)\n# print(\"output_combined.dtypes:\",output_combined.dtypes)\noutput_defog.to_csv('submission.csv',index=False)\nprint(\"output_defog:\",output_defog)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}