{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-20T22:48:42.978052Z","iopub.execute_input":"2023-04-20T22:48:42.978494Z","iopub.status.idle":"2023-04-20T22:48:43.112075Z","shell.execute_reply.started":"2023-04-20T22:48:42.978459Z","shell.execute_reply":"2023-04-20T22:48:43.111164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn import *\nimport glob\nimport gc\nfrom pathlib import Path","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:48:43.113716Z","iopub.execute_input":"2023-04-20T22:48:43.114764Z","iopub.status.idle":"2023-04-20T22:48:44.590803Z","shell.execute_reply.started":"2023-04-20T22:48:43.114723Z","shell.execute_reply":"2023-04-20T22:48:44.589686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_metadata = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv')\ntdcsfog_meta = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv')\nsubject_df = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/subjects.csv')","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:48:44.592138Z","iopub.execute_input":"2023-04-20T22:48:44.592758Z","iopub.status.idle":"2023-04-20T22:48:44.619639Z","shell.execute_reply.started":"2023-04-20T22:48:44.592703Z","shell.execute_reply":"2023-04-20T22:48:44.618415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X=pd.concat([defog_metadata,tdcsfog_meta])","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:48:44.624451Z","iopub.execute_input":"2023-04-20T22:48:44.624819Z","iopub.status.idle":"2023-04-20T22:48:44.637845Z","shell.execute_reply.started":"2023-04-20T22:48:44.624786Z","shell.execute_reply":"2023-04-20T22:48:44.636820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X['Medication'] = np.where(X['Medication'] != 'on', 0, 1)\nX_new=X.merge(subject_df,how='left',on='Subject').copy()\nX_new.fillna(0,inplace=True)\nX_new['Sex'] = np.where(X_new['Sex'] != 'M', 0, 1)","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:48:44.639114Z","iopub.execute_input":"2023-04-20T22:48:44.639522Z","iopub.status.idle":"2023-04-20T22:48:44.661793Z","shell.execute_reply.started":"2023-04-20T22:48:44.639491Z","shell.execute_reply":"2023-04-20T22:48:44.660457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bins_Age = [0,47,61,67,75,80,87,100]\nlabels_Age = [0,1,2,3,4,5,6]\nX_new['Age_Category'] = pd.cut(X_new['Age'],bins_Age,labels=labels_Age)","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:48:44.663958Z","iopub.execute_input":"2023-04-20T22:48:44.664788Z","iopub.status.idle":"2023-04-20T22:48:44.675716Z","shell.execute_reply.started":"2023-04-20T22:48:44.664735Z","shell.execute_reply":"2023-04-20T22:48:44.674527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bins_YearsDx = [0,3.38,6.34,9.29,12.25,15.21,18.17,21.13,24.08,27.04,30]\nlabels_YearsDx = [0,1,2,3,4,5,6,7,8,9]\nX_new['YearsSinceDx_Category'] = pd.cut(X_new['YearsSinceDx'],bins_YearsDx,labels=labels_YearsDx)","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:48:44.677297Z","iopub.execute_input":"2023-04-20T22:48:44.677619Z","iopub.status.idle":"2023-04-20T22:48:44.690279Z","shell.execute_reply.started":"2023-04-20T22:48:44.677589Z","shell.execute_reply":"2023-04-20T22:48:44.689434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_new.drop(['UPDRSIII_On', 'UPDRSIII_Off', 'NFOGQ', 'Visit_x', 'Visit_y', 'Test'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:48:44.691701Z","iopub.execute_input":"2023-04-20T22:48:44.692304Z","iopub.status.idle":"2023-04-20T22:48:44.703018Z","shell.execute_reply.started":"2023-04-20T22:48:44.692271Z","shell.execute_reply":"2023-04-20T22:48:44.701624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_new = X_new.drop_duplicates()\nX_new = X_new.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:48:44.704321Z","iopub.execute_input":"2023-04-20T22:48:44.705244Z","iopub.status.idle":"2023-04-20T22:48:44.720029Z","shell.execute_reply.started":"2023-04-20T22:48:44.705209Z","shell.execute_reply":"2023-04-20T22:48:44.719128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path=\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/\"\n\ntrain_defog = glob.glob(path+'train/defog/**')\ntrain_tdcsfog = glob.glob(path+'train/tdcsfog/**')","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:48:44.723726Z","iopub.execute_input":"2023-04-20T22:48:44.724408Z","iopub.status.idle":"2023-04-20T22:48:44.733724Z","shell.execute_reply.started":"2023-04-20T22:48:44.724341Z","shell.execute_reply":"2023-04-20T22:48:44.732797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_data(f):\n    df = pd.read_csv(f)\n    df['Id'] = f.split('/')[-1].split('.')[0]\n    df['data_type'] = f.split('/')[-2]\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:48:44.734784Z","iopub.execute_input":"2023-04-20T22:48:44.735472Z","iopub.status.idle":"2023-04-20T22:48:44.741185Z","shell.execute_reply.started":"2023-04-20T22:48:44.735439Z","shell.execute_reply":"2023-04-20T22:48:44.739955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_defog = pd.concat([get_data(f) for f in train_defog])\ndf_train_tdcsfog = pd.concat([get_data(f) for f in train_tdcsfog])","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:48:44.742700Z","iopub.execute_input":"2023-04-20T22:48:44.743012Z","iopub.status.idle":"2023-04-20T22:49:27.659946Z","shell.execute_reply.started":"2023-04-20T22:48:44.742982Z","shell.execute_reply":"2023-04-20T22:49:27.658623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_time(df, freq):\n    if freq == 100:\n        df['Time'] /= 100\n        df['Time'] = round(df['Time'], 3)\n    elif freq == 128:\n        df['Time'] = round(df['Time'] / 128, )\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:49:27.661532Z","iopub.execute_input":"2023-04-20T22:49:27.661871Z","iopub.status.idle":"2023-04-20T22:49:27.667932Z","shell.execute_reply.started":"2023-04-20T22:49:27.661838Z","shell.execute_reply":"2023-04-20T22:49:27.666885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_defog = convert_time(df_train_defog, 100)\ndf_train_tdcsfog = convert_time(df_train_tdcsfog, 128)","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:49:27.669271Z","iopub.execute_input":"2023-04-20T22:49:27.669979Z","iopub.status.idle":"2023-04-20T22:49:27.940295Z","shell.execute_reply.started":"2023-04-20T22:49:27.669935Z","shell.execute_reply":"2023-04-20T22:49:27.938955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_acc(df):\n    df['AccV'] *= 9.80665\n    df['AccML'] *= 9.80665\n    df['AccAP'] *= 9.80665\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:49:27.941817Z","iopub.execute_input":"2023-04-20T22:49:27.942275Z","iopub.status.idle":"2023-04-20T22:49:27.948535Z","shell.execute_reply.started":"2023-04-20T22:49:27.942225Z","shell.execute_reply":"2023-04-20T22:49:27.947340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_defog = convert_acc(df_train_defog)","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:49:27.950295Z","iopub.execute_input":"2023-04-20T22:49:27.950734Z","iopub.status.idle":"2023-04-20T22:49:28.266188Z","shell.execute_reply.started":"2023-04-20T22:49:27.950678Z","shell.execute_reply":"2023-04-20T22:49:28.264949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=pd.concat([df_train_defog,df_train_tdcsfog])","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:49:28.267527Z","iopub.execute_input":"2023-04-20T22:49:28.267863Z","iopub.status.idle":"2023-04-20T22:49:30.872456Z","shell.execute_reply.started":"2023-04-20T22:49:28.267831Z","shell.execute_reply":"2023-04-20T22:49:30.871238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:49:30.875530Z","iopub.execute_input":"2023-04-20T22:49:30.876276Z","iopub.status.idle":"2023-04-20T22:49:44.144758Z","shell.execute_reply.started":"2023-04-20T22:49:30.876226Z","shell.execute_reply":"2023-04-20T22:49:44.143411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combined_train = train.merge(X_new,how='left',on='Id').copy()","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:49:44.146114Z","iopub.execute_input":"2023-04-20T22:49:44.146460Z","iopub.status.idle":"2023-04-20T22:49:58.193783Z","shell.execute_reply.started":"2023-04-20T22:49:44.146427Z","shell.execute_reply":"2023-04-20T22:49:58.192567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combined_train['data_type'] = np.where(combined_train['data_type'] != 'defog', 0, 1)","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:49:58.195408Z","iopub.execute_input":"2023-04-20T22:49:58.195748Z","iopub.status.idle":"2023-04-20T22:50:01.302565Z","shell.execute_reply.started":"2023-04-20T22:49:58.195716Z","shell.execute_reply":"2023-04-20T22:50:01.301272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features=['Time', 'AccV', 'AccML', 'AccAP', 'Age_Category', 'Medication', 'YearsSinceDx_Category', 'data_type']\nTargets=['StartHesitation', 'Turn' , 'Walking']","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:50:01.303874Z","iopub.execute_input":"2023-04-20T22:50:01.304197Z","iopub.status.idle":"2023-04-20T22:50:01.310026Z","shell.execute_reply.started":"2023-04-20T22:50:01.304165Z","shell.execute_reply":"2023-04-20T22:50:01.308870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = model_selection.train_test_split(combined_train[features], combined_train[Targets], test_size=.30, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:50:01.311756Z","iopub.execute_input":"2023-04-20T22:50:01.312172Z","iopub.status.idle":"2023-04-20T22:50:08.187042Z","shell.execute_reply.started":"2023-04-20T22:50:01.312129Z","shell.execute_reply":"2023-04-20T22:50:08.185462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = ensemble.RandomForestRegressor(max_depth= 10, min_samples_leaf=1, min_samples_split=2, n_estimators=125, max_features='sqrt', random_state=42, n_jobs=-1)\nmodel.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-04-20T22:50:08.188658Z","iopub.execute_input":"2023-04-20T22:50:08.189619Z","iopub.status.idle":"2023-04-20T23:15:47.165281Z","shell.execute_reply.started":"2023-04-20T22:50:08.189578Z","shell.execute_reply":"2023-04-20T23:15:47.162204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(X_valid)\nprint(metrics.average_precision_score(y_valid, y_pred.clip(0.0,1.0)))","metadata":{"execution":{"iopub.status.busy":"2023-04-20T23:15:47.170533Z","iopub.execute_input":"2023-04-20T23:15:47.171100Z","iopub.status.idle":"2023-04-20T23:16:37.872541Z","shell.execute_reply.started":"2023-04-20T23:15:47.171045Z","shell.execute_reply":"2023-04-20T23:16:37.871233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_files = glob.glob(path + 'test/**/*.csv', recursive=True)\nsub = pd.read_csv(path+'sample_submission.csv')\nsub['t'] = 0\ntargets = ['Id', 'StartHesitation', 'Turn' , 'Walking']\n\nsubmit = []\n\nfor file in test_files:\n    test_df = pd.read_csv(file)\n    id_str = file.split('/')[-1].split('.')[0]\n\n    data_type = 1 if 'defog' in id_str else 0\n    df = test_df.assign(Id=id_str, data_type=data_type)\n    df = convert_time(df, 100) if data_type else convert_time(df, 128)\n    df = convert_acc(df) if data_type else df\n    df = df.merge(X_new, how='left', on='Id')\n    df = df.fillna(0).reset_index(drop=True)\n\n    results = pd.DataFrame(np.round(model.predict(df[features]),3), columns=Targets)\n    \n    df_ = pd.concat([df[['Id']], results], axis=1)\n    \n    df_['Id'] = df_['Id'].astype(str) + '_' + df_.index.astype(str)\n \n    submit.append(df_[targets])    \n    \nsubmit = pd.concat(submit)\nsubmit = pd.merge(sub[['Id']], submit, how='left', on='Id').fillna(0.0)\nsubmit[targets].to_csv('submit.csv', index=False)\nsubmit","metadata":{"execution":{"iopub.status.busy":"2023-04-20T23:20:33.334732Z","iopub.execute_input":"2023-04-20T23:20:33.335248Z","iopub.status.idle":"2023-04-20T23:20:36.185295Z","shell.execute_reply.started":"2023-04-20T23:20:33.335207Z","shell.execute_reply":"2023-04-20T23:20:36.184187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit[targets]","metadata":{"execution":{"iopub.status.busy":"2023-04-20T23:21:06.851847Z","iopub.execute_input":"2023-04-20T23:21:06.852278Z","iopub.status.idle":"2023-04-20T23:21:06.875815Z","shell.execute_reply.started":"2023-04-20T23:21:06.852238Z","shell.execute_reply":"2023-04-20T23:21:06.874425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit[targets].to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-20T23:21:19.740306Z","iopub.execute_input":"2023-04-20T23:21:19.740924Z","iopub.status.idle":"2023-04-20T23:21:20.575147Z","shell.execute_reply.started":"2023-04-20T23:21:19.740875Z","shell.execute_reply":"2023-04-20T23:21:20.573887Z"},"trusted":true},"execution_count":null,"outputs":[]}]}