{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n#from sklearn import *\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.metrics import accuracy_score\nimport glob\n\ndef get_id(file):\n    return file.split('/')[-1].split('.')[0]\n\ndef build_submission(files, model, features):\n    \n    submission = []\n    \n    for file in files:\n        \n        df = pd.read_csv(file)\n        df['Id'] = get_id(file)\n        #df = df.fillna(0).reset_index(drop=True)\n        res = pd.DataFrame(np.round(model.predict(df[features]),3), columns=['StartHesitation', 'Turn' , 'Walking'])\n        df = pd.concat([df,res], axis=1)\n        #df['Id'] = df['Id'].astype(str) + '_' + df['Time'].astype(str)\n        submission.append(df[['Id','StartHesitation', 'Turn' , 'Walking']])\n\ncount = 0\n#def get_file_data(file):\ndef get_file_data(datasets, useBreak=False):\n    \n    data_frames = []\n    count = 0\n    \n    for file in datasets:\n    \n        if count > 200 and useBreak == True:\n           break\n        #print(file)\n        df = pd.read_csv(file)\n\n        df['Id'] = get_id(file)\n        df['Type'] = file.split('/')[-2]\n        \n        # filter out rows where either Valid or Task is false\n        try:\n            df = df[(df['Valid'] == True) & (df['Task'] == True)]        \n        except: \n            print(\"D\")\n\n        data_frames.append(df)\n        \n        count = count + 1\n        \n    return pd.concat(data_frames)\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-17T01:44:35.509424Z","iopub.execute_input":"2023-04-17T01:44:35.509850Z","iopub.status.idle":"2023-04-17T01:44:35.522438Z","shell.execute_reply.started":"2023-04-17T01:44:35.509813Z","shell.execute_reply":"2023-04-17T01:44:35.521240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2\n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-04-16T19:38:12.126574Z","iopub.execute_input":"2023-04-16T19:38:12.127017Z","iopub.status.idle":"2023-04-16T19:38:12.146690Z","shell.execute_reply.started":"2023-04-16T19:38:12.126981Z","shell.execute_reply":"2023-04-16T19:38:12.145273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = glob.glob('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/**/**')\ntest = glob.glob('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/**/**')\nsubjects = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/subjects.csv')\ntasks = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tasks.csv')\n\n#train","metadata":{"execution":{"iopub.status.busy":"2023-04-17T01:18:10.768799Z","iopub.execute_input":"2023-04-17T01:18:10.769347Z","iopub.status.idle":"2023-04-17T01:18:10.877929Z","shell.execute_reply.started":"2023-04-17T01:18:10.769297Z","shell.execute_reply":"2023-04-17T01:18:10.876291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = get_file_data(train, True)\ntest_data = get_file_data(test)","metadata":{"execution":{"iopub.status.busy":"2023-04-17T01:18:12.914499Z","iopub.execute_input":"2023-04-17T01:18:12.914955Z","iopub.status.idle":"2023-04-17T01:18:41.601434Z","shell.execute_reply.started":"2023-04-17T01:18:12.914899Z","shell.execute_reply":"2023-04-17T01:18:41.600257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data","metadata":{"execution":{"iopub.status.busy":"2023-04-16T19:38:39.418141Z","iopub.execute_input":"2023-04-16T19:38:39.418821Z","iopub.status.idle":"2023-04-16T19:38:39.449336Z","shell.execute_reply.started":"2023-04-16T19:38:39.418781Z","shell.execute_reply":"2023-04-16T19:38:39.448198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data","metadata":{"execution":{"iopub.status.busy":"2023-04-16T19:38:39.450776Z","iopub.execute_input":"2023-04-16T19:38:39.451159Z","iopub.status.idle":"2023-04-16T19:38:39.468462Z","shell.execute_reply.started":"2023-04-16T19:38:39.451124Z","shell.execute_reply":"2023-04-16T19:38:39.466888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = reduce_mem_usage(train_data)\ntest_data = reduce_mem_usage(test_data)\n\nprint(train_data.columns)\n#print(subjects.columns)\nprint(tasks.columns)\nprint(test_data.columns)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T19:38:39.470249Z","iopub.execute_input":"2023-04-16T19:38:39.470763Z","iopub.status.idle":"2023-04-16T19:38:39.774512Z","shell.execute_reply.started":"2023-04-16T19:38:39.470715Z","shell.execute_reply":"2023-04-16T19:38:39.773632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.merge(train_data, tasks, on='Id')","metadata":{"execution":{"iopub.status.busy":"2023-04-16T19:38:39.775622Z","iopub.execute_input":"2023-04-16T19:38:39.776155Z","iopub.status.idle":"2023-04-16T19:38:51.938708Z","shell.execute_reply.started":"2023-04-16T19:38:39.776119Z","shell.execute_reply":"2023-04-16T19:38:51.937233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data","metadata":{"execution":{"iopub.status.busy":"2023-04-16T19:38:51.940472Z","iopub.execute_input":"2023-04-16T19:38:51.940954Z","iopub.status.idle":"2023-04-16T19:38:51.967787Z","shell.execute_reply.started":"2023-04-16T19:38:51.940913Z","shell.execute_reply":"2023-04-16T19:38:51.966535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"merge testing\")\ntest_data = pd.merge(test_data, tasks, on='Id')","metadata":{"execution":{"iopub.status.busy":"2023-04-16T19:38:51.969494Z","iopub.execute_input":"2023-04-16T19:38:51.969835Z","iopub.status.idle":"2023-04-16T19:38:53.404089Z","shell.execute_reply.started":"2023-04-16T19:38:51.969804Z","shell.execute_reply":"2023-04-16T19:38:53.402463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data","metadata":{"execution":{"iopub.status.busy":"2023-04-16T19:38:53.405809Z","iopub.execute_input":"2023-04-16T19:38:53.406390Z","iopub.status.idle":"2023-04-16T19:38:53.429706Z","shell.execute_reply.started":"2023-04-16T19:38:53.406338Z","shell.execute_reply":"2023-04-16T19:38:53.428480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"training\")\nX = train_data[['AccV', 'AccML', 'AccAP']]\nX_test = test_data[['AccV', 'AccML', 'AccAP']]\ny = train_data['Valid'].values.ravel()\ny_test = test_data[['Task']]\n\n\n#print(train_data.shape)\n#print(test_data.shape)\n\n#print(X.shape, y.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-16T19:38:53.431279Z","iopub.execute_input":"2023-04-16T19:38:53.431772Z","iopub.status.idle":"2023-04-16T19:39:05.281480Z","shell.execute_reply.started":"2023-04-16T19:38:53.431734Z","shell.execute_reply":"2023-04-16T19:39:05.279949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.cluster import KMeans\n\nn_clusters = 5\nkmeans = KMeans(n_clusters=n_clusters)\nkmeans.fit(X)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T19:46:44.313311Z","iopub.execute_input":"2023-04-16T19:46:44.313805Z","iopub.status.idle":"2023-04-16T19:52:33.005172Z","shell.execute_reply.started":"2023-04-16T19:46:44.313759Z","shell.execute_reply":"2023-04-16T19:52:33.003687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path=\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/\"\nsub = pd.read_csv(path+'sample_submission.csv')\ntest = glob.glob(path+'test/**/**')\n\ncount = 0\nsub['t'] = 0\nsubmission = []\nfor f in test:\n    print(\"Working test \", count)\n    #if count > 30:\n    #    break\n    count = count + 1\n    df = pd.read_csv(f)\n    df['Id'] = f.split('/')[-1].split('.')[0]\n    df = df.fillna(0).reset_index(drop=True)\n    res = kmeans.predict(df[['AccV', 'AccML', 'AccAP']])\n    n = len(res)\n    if n % 3 != 0:\n        res = np.concatenate([res, np.zeros(3 - n % 3)])\n    res = res.reshape(-1, 3).astype('float64')\n    res = pd.DataFrame(res, columns=['StartHesitation', 'Turn', 'Walking'])\n    df = pd.concat([df, res], axis=1)\n    df['Id'] = df['Id'].astype(str) + '_' + df['Time'].astype(str)\n    submission.append(df[['Id', 'StartHesitation', 'Turn', 'Walking']])\nsubmission = pd.concat(submission)\nsubmission = pd.merge(sub[['Id', 't']], submission, how='left', on='Id').fillna(0.0)\nsubmission[['Id', 'StartHesitation', 'Turn', 'Walking']].to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-16T19:43:49.499859Z","iopub.execute_input":"2023-04-16T19:43:49.500250Z","iopub.status.idle":"2023-04-16T19:43:51.300994Z","shell.execute_reply.started":"2023-04-16T19:43:49.500211Z","shell.execute_reply":"2023-04-16T19:43:51.299917Z"},"trusted":true},"execution_count":null,"outputs":[]}]}