{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-14T20:58:58.409842Z","iopub.execute_input":"2023-04-14T20:58:58.410313Z","iopub.status.idle":"2023-04-14T20:58:58.557612Z","shell.execute_reply.started":"2023-04-14T20:58:58.410265Z","shell.execute_reply":"2023-04-14T20:58:58.556254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport glob # For importing datasets\nfrom tqdm.auto import tqdm # For progress bar\nfrom sklearn import *\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\np = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/'\n\n#train = glob.glob(p+'train/**/**') # Grabs all three training datasets\n#train_tdcsfog_csv_list = glob.glob(p+'train/tdcsfog/**') # Grabs only the tdcsfog train dataset\ntrain_defog_csv_list = glob.glob(p+'train/defog/**') # Grabs only the tdcsfog train dataset\ntrain_notype_csv_list = glob.glob(p+'train/notype/**') \n#test = glob.glob(p+'test/**/**')\n#subjects = pd.read_csv(p+'subjects.csv')\n#tasks = pd.read_csv(p+'tasks.csv')\n#sub = pd.read_csv(p+'sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-04-14T23:25:08.528392Z","iopub.execute_input":"2023-04-14T23:25:08.529668Z","iopub.status.idle":"2023-04-14T23:25:10.253383Z","shell.execute_reply.started":"2023-04-14T23:25:08.529622Z","shell.execute_reply":"2023-04-14T23:25:10.251848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pathlib\ndef reader(f):\n    try:\n        df = pd.read_csv(f,\n            usecols=['Time', 'Valid', 'Task']\n        )        \n        df['Id'] = f.split('/')[-1].split('.')[0]\n        df['Dataset'] = pathlib.Path(f).parts[-2]\n        #df = pd.merge(df, meta, how='left', on='Id')\n        return df\n    except: pass\n\n# Concatenates the defog train rows\ntrain_defog = pd.concat([reader(f) for f in tqdm(train_defog_csv_list)])\ntrain_defog = train_defog.reset_index(drop=True)\nprint(train_defog.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-14T23:28:28.016411Z","iopub.execute_input":"2023-04-14T23:28:28.016816Z","iopub.status.idle":"2023-04-14T23:28:53.371500Z","shell.execute_reply.started":"2023-04-14T23:28:28.016782Z","shell.execute_reply":"2023-04-14T23:28:53.370105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_df = pd.DataFrame(train_defog)","metadata":{"execution":{"iopub.status.busy":"2023-04-14T23:30:23.820095Z","iopub.execute_input":"2023-04-14T23:30:23.821100Z","iopub.status.idle":"2023-04-14T23:30:23.827158Z","shell.execute_reply.started":"2023-04-14T23:30:23.821042Z","shell.execute_reply":"2023-04-14T23:30:23.825628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_defog['Valid'].value_counts()/train_defog.shape[0]","metadata":{"execution":{"iopub.status.busy":"2023-04-11T20:43:13.094546Z","iopub.execute_input":"2023-04-11T20:43:13.094995Z","iopub.status.idle":"2023-04-11T20:43:13.197296Z","shell.execute_reply.started":"2023-04-11T20:43:13.094952Z","shell.execute_reply":"2023-04-11T20:43:13.195835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(np.unique(train_defog['Id']))","metadata":{"execution":{"iopub.status.busy":"2023-04-14T21:05:22.876324Z","iopub.execute_input":"2023-04-14T21:05:22.876896Z","iopub.status.idle":"2023-04-14T21:05:33.255339Z","shell.execute_reply.started":"2023-04-14T21:05:22.876855Z","shell.execute_reply":"2023-04-14T21:05:33.254028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Concatenates the notype train rows\ntrain_notype = pd.concat([reader(f) for f in tqdm(train_notype_csv_list)])\ntrain_notype = train_notype.reset_index(drop=True)\nprint(train_notype.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-15T00:16:54.611835Z","iopub.execute_input":"2023-04-15T00:16:54.613144Z","iopub.status.idle":"2023-04-15T00:17:12.642360Z","shell.execute_reply.started":"2023-04-15T00:16:54.613100Z","shell.execute_reply":"2023-04-15T00:17:12.641063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Defog","metadata":{}},{"cell_type":"markdown","source":"## stem plot","metadata":{}},{"cell_type":"code","source":"trial_length_defog = pd.DataFrame(defog_df[['Id']].groupby(['Id']).size(),columns = [\"Length\"])\nplot_order = trial_length_defog.rank().astype(int)-1\ntrial_length_defog['Order'] = plot_order\ntrial_length_defog.sort_values('Order', ascending = False, inplace = True)\n\n\nfig, ax1 = plt.subplots()\nax2 = ax1.twinx()\nax1.stem(np.arange(0,91,1),trial_length_defog['Length']/100)\n\nmn, mx = ax1.get_ylim()\nax2.set_ylim(0, mx/60)\n\nax1.set_xlabel('Trial')\nax1.set_ylabel('Seconds', color='g')\nax2.set_ylabel('Minutes', color='orange')\n\nplt.title(\"Length of Defog Trials\")\nplt.savefig(\"defog lengths.png\", dpi=300,bbox_inches='tight')","metadata":{"execution":{"iopub.status.busy":"2023-04-15T00:24:19.756145Z","iopub.execute_input":"2023-04-15T00:24:19.757487Z","iopub.status.idle":"2023-04-15T00:24:22.042301Z","shell.execute_reply.started":"2023-04-15T00:24:19.757442Z","shell.execute_reply":"2023-04-15T00:24:22.040947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## histogram","metadata":{}},{"cell_type":"code","source":"defog_valid = defog_df[['Id','Valid']].groupby(['Id']).mean()\ndefog_valid.head()\n\ndefog_task = defog_df[['Id','Task']].groupby(['Id']).mean()\ndefog_task.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-15T00:26:35.399191Z","iopub.execute_input":"2023-04-15T00:26:35.399607Z","iopub.status.idle":"2023-04-15T00:26:38.736395Z","shell.execute_reply.started":"2023-04-15T00:26:35.399572Z","shell.execute_reply":"2023-04-15T00:26:38.735212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(defog_valid['Valid'],bins = 50)\nplt.xlabel(\"Percent Valid Timestamps\")\nplt.ylabel(\"Frequency\")\nplt.title(\"Distribution of Percentage of Valid Timestamps\")\nplt.savefig(\"valid defog percent.png\", dpi = 300)","metadata":{"execution":{"iopub.status.busy":"2023-04-15T00:27:09.919576Z","iopub.execute_input":"2023-04-15T00:27:09.920399Z","iopub.status.idle":"2023-04-15T00:27:10.991911Z","shell.execute_reply.started":"2023-04-15T00:27:09.920353Z","shell.execute_reply":"2023-04-15T00:27:10.990639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(defog_task['Task'],bins = 50)\nplt.xlabel(\"Percent Task Timestamps\")\nplt.ylabel(\"Frequency\")\nplt.title(\"Distribution of Percentage of Task Timestamps\")\nplt.savefig(\"task defog percent.png\", dpi = 300)","metadata":{"execution":{"iopub.status.busy":"2023-04-15T00:27:36.081847Z","iopub.execute_input":"2023-04-15T00:27:36.082258Z","iopub.status.idle":"2023-04-15T00:27:36.641541Z","shell.execute_reply.started":"2023-04-15T00:27:36.082226Z","shell.execute_reply":"2023-04-15T00:27:36.640036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## heatmap","metadata":{}},{"cell_type":"code","source":"defog_valid_wide = defog_df.pivot(index = \"Id\", columns = \"Time\", values = \"Valid\")","metadata":{"execution":{"iopub.status.busy":"2023-04-15T00:30:51.000646Z","iopub.execute_input":"2023-04-15T00:30:51.001081Z","iopub.status.idle":"2023-04-15T00:31:13.403306Z","shell.execute_reply.started":"2023-04-15T00:30:51.001045Z","shell.execute_reply":"2023-04-15T00:31:13.402189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(defog_valid_ordered.astype(float),cmap=\"crest\")\nplt.xlabel(\"Timestamp\")\nplt.ylabel(\"Trial\")\nplt.title(\"Valid Timestamps: Defog\")\nplt.savefig(\"valid defog.png\",dpi = 300, bbox_inches='tight')","metadata":{"execution":{"iopub.status.busy":"2023-04-15T01:12:40.619577Z","iopub.execute_input":"2023-04-15T01:12:40.620015Z","iopub.status.idle":"2023-04-15T01:14:31.493182Z","shell.execute_reply.started":"2023-04-15T01:12:40.619977Z","shell.execute_reply":"2023-04-15T01:14:31.491879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_task_wide = task_df.pivot(index = \"Id\", columns = \"Time\", values = \"Valid\")","metadata":{},"execution_count":null,"outputs":[]}]}