{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nfrom glob import glob \nimport os\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-03T12:05:04.048041Z","iopub.execute_input":"2023-06-03T12:05:04.048428Z","iopub.status.idle":"2023-06-03T12:05:04.083544Z","shell.execute_reply.started":"2023-06-03T12:05:04.048396Z","shell.execute_reply":"2023-06-03T12:05:04.082420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import the DataFrames","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom os import path\nfrom pathlib import Path\nimport re\n\nroot = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/'\n\ntrain = glob(path.join(root, 'train/**/**'))\ntest = glob(path.join(root, 'test/**/**'))\n\nsubjects = pd.read_csv(path.join(root, 'subjects.csv'))\ntasks = pd.read_csv(path.join(root, 'tasks.csv'))\nevents = pd.read_csv(path.join(root, 'events.csv'))\n\ntdcsfog_metadata = pd.read_csv(path.join(root, 'tdcsfog_metadata.csv'))\ndefog_metadata = pd.read_csv(path.join(root, 'defog_metadata.csv')) \n\ntdcsfog_metadata['Method'] = 'tdcsfog'\ndefog_metadata['Method'] = 'defog'\n\n#tdcsfog_metadata['Module'] = 'tdcsfog'\n#defog_metadata['Module'] = 'defog'\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:05:04.086246Z","iopub.execute_input":"2023-06-03T12:05:04.086650Z","iopub.status.idle":"2023-06-03T12:05:05.006561Z","shell.execute_reply.started":"2023-06-03T12:05:04.086614Z","shell.execute_reply":"2023-06-03T12:05:05.005790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's look at the explanation of DF's\ndef explanation_of_dataframes(df) : \n    print(df.info(verbose=True,show_counts=True))\n    print('\\n')\n    print('\\n')\n    print(df.isna().sum())\n    print('\\n')\n    print('\\n')\n    a = df.describe().style.format('{:.3f}', na_rep=\"\")\\\n         .bar(align=0, vmin=-2.5, vmax=2.5, cmap=\"bwr\", height=50,\n              width=60, props=\"width: 120px; border-right: 1px solid black;\")\\\n         .text_gradient(cmap=\"bwr\", vmin=0, vmax=10)\n    return a","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:05:05.010364Z","iopub.execute_input":"2023-06-03T12:05:05.012639Z","iopub.status.idle":"2023-06-03T12:05:05.021197Z","shell.execute_reply.started":"2023-06-03T12:05:05.012600Z","shell.execute_reply":"2023-06-03T12:05:05.019081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explanation_of_dataframes(subjects)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:05:05.026102Z","iopub.execute_input":"2023-06-03T12:05:05.026612Z","iopub.status.idle":"2023-06-03T12:05:05.226118Z","shell.execute_reply.started":"2023-06-03T12:05:05.026586Z","shell.execute_reply":"2023-06-03T12:05:05.225262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explanation_of_dataframes(tasks)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:05:05.227697Z","iopub.execute_input":"2023-06-03T12:05:05.228256Z","iopub.status.idle":"2023-06-03T12:05:05.267117Z","shell.execute_reply.started":"2023-06-03T12:05:05.228225Z","shell.execute_reply":"2023-06-03T12:05:05.265791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explanation_of_dataframes(events)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:05:05.269082Z","iopub.execute_input":"2023-06-03T12:05:05.269577Z","iopub.status.idle":"2023-06-03T12:05:05.319553Z","shell.execute_reply.started":"2023-06-03T12:05:05.269535Z","shell.execute_reply":"2023-06-03T12:05:05.318176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explanation_of_dataframes(defog_metadata)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:05:05.320969Z","iopub.execute_input":"2023-06-03T12:05:05.321488Z","iopub.status.idle":"2023-06-03T12:05:05.350589Z","shell.execute_reply.started":"2023-06-03T12:05:05.321458Z","shell.execute_reply":"2023-06-03T12:05:05.349191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explanation_of_dataframes(tdcsfog_metadata)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:05:05.352006Z","iopub.execute_input":"2023-06-03T12:05:05.352329Z","iopub.status.idle":"2023-06-03T12:05:05.386225Z","shell.execute_reply.started":"2023-06-03T12:05:05.352300Z","shell.execute_reply":"2023-06-03T12:05:05.385371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_metadata.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:05:05.387199Z","iopub.execute_input":"2023-06-03T12:05:05.388420Z","iopub.status.idle":"2023-06-03T12:05:05.411609Z","shell.execute_reply.started":"2023-06-03T12:05:05.388365Z","shell.execute_reply":"2023-06-03T12:05:05.410104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndefog_metadata.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:05:05.414950Z","iopub.execute_input":"2023-06-03T12:05:05.415919Z","iopub.status.idle":"2023-06-03T12:05:05.428037Z","shell.execute_reply.started":"2023-06-03T12:05:05.415876Z","shell.execute_reply":"2023-06-03T12:05:05.426330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#https://www.kaggle.com/code/arjanso/reducing-dataframe-memory-size-by-65\ndef reduce_mem_usage(props):\n    start_mem_usg = props.memory_usage().sum() / 1024**2 \n    print(\"Memory usage of properties dataframe is :\",start_mem_usg,\" MB\")\n    NAlist = [] # Keeps track of columns that have missing values filled in. \n    for col in props.columns:\n        if props[col].dtype != object:  # Exclude strings\n            \n            # Print current column type\n            print(\"******************************\")\n            print(\"Column: \",col)\n            print(\"dtype before: \",props[col].dtype)\n            \n            # make variables for Int, max and min\n            IsInt = False\n            mx = props[col].max()\n            mn = props[col].min()\n            \n            # Integer does not support NA, therefore, NA needs to be filled\n            if not np.isfinite(props[col]).all(): \n                NAlist.append(col)\n                props[col].fillna(mn-1,inplace=True)  \n                   \n            # test if column can be converted to an integer\n            asint = props[col].fillna(0).astype(np.int64)\n            result = (props[col] - asint)\n            result = result.sum()\n            if result > -0.01 and result < 0.01:\n                IsInt = True\n\n            \n            # Make Integer/unsigned Integer datatypes\n            if IsInt:\n                if mn >= 0:\n                    if mx < 255:\n                        props[col] = props[col].astype(np.uint8)\n                    elif mx < 65535:\n                        props[col] = props[col].astype(np.uint16)\n                    elif mx < 4294967295:\n                        props[col] = props[col].astype(np.uint32)\n                    else:\n                        props[col] = props[col].astype(np.uint64)\n                else:\n                    if mn > np.iinfo(np.int8).min and mx < np.iinfo(np.int8).max:\n                        props[col] = props[col].astype(np.int8)\n                    elif mn > np.iinfo(np.int16).min and mx < np.iinfo(np.int16).max:\n                        props[col] = props[col].astype(np.int16)\n                    elif mn > np.iinfo(np.int32).min and mx < np.iinfo(np.int32).max:\n                        props[col] = props[col].astype(np.int32)\n                    elif mn > np.iinfo(np.int64).min and mx < np.iinfo(np.int64).max:\n                        props[col] = props[col].astype(np.int64)    \n            \n            # Make float datatypes 32 bit\n            else:\n                props[col] = props[col].astype(np.float32)\n            \n            # Print new column type\n            print(\"dtype after: \",props[col].dtype)\n            print(\"******************************\")\n    \n    # Print final result\n    print(\"___MEMORY USAGE AFTER COMPLETION:___\")\n    mem_usg = props.memory_usage().sum() / 1024**2 \n    print(\"Memory usage is: \",mem_usg,\" MB\")\n    print(\"This is \",100*mem_usg/start_mem_usg,\"% of the initial size\")\n    return props, NAlist","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:05:05.429368Z","iopub.execute_input":"2023-06-03T12:05:05.430234Z","iopub.status.idle":"2023-06-03T12:05:05.444089Z","shell.execute_reply.started":"2023-06-03T12:05:05.430195Z","shell.execute_reply":"2023-06-03T12:05:05.442887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_metadata, NAlist = reduce_mem_usage(defog_metadata)\nprint(\"_________________\")\nprint(\"\")\nprint(\"Warning: the following columns have missing values filled with 'df['column_name'].min() -1': \")\nprint(\"_________________\")\nprint(\"\")\nprint(NAlist)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:05:05.445282Z","iopub.execute_input":"2023-06-03T12:05:05.445942Z","iopub.status.idle":"2023-06-03T12:05:05.465432Z","shell.execute_reply.started":"2023-06-03T12:05:05.445907Z","shell.execute_reply":"2023-06-03T12:05:05.463950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_metadata, NAlist = reduce_mem_usage(tdcsfog_metadata)\nprint(\"_________________\")\nprint(\"\")\nprint(\"Warning: the following columns have missing values filled with 'df['column_name'].min() -1': \")\nprint(\"_________________\")\nprint(\"\")\nprint(NAlist)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:05:05.466700Z","iopub.execute_input":"2023-06-03T12:05:05.467595Z","iopub.status.idle":"2023-06-03T12:05:05.479417Z","shell.execute_reply.started":"2023-06-03T12:05:05.467562Z","shell.execute_reply":"2023-06-03T12:05:05.478425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects, NAlist = reduce_mem_usage(subjects)\nprint(\"_________________\")\nprint(\"\")\nprint(\"Warning: the following columns have missing values filled with 'df['column_name'].min() -1': \")\nprint(\"_________________\")\nprint(\"\")\nprint(NAlist)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:05:05.480522Z","iopub.execute_input":"2023-06-03T12:05:05.481477Z","iopub.status.idle":"2023-06-03T12:05:05.498222Z","shell.execute_reply.started":"2023-06-03T12:05:05.481449Z","shell.execute_reply":"2023-06-03T12:05:05.497565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tasks, NAlist = reduce_mem_usage(tasks)\nprint(\"_________________\")\nprint(\"\")\nprint(\"Warning: the following columns have missing values filled with 'df['column_name'].min() -1': \")\nprint(\"_________________\")\nprint(\"\")\nprint(NAlist)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:05:05.499407Z","iopub.execute_input":"2023-06-03T12:05:05.499864Z","iopub.status.idle":"2023-06-03T12:05:05.510644Z","shell.execute_reply.started":"2023-06-03T12:05:05.499841Z","shell.execute_reply":"2023-06-03T12:05:05.509501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_metadata","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:05:05.511823Z","iopub.execute_input":"2023-06-03T12:05:05.512463Z","iopub.status.idle":"2023-06-03T12:05:05.536466Z","shell.execute_reply.started":"2023-06-03T12:05:05.512438Z","shell.execute_reply":"2023-06-03T12:05:05.535682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## DeFOG & tDCSFOG to Single DataFrame","metadata":{}},{"cell_type":"code","source":"from tqdm import tqdm\nfrom tqdm.auto import trange\nimport gc\ndef complete_data(file_name='defog'):\n    df_genel = pd.DataFrame()\n\n    for filename in tqdm(glob(root + f'train/{file_name}/**')):\n\n        file_id = re.split(r\"/|\\.\", filename)[-2]\n        df_iteration = pd.read_csv(filename)\n        df_iteration['Id'] = file_id\n\n        if file_name == 'defog':\n            df = defog_metadata.copy()\n        else:\n            df = tdcsfog_metadata.copy()\n        \n\n        df_iteration['Visit'] = df.loc[df['Id'] == file_id, 'Visit'].to_list()[0]\n        df_iteration['Medication'] = df.loc[df['Id'] == file_id, 'Medication'].to_list()[0]\n        df_iteration['Subject'] = df.loc[df['Id'] == file_id, 'Subject'].to_list()[0]\n        df_iteration['Method'] = df.loc[df['Id'] == file_id, 'Method'].to_list()[0]\n\n        if file_name != 'defog':\n            df_iteration['Test'] = df.loc[df['Id'] == file_id, 'Test'].to_list()[0]\n\n\n        df_genel = pd.concat([df_genel, df_iteration]).reset_index(drop=True)\n        del df\n        gc.collect()\n\n    return df_genel","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:05:05.537874Z","iopub.execute_input":"2023-06-03T12:05:05.538350Z","iopub.status.idle":"2023-06-03T12:05:05.551564Z","shell.execute_reply.started":"2023-06-03T12:05:05.538324Z","shell.execute_reply":"2023-06-03T12:05:05.550114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_metadata = complete_data('defog')","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:05:05.553243Z","iopub.execute_input":"2023-06-03T12:05:05.553886Z","iopub.status.idle":"2023-06-03T12:08:10.252832Z","shell.execute_reply.started":"2023-06-03T12:05:05.553848Z","shell.execute_reply":"2023-06-03T12:08:10.250898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_metadata","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:08:10.254593Z","iopub.execute_input":"2023-06-03T12:08:10.254975Z","iopub.status.idle":"2023-06-03T12:08:10.278689Z","shell.execute_reply.started":"2023-06-03T12:08:10.254947Z","shell.execute_reply":"2023-06-03T12:08:10.277172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_metadata = complete_data('tdcsfog')\ntdcsfog_metadata","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:08:10.280557Z","iopub.execute_input":"2023-06-03T12:08:10.280988Z","iopub.status.idle":"2023-06-03T12:22:46.538100Z","shell.execute_reply.started":"2023-06-03T12:08:10.280955Z","shell.execute_reply":"2023-06-03T12:22:46.537120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I want to merge other files but it increases to 30M parameters, so i try to reduce size\n# garbage collector for gain memory usage \n#defog_sub = defog_metadata.merge(subjects, on='Subject')\n#del defog_metadata\n#gc.collect()\n#defog_sub.drop('Visit_y', axis = 1, inplace = True)\n#defog_sub.rename(columns = {'Visit_x':'Visit'}, inplace = True)\n#defog_metadata = defog_metadata.merge(tasks, on='Id')\n#defog_metadata = defog_metadata.merge(events, on='Id')\n\n#tdcsfog_sub = tdcsfog_metadata.merge(subjects, on='Subject')\n#del tdcsfog_metadata\n#gc.collect()\n#tdcsfog_sub.drop('Visit_y', axis = 1, inplace = True)\n#tdcsfog_sub.rename(columns = {'Visit_x':'Visit'}, inplace = True)\n#tdcsfog_metadata = tdcsfog_metadata.merge(tasks, on='Id')","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:22:46.539192Z","iopub.execute_input":"2023-06-03T12:22:46.539473Z","iopub.status.idle":"2023-06-03T12:22:46.545481Z","shell.execute_reply.started":"2023-06-03T12:22:46.539447Z","shell.execute_reply":"2023-06-03T12:22:46.543925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# when i look at the df's, i saw a relations between Task/Valid and fog files but i won't use them\ndefog_metadata.drop(['Valid','Task'], axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:22:46.547013Z","iopub.execute_input":"2023-06-03T12:22:46.547316Z","iopub.status.idle":"2023-06-03T12:22:47.376339Z","shell.execute_reply.started":"2023-06-03T12:22:46.547291Z","shell.execute_reply":"2023-06-03T12:22:47.375284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_metadata.drop('Test', axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:22:47.377740Z","iopub.execute_input":"2023-06-03T12:22:47.378062Z","iopub.status.idle":"2023-06-03T12:22:47.805350Z","shell.execute_reply.started":"2023-06-03T12:22:47.378036Z","shell.execute_reply":"2023-06-03T12:22:47.804197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Concatenate two files in one DF but it4s very high dimension maybe someone can use Polar python for fast solving\nfog = pd.concat([defog_metadata,tdcsfog_metadata])\n\nfog\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:22:47.806783Z","iopub.execute_input":"2023-06-03T12:22:47.807193Z","iopub.status.idle":"2023-06-03T12:22:49.722215Z","shell.execute_reply.started":"2023-06-03T12:22:47.807158Z","shell.execute_reply":"2023-06-03T12:22:49.720921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fog.Subject.value_counts() # just control","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:22:49.723759Z","iopub.execute_input":"2023-06-03T12:22:49.724201Z","iopub.status.idle":"2023-06-03T12:22:50.438547Z","shell.execute_reply.started":"2023-06-03T12:22:49.724161Z","shell.execute_reply":"2023-06-03T12:22:50.436280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I dropped non-parametric values\nfog.drop(['Id', 'Subject','Method'], axis = 1, inplace = True)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:22:50.440298Z","iopub.execute_input":"2023-06-03T12:22:50.440665Z","iopub.status.idle":"2023-06-03T12:22:51.220764Z","shell.execute_reply.started":"2023-06-03T12:22:50.440635Z","shell.execute_reply":"2023-06-03T12:22:51.219254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Categorical encoder for Sex and Medication but i won use them my model just for the EDA\n\nfog['Medication'] = fog['Medication'].replace({'on' : 1, 'off' : 0})\n","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:22:51.222123Z","iopub.execute_input":"2023-06-03T12:22:51.222432Z","iopub.status.idle":"2023-06-03T12:22:56.991057Z","shell.execute_reply.started":"2023-06-03T12:22:51.222406Z","shell.execute_reply":"2023-06-03T12:22:56.989635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fog_hesitation = fog[['Time', 'AccV', 'AccML', 'AccAP', 'StartHesitation', \n                       'Visit', 'Medication']]\nfog_turn= fog[['Time', 'AccV', 'AccML', 'AccAP', 'Turn',\n               'Visit', 'Medication' ]]\nfog_walking = fog[['Time', 'AccV', 'AccML', 'AccAP', 'Walking',\n               'Visit', 'Medication']]\n\ndel fog\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:22:56.996775Z","iopub.execute_input":"2023-06-03T12:22:56.997137Z","iopub.status.idle":"2023-06-03T12:22:58.140807Z","shell.execute_reply.started":"2023-06-03T12:22:56.997109Z","shell.execute_reply":"2023-06-03T12:22:58.139878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking nb of not null values in the columns\n\ndef not_null(df):\n    \n    # Get the name of the DataFrame\n    def get_dataframe_name(df):\n        for name, obj in globals().items():  # Use locals() for local namespace\n            if obj is df:\n                return name\n        return None\n    dataframe_name = get_dataframe_name(df)\n    \n\n    nb_null_ = pd.DataFrame((df.isna()).sum(axis =0), columns=['nb'])\n    nb_null_.sort_values(by=['nb'], axis=0, ascending=True, inplace=True)\n    \n    return nb_null_.T.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:22:58.142080Z","iopub.execute_input":"2023-06-03T12:22:58.143061Z","iopub.status.idle":"2023-06-03T12:22:58.150383Z","shell.execute_reply.started":"2023-06-03T12:22:58.143007Z","shell.execute_reply":"2023-06-03T12:22:58.148973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"not_null(fog_hesitation)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:22:58.152001Z","iopub.execute_input":"2023-06-03T12:22:58.152623Z","iopub.status.idle":"2023-06-03T12:22:58.292028Z","shell.execute_reply.started":"2023-06-03T12:22:58.152598Z","shell.execute_reply":"2023-06-03T12:22:58.291065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\ndef plot_hist(df, cols):\n    num_plots = len(cols)\n    num_cols = math.ceil(np.sqrt(num_plots))\n    num_rows = math.ceil(num_plots/num_cols)\n\n    fig, axs = plt.subplots(num_rows, num_cols, figsize=(22,16))\n\n    for ind, col in enumerate(cols):\n        i = math.floor(ind/num_cols)\n        j = ind - i*num_cols\n\n        if num_rows == 1:\n                sns.histplot(df[col], kde=True, ax=axs[j],bins=20,color='b', edgecolor='k', alpha=0.5)\n                axs[j].set_title('Distribution of ' + col)\n\n        else:\n            sns.histplot(df[col], kde=True, ax=axs[i, j],bins=20,color='b', edgecolor='k', alpha=0.5)\n            axs[i, j].set_title('Distribution of ' + col)\n        ","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:22:58.293456Z","iopub.execute_input":"2023-06-03T12:22:58.293874Z","iopub.status.idle":"2023-06-03T12:22:58.303313Z","shell.execute_reply.started":"2023-06-03T12:22:58.293843Z","shell.execute_reply":"2023-06-03T12:22:58.301746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# distribution plot\nfog_sample = fog_hesitation.sample(frac = 0.1)\nplot_hist(fog_sample, ['AccV','AccML','AccAP'])\n","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:22:58.304640Z","iopub.execute_input":"2023-06-03T12:22:58.305043Z","iopub.status.idle":"2023-06-03T12:23:30.680240Z","shell.execute_reply.started":"2023-06-03T12:22:58.305010Z","shell.execute_reply":"2023-06-03T12:23:30.678935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# MULTIVARIATE ANALYSIS","metadata":{}},{"cell_type":"code","source":"def description_var(df, col):\n    colData=df[col]\n    moyenne=np.mean(colData)\n    mediane = np.median(colData)\n    Q1 = np.percentile(colData, 25)\n    Q3 = np.percentile(colData, 75)\n    max = colData.max()\n    min = colData.min()\n    variance = np.var(colData)\n    sd = np.std(colData)\n    skew=pd.DataFrame(colData).skew()[0]\n    kurt=pd.DataFrame(colData).kurtosis()[0]\n\n    print(\"Measures statistiques of variable {}\" .format(col))\n    print(\" \")\n    print(\"Mean of variable {} is : {} \".format(col,round(moyenne, 2)))\n    print(\"MEdian of variable {} is : {} \".format(col,round(mediane, 2)))\n    print(\"Quartile Q1 value : {} \".format(round(Q1, 2)))\n    print(\"Quartile Q3 value : {} \".format(round(Q3, 2)))\n    print(\"Maximum Value : {} \".format(max))\n    print(\"Minimum Value : {} \".format(min))\n    print(\" \")\n    print(\" \")\n    print(\"Distribution of variable {}\" .format(col))\n    print(\" \")\n    print(\"Variance of variable {} is : {} \" .format(col,round(variance, 2)))\n    print(\"Eta-squared of variable {} is : {} \" .format(col,round(sd, 2)))\n    print(\"Eta-squared interquartile of variable {} is : {} \" .format(col,round(Q3-Q1, 2)))\n    print(\" \")\n    print(\" \")\n    print(\"Measures de forme for variable {}\" .format(col))\n    print(\" \")\n    print(\"Skewewness empirique of variable {} is{} \" .format(col,round(skew, 4)))\n    if (skew==0):\n        print(\"Distribution of variable {} is symétrique.\" .format(col))\n    elif (skew>0):\n        print(\"Distribution of variable {} is skewed to the right.\" .format(col))\n    else:\n        print(\"Distribution of variable {} is skewed to the left.\" .format(col))\n    print(\" \")\n    print(\"Kurtosis  for variable {} is {} \" .format(col, round(kurt, 4)))\n    if kurt==0:\n        print(\"Distribution of variable {} has the same flattening as the normal distribution\" .format(col))\n    elif kurt>0:\n        print(\"Distribution of variable {} is less flat than the normal distribution, observations are more concentrated..\" .format(col))\n    else:\n        print(\"The distribution of the variable {} is more flattened than the normal distribution, the observations are less concentrated.\" .format(col))\n    print(\" \")\n    plt.figure(figsize=(10,8))\n    df[col].hist(color = '#ffb2ff', edgecolor = '#7200ca', log = True )\n    plt.title(\"Representation statistique of variable {}\".format(col))\n    plt.xlabel(\"{}\".format(col))\n    plt.ylabel(\"Log Count\")\n    plt.show()\n    print(\" \")\n    print(\"Boxplot of variable {}\".format(col))\n    plt.figure(figsize=(10,8))\n    df.boxplot(column=[col], return_type='axes', vert=True, showfliers=False, showcaps=True, patch_artist=True, color='tan', medianprops={'linestyle': '-', 'linewidth': 2, 'color': 'red'}, whiskerprops={'linestyle': '-', 'linewidth': 2, 'color' : 'blue'}, capprops={'linestyle': '-', 'linewidth': 2, 'color':'blue'})\n    plt.ylabel(\"Values of {}\".format(col))\n    plt.show()\n    del skew\n    del kurt","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:23:30.681785Z","iopub.execute_input":"2023-06-03T12:23:30.682105Z","iopub.status.idle":"2023-06-03T12:23:30.703312Z","shell.execute_reply.started":"2023-06-03T12:23:30.682077Z","shell.execute_reply":"2023-06-03T12:23:30.702516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"description_var(fog_sample, 'AccV')","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:23:30.704110Z","iopub.execute_input":"2023-06-03T12:23:30.704372Z","iopub.status.idle":"2023-06-03T12:23:35.387552Z","shell.execute_reply.started":"2023-06-03T12:23:30.704349Z","shell.execute_reply":"2023-06-03T12:23:35.386566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"description_var(fog_sample, 'AccML')","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:23:35.388898Z","iopub.execute_input":"2023-06-03T12:23:35.389823Z","iopub.status.idle":"2023-06-03T12:23:40.075201Z","shell.execute_reply.started":"2023-06-03T12:23:35.389783Z","shell.execute_reply":"2023-06-03T12:23:40.073540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"description_var(fog_sample, 'AccAP')","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:23:40.076750Z","iopub.execute_input":"2023-06-03T12:23:40.077123Z","iopub.status.idle":"2023-06-03T12:23:44.782043Z","shell.execute_reply.started":"2023-06-03T12:23:40.077092Z","shell.execute_reply":"2023-06-03T12:23:44.781061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 8))\nmask = np.triu(np.ones_like(fog_sample.corr(), dtype=np.bool_))\nheatmap = sns.heatmap(fog_sample.corr(method = 'pearson'), mask=mask, vmin=-1, vmax=1, annot=True, cmap='BrBG')\nheatmap.set_title('Relation between the variables', fontdict={'fontsize':18}, pad=16);","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:23:44.783147Z","iopub.execute_input":"2023-06-03T12:23:44.783410Z","iopub.status.idle":"2023-06-03T12:23:45.894318Z","shell.execute_reply.started":"2023-06-03T12:23:44.783387Z","shell.execute_reply":"2023-06-03T12:23:45.893021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There is visible relation between Visit and AccaP/AccV ","metadata":{}},{"cell_type":"code","source":"#plt.figure(figsize=(20,12)) \n#sns.pairplot(fog_sample,hue = 'Sex',palette=\"husl\",size=5)\n#plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:23:45.895884Z","iopub.execute_input":"2023-06-03T12:23:45.896263Z","iopub.status.idle":"2023-06-03T12:23:45.901303Z","shell.execute_reply.started":"2023-06-03T12:23:45.896234Z","shell.execute_reply":"2023-06-03T12:23:45.899799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# HYPOTHESIS TEST, F-TEST, ETA_SQUARED","metadata":{}},{"cell_type":"markdown","source":"\n- H₀: μ₁= μ₂ = μ₃ = … = μ<sub>9</sub>\n- H₁: Means of the variable ... aren't same\n- α = 0.05\nSelon F test statistique:","metadata":{}},{"cell_type":"code","source":"def hypothesis_test(df, col):\n    data = [['Var inter gr', '', '', '', '', '', '',''], ['Var intra gr', '', '', '', '', '', '',''], ['Total', '', '', '', '', '', '','']]\n    anova_table = pd.DataFrame(data, columns = ['Source of Variation', 'SS', 'df', 'MS', 'F', 'P-value', 'F crit', 'Eta-squared'])\n    anova_table.set_index('Source of Variation', inplace = True)\n    print('\\n')\n    print(col + ' Hypothesis Test')\n\n    # calculate SSTR and update anova table\n    x_bar = df[col].mean()\n    SSTR = df.groupby('Visit').count() * (df.groupby('Visit').mean() - x_bar)**2\n    anova_table['SS']['Var inter gr'] = SSTR[col].sum()\n\n    # calculate SSE and update anova table\n    SSE = (df.groupby('Visit').count() - 1) * df.groupby('Visit').std()**2\n    anova_table['SS']['Var intra gr'] = SSE[col].sum()\n\n    # calculate SSTR and update anova table\n    SSTR = SSTR[col].sum() + SSE[col].sum()\n    anova_table['SS']['Total'] = SSTR\n    # update degree of freedom\n    anova_table['df']['Var inter gr'] = df['Visit'].nunique() - 1\n    anova_table['df']['Var intra gr'] = df.shape[0] - df['Visit'].nunique()\n    anova_table['df']['Total'] = df.shape[0] - 1\n    # calculate MS\n    anova_table['MS'] = anova_table['SS'] / anova_table['df']\n\n    # calculate F\n    F = anova_table['MS']['Var inter gr'] / anova_table['MS']['Var intra gr']\n    anova_table['F']['Var inter gr'] = F\n    eta_carré = anova_table['SS']['Var inter gr'] / anova_table['SS']['Total']\n    anova_table['Eta-squared']['Var inter gr'] = eta_carré\n    # p-value\n    anova_table['P-value']['Var inter gr'] = 1 - stats.f.cdf(F, anova_table['df']['Var inter gr'], anova_table['df']['Var intra gr'])\n    # F critical\n    alpha = 0.05\n    # possible types \"right-tailed, left-tailed, two-tailed\"\n    latéral_hypothesis_type = \"bilatéral\"\n    if latéral_hypothesis_type == \"bilatéral\":\n        alpha /= 2\n    anova_table['F crit']['Var inter gr'] = stats.f.ppf(1-alpha, anova_table['df']['Var inter gr'], anova_table['df']['Var intra gr'])\n\n    print(\"The Critical Value Approach for Hypothesis Testing in the Decision Rule\")\n    conclusion = \"You cannot reject the null hypothesis.\"\n    if anova_table['F']['Var inter gr'] > anova_table['F crit']['Var inter gr']:\n        conclusion = \"Rejecte null hypothes.\"\n    print(\"F-score is:\", anova_table['F']['Var inter gr'], \" and critical value is:\", anova_table['F crit']['Var inter gr'])\n    print(\"P-value is:\", anova_table['P-value']['Var inter gr'])\n    print(\"Eta-squared score is:\", anova_table['Eta-squared']['Var inter gr'])\n    ''' L\\’Eta-squared therefore represents the proportion of variance of the dependent variable (the tested variable) explained by the independent variable (HSR_rating). This index varies between 0 and 1 and the following markers were developed by Cohen (1988) to guide its interpretation.'''\n    if eta_carré < 0.02 :\n        print('Eta-squared has the effect of small size')\n    elif 0.14 > eta_carré > 0.06 :\n        print('Eta-squared has the effect of mean size')\n    elif eta_carré > 0.14 :\n        print('Eta-squared has the effect of big size')\n    print(conclusion)\n\n    print('\\n')\n    print(anova_table)\n    print(\"\\n---------------\")","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:23:45.902892Z","iopub.execute_input":"2023-06-03T12:23:45.903283Z","iopub.status.idle":"2023-06-03T12:23:45.920451Z","shell.execute_reply.started":"2023-06-03T12:23:45.903248Z","shell.execute_reply":"2023-06-03T12:23:45.919558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy import stats\nvariables = ['Visit','AccV','AccML','AccAP']\nvar2 = ['AccV','AccML','AccAP']\ndf_hyp = fog_sample[variables]\n\nfor i in var2  :\n    hypothesis_test(df_hyp,i)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:23:45.922268Z","iopub.execute_input":"2023-06-03T12:23:45.923751Z","iopub.status.idle":"2023-06-03T12:23:46.521425Z","shell.execute_reply.started":"2023-06-03T12:23:45.923651Z","shell.execute_reply":"2023-06-03T12:23:46.520275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_hyp\ngc.collect()\n\n\nfog_hes = fog_hesitation[['AccV','AccML','AccAP','StartHesitation']]\nfog_tu = fog_turn[['AccV','AccML','AccAP','Turn']]\nfog_walk = fog_walking[['AccV','AccML','AccAP', 'Walking']]\ndel fog_hesitation\ngc.collect()\ndel fog_turn\ngc.collect()\ndel fog_walking\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:23:46.523279Z","iopub.execute_input":"2023-06-03T12:23:46.523894Z","iopub.status.idle":"2023-06-03T12:23:47.627335Z","shell.execute_reply.started":"2023-06-03T12:23:46.523863Z","shell.execute_reply":"2023-06-03T12:23:47.625993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### There were a PCA Test analysis but becaus of the memory, i cleaned it","metadata":{}},{"cell_type":"markdown","source":"# TRAIN - TEST PHASE","metadata":{}},{"cell_type":"markdown","source":"### for fog_hesitation","metadata":{}},{"cell_type":"code","source":"import optuna\nimport xgboost as xgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import average_precision_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import StandardScaler\n","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:23:47.628739Z","iopub.execute_input":"2023-06-03T12:23:47.629140Z","iopub.status.idle":"2023-06-03T12:23:48.489187Z","shell.execute_reply.started":"2023-06-03T12:23:47.629105Z","shell.execute_reply":"2023-06-03T12:23:48.488079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The objective function is passed an Optuna specific argument of trial\nN_SPLITS = 5\nN_REPEATS = 2\nN_TRIALS = 25\nRS = 0\n\ndef objective(trial,\n                X,\n                y,\n                random_state=0,\n                n_splits=N_SPLITS ,\n                n_repeats=N_REPEATS,\n                n_jobs=1,\n                early_stopping_rounds=10,\n            ):\n    \n    \n# params specifies the XGBoost hyperparameters to be tuned\n    params = {\n        'objective': 'binary:logistic',\n        'eval_metric': 'logloss',\n        'booster': trial.suggest_categorical('booster', ['gbtree', 'dart']),\n        'max_depth': trial.suggest_int('max_depth', 5, 50),\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.1),\n        'subsample': trial.suggest_float('subsample', 0.50, 1),\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.50, 1),\n        'gamma': trial.suggest_int('gamma', 0, 10),\n        'objective': 'binary:logistic'\n    }  \n    # Perform stratified K-fold cross-validation\n    skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n    mean_precision = []\n    for train_index, test_index in tqdm(skf.split(X, y)):\n        X_train, X_test = X.iloc[train_index], X.iloc[test_index]\n        y_train, y_test = y.iloc[train_index], y.iloc[test_index]\n\n        # Create the DMatrix for XGBoost\n        dtrain = xgb.DMatrix(X_train, label=y_train)\n        dtest = xgb.DMatrix(X_test)\n\n        # Train the model with the current hyperparameters\n        model = xgb.train(params, dtrain)\n\n        # Predict on the test set\n        y_pred = model.predict(dtest)\n\n        # Convert probabilities to binary predictions\n        y_pred_binary = [1 if pred >= 0.5 else 0 for pred in y_pred]\n\n        # Convert probabilities to binary predictions\n        y_pred_binary = [1 if pred >= 0.5 else 0 for pred in y_pred]\n\n        # Calculate accuracy for this fold\n        mAP = average_precision_score(y_test, y_pred_binary, average='macro')\n        mean_precision.append(mAP)\n\n    # Calculate the mean accuracy across folds\n    mean_of_precision = sum(mean_precision) / len(mean_precision)\n  \n    print(\"Mean Average Precision (mAP):\", mean_of_precision)\n    \n    return mean_of_precision ","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:23:48.490408Z","iopub.execute_input":"2023-06-03T12:23:48.490688Z","iopub.status.idle":"2023-06-03T12:23:48.501877Z","shell.execute_reply.started":"2023-06-03T12:23:48.490664Z","shell.execute_reply":"2023-06-03T12:23:48.500561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prep_data(X,y):\n    a=[]\n    study = optuna.create_study(direction=\"maximize\",pruner=optuna.pruners.MedianPruner())\n\n    study.optimize(lambda trial :objective(trial,X, y,\n                    random_state=RS,\n                    n_splits=N_SPLITS,\n                    n_repeats=N_REPEATS,\n                    n_jobs=-1,\n                ),\n                n_trials=N_TRIALS,\n                n_jobs=-1,\n            )\n\n    print(\"Number of finished trials: \", len(study.trials))\n    print(\"Best trial:\")\n    best_par = study.best_params\n    trial = study.best_trial\n\n    print(\"  Value: {}\".format(trial.value))\n    print(\"  Params: \")\n    for key, value in trial.params.items():\n        print(\"    {}: {}\".format(key, value))\n        \n    best_params = trial.params\n    best_params['objective'] = 'binary:logistic'\n\n    \n    return best_par\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:23:48.503137Z","iopub.execute_input":"2023-06-03T12:23:48.503411Z","iopub.status.idle":"2023-06-03T12:23:48.519053Z","shell.execute_reply.started":"2023-06-03T12:23:48.503386Z","shell.execute_reply":"2023-06-03T12:23:48.518159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def final_class(df, col ):\n    # save the labels to a Pandas series target\n    y = df[col]\n    # Drop the label feature\n    X = df.drop(col,axis=1)\n    #Apply StandardScaler to the input features\n    scaler = StandardScaler()\n    X_scaled = scaler.fit_transform(X)\n    # Convert the scaled data back to pandas DataFrame\n    X = pd.DataFrame(X_scaled, columns=X.columns)\n    best_param=prep_data(X,y)\n    return best_param","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:23:48.520280Z","iopub.execute_input":"2023-06-03T12:23:48.520505Z","iopub.status.idle":"2023-06-03T12:23:48.529246Z","shell.execute_reply.started":"2023-06-03T12:23:48.520484Z","shell.execute_reply":"2023-06-03T12:23:48.528542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nbest_params = final_class(fog_hes, 'StartHesitation' )","metadata":{"execution":{"iopub.status.busy":"2023-06-03T12:23:48.530137Z","iopub.execute_input":"2023-06-03T12:23:48.530479Z","iopub.status.idle":"2023-06-03T13:37:13.362185Z","shell.execute_reply.started":"2023-06-03T12:23:48.530459Z","shell.execute_reply":"2023-06-03T13:37:13.358766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ny = fog_hes['StartHesitation'] \n# Drop the label feature\nX = fog_hes.drop('StartHesitation',axis=1)\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=0)\n\noptimal_hesitation = xgb.XGBClassifier(**best_params)\noptimal_hesitation.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T13:37:13.367330Z","iopub.execute_input":"2023-06-03T13:37:13.368102Z","iopub.status.idle":"2023-06-03T13:44:32.679703Z","shell.execute_reply.started":"2023-06-03T13:37:13.368036Z","shell.execute_reply":"2023-06-03T13:44:32.678779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### FOG_Turn","metadata":{}},{"cell_type":"code","source":"\nbest_params_turn= final_class(fog_tu, 'Turn' )","metadata":{"execution":{"iopub.status.busy":"2023-06-03T13:44:32.681250Z","iopub.execute_input":"2023-06-03T13:44:32.681557Z","iopub.status.idle":"2023-06-03T15:18:26.551622Z","shell.execute_reply.started":"2023-06-03T13:44:32.681526Z","shell.execute_reply":"2023-06-03T15:18:26.548643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ny = fog_tu['Turn'] \n# Drop the label feature\nX = fog_tu.drop('Turn',axis=1)\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=0)\n\noptimal_turn = xgb.XGBClassifier(**best_params_turn)\n\noptimal_turn.fit(X_train, y_train)\n    \n","metadata":{"execution":{"iopub.status.busy":"2023-06-03T15:18:26.557445Z","iopub.execute_input":"2023-06-03T15:18:26.557998Z","iopub.status.idle":"2023-06-03T15:29:26.435120Z","shell.execute_reply.started":"2023-06-03T15:18:26.557973Z","shell.execute_reply":"2023-06-03T15:29:26.433744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### FOG_WALKING","metadata":{}},{"cell_type":"code","source":"best_params_walk = final_class(fog_walk, 'Walking' )","metadata":{"execution":{"iopub.status.busy":"2023-06-03T15:29:26.437132Z","iopub.execute_input":"2023-06-03T15:29:26.437414Z","iopub.status.idle":"2023-06-03T17:00:48.197619Z","shell.execute_reply.started":"2023-06-03T15:29:26.437390Z","shell.execute_reply":"2023-06-03T17:00:48.193991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ny = fog_walk['Walking'] \n# Drop the label feature\nX = fog_walk.drop('Walking',axis=1)\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=0)\n\noptimal_walking = xgb.XGBClassifier(**best_params_walk)\noptimal_walking.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T17:00:48.204051Z","iopub.execute_input":"2023-06-03T17:00:48.204619Z","iopub.status.idle":"2023-06-03T17:08:51.907276Z","shell.execute_reply.started":"2023-06-03T17:00:48.204583Z","shell.execute_reply":"2023-06-03T17:08:51.905626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PREPARE TEST DATA","metadata":{}},{"cell_type":"code","source":"def prepare_test_data (df_type):\n    full_data = pd.DataFrame()\n    subdatas = glob(root + f'test/{df_type}/*')\n    \n    for subdata in subdatas:\n        sub_data = pd.read_csv(subdata)\n        sub_data['Id'] = subdata.split(sep='/')[-1].split(sep='.')[0] + '_' + sub_data['Time'].astype(str)\n        full_data = pd.concat([full_data, sub_data]).reset_index(drop=True)\n            \n    return full_data","metadata":{"execution":{"iopub.status.busy":"2023-06-03T17:08:51.909152Z","iopub.execute_input":"2023-06-03T17:08:51.909457Z","iopub.status.idle":"2023-06-03T17:08:51.917963Z","shell.execute_reply.started":"2023-06-03T17:08:51.909433Z","shell.execute_reply":"2023-06-03T17:08:51.916787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_defog = prepare_test_data('defog')\ntest_defog.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-03T17:08:51.919425Z","iopub.execute_input":"2023-06-03T17:08:51.919784Z","iopub.status.idle":"2023-06-03T17:08:52.443538Z","shell.execute_reply.started":"2023-06-03T17:08:51.919754Z","shell.execute_reply":"2023-06-03T17:08:52.441134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tdcsfog = prepare_test_data('tdcsfog')\ntest_tdcsfog.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-03T17:08:52.445280Z","iopub.execute_input":"2023-06-03T17:08:52.445669Z","iopub.status.idle":"2023-06-03T17:08:52.474394Z","shell.execute_reply.started":"2023-06-03T17:08:52.445637Z","shell.execute_reply":"2023-06-03T17:08:52.472790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data for StartHesitation/Turn models\ntest_defog_short = test_defog.drop(['Time', 'Id'], axis=1)\n\n# defog_StartHesitation\ndefog_StartHesitation_test = optimal_hesitation.predict(test_defog_short)\n\n# defog_Turn\ndefog_Turn_test = optimal_turn.predict(test_defog_short)\n\n# defog_Walking\ndefog_Walking_test = optimal_walking.predict(test_defog_short)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-03T17:08:52.476244Z","iopub.execute_input":"2023-06-03T17:08:52.476793Z","iopub.status.idle":"2023-06-03T17:08:53.461728Z","shell.execute_reply.started":"2023-06-03T17:08:52.476748Z","shell.execute_reply":"2023-06-03T17:08:53.460922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# defog table\n\ndefog_StartHesitation_test = pd.Series(defog_StartHesitation_test, name='StartHesitation')\ndefog_Turn_test = pd.Series(defog_Turn_test, name='Turn')\ndefog_Walking_test = pd.Series(defog_Walking_test, name='Walking')\n\ntest_defog_predicted = pd.concat([defog_StartHesitation_test, \n                                  defog_Turn_test, \n                                  defog_Walking_test], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T17:08:53.464303Z","iopub.execute_input":"2023-06-03T17:08:53.464657Z","iopub.status.idle":"2023-06-03T17:08:53.474815Z","shell.execute_reply.started":"2023-06-03T17:08:53.464626Z","shell.execute_reply":"2023-06-03T17:08:53.473617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data for StartHesitation/Turn models\ntest_tdcsfog_pred = test_tdcsfog.drop('Id', axis=1)\ntest_tdcsfog_short = test_tdcsfog.drop(['Time', 'Id'], axis=1)\n\n# defog_StartHesitation\ntdcsfog_StartHesitation_test = optimal_hesitation.predict(test_tdcsfog_short)\n\n# defog_Turn\ntdcsfog_Turn_test = optimal_turn.predict(test_tdcsfog_short)\n\n# defog_Walking\ntdcsfog_Walking_test = optimal_walking.predict(test_tdcsfog_short)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T17:08:53.476510Z","iopub.execute_input":"2023-06-03T17:08:53.477248Z","iopub.status.idle":"2023-06-03T17:08:53.513783Z","shell.execute_reply.started":"2023-06-03T17:08:53.477215Z","shell.execute_reply":"2023-06-03T17:08:53.512643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tdcsfog table\n\ntdcsfog_StartHesitation_test = pd.Series(tdcsfog_StartHesitation_test, name='StartHesitation')\ntdcsfog_Turn_test = pd.Series(tdcsfog_Turn_test, name='Turn')\ntdcsfog_Walking_test = pd.Series(tdcsfog_Walking_test, name='Walking')\n\ntest_tdcsfog_predicted = pd.concat([tdcsfog_StartHesitation_test, \n                                  tdcsfog_Turn_test, \n                                  tdcsfog_Walking_test], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T17:08:53.515408Z","iopub.execute_input":"2023-06-03T17:08:53.516167Z","iopub.status.idle":"2023-06-03T17:08:53.525217Z","shell.execute_reply.started":"2023-06-03T17:08:53.516126Z","shell.execute_reply":"2023-06-03T17:08:53.523963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_data = (pd.concat(\n    [test_tdcsfog_predicted, \n     test_defog_predicted])\n              .reset_index(drop=True))\n\ntotal_data.head(),\\\ntotal_data.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-03T17:08:53.527335Z","iopub.execute_input":"2023-06-03T17:08:53.527781Z","iopub.status.idle":"2023-06-03T17:08:53.548419Z","shell.execute_reply.started":"2023-06-03T17:08:53.527739Z","shell.execute_reply":"2023-06-03T17:08:53.547454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def id_for_submission(tdcsfog, defog):\n    df_tdcsfog = tdcsfog\n    df_defog = defog\n    \n    df = pd.concat([df_tdcsfog,\n                    df_defog]).reset_index(drop=True)\n    \n    submission_id = df.drop(['Time',\n                         'AccV', \n                         'AccML', \n                         'AccAP'], axis=1)\n    \n    return submission_id","metadata":{"execution":{"iopub.status.busy":"2023-06-03T17:08:53.549587Z","iopub.execute_input":"2023-06-03T17:08:53.550581Z","iopub.status.idle":"2023-06-03T17:08:53.557523Z","shell.execute_reply.started":"2023-06-03T17:08:53.550558Z","shell.execute_reply":"2023-06-03T17:08:53.555363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save predictions in format used for competition scoring\n\nsubmission = id_for_submission(test_tdcsfog, test_defog)\nsubmission = pd.concat([submission, total_data], axis=1)\n\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T17:08:53.559505Z","iopub.execute_input":"2023-06-03T17:08:53.559916Z","iopub.status.idle":"2023-06-03T17:08:54.017178Z","shell.execute_reply.started":"2023-06-03T17:08:53.559884Z","shell.execute_reply":"2023-06-03T17:08:54.015156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2023-06-03T17:08:54.020124Z","iopub.execute_input":"2023-06-03T17:08:54.020655Z","iopub.status.idle":"2023-06-03T17:08:54.035311Z","shell.execute_reply.started":"2023-06-03T17:08:54.020594Z","shell.execute_reply":"2023-06-03T17:08:54.034469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}