{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":41880,"databundleVersionId":5677426,"sourceType":"competition"}],"dockerImageVersionId":30699,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-09T15:08:31.310207Z","iopub.execute_input":"2024-05-09T15:08:31.310971Z","iopub.status.idle":"2024-05-09T15:08:32.506385Z","shell.execute_reply.started":"2024-05-09T15:08:31.31094Z","shell.execute_reply":"2024-05-09T15:08:32.505394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport glob\nimport lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:08:32.508285Z","iopub.execute_input":"2024-05-09T15:08:32.508654Z","iopub.status.idle":"2024-05-09T15:08:37.17608Z","shell.execute_reply.started":"2024-05-09T15:08:32.508631Z","shell.execute_reply":"2024-05-09T15:08:37.175019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\n\nfrom sklearn.impute import SimpleImputer\nfrom imblearn.over_sampling import SMOTE\nfrom imblearn.under_sampling import RandomUnderSampler\nfrom lightgbm import LGBMClassifier\n\nfrom sklearn.cluster import KMeans\nfrom sklearn.feature_selection import mutual_info_classif\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import GridSearchCV\n\nfrom sklearn.metrics import f1_score, accuracy_score, precision_score, recall_score\nfrom sklearn.metrics import confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:08:37.177421Z","iopub.execute_input":"2024-05-09T15:08:37.177861Z","iopub.status.idle":"2024-05-09T15:08:37.50803Z","shell.execute_reply.started":"2024-05-09T15:08:37.177827Z","shell.execute_reply":"2024-05-09T15:08:37.507276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/'\n# Read the dataset\n\ndefog_metadata_file = f'{data_dir}defog_metadata.csv'\ntdcsfog_metadata_file = f'{data_dir}tdcsfog_metadata.csv'\nevents_data_file = f'{data_dir}events.csv'\nsubjects_data_file = f'{data_dir}subjects.csv'\ntasks_data_file = f'{data_dir}tasks.csv'\n\n# Read the meta data\n\ndefog_metadata = pd.read_csv(defog_metadata_file)\ntdcsfog_metadata = pd.read_csv(tdcsfog_metadata_file)\nfull_metadata = pd.concat([tdcsfog_metadata, defog_metadata])\n\nevents_data = pd.read_csv(events_data_file)\nsubjects_data = pd.read_csv(subjects_data_file)\ntasks_data = pd.read_csv(tasks_data_file)","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:08:37.509994Z","iopub.execute_input":"2024-05-09T15:08:37.510577Z","iopub.status.idle":"2024-05-09T15:08:37.566932Z","shell.execute_reply.started":"2024-05-09T15:08:37.510549Z","shell.execute_reply":"2024-05-09T15:08:37.566095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the data and append the Id\ndef read_data(path):\n    df = pd.read_csv(path)\n    df['Id'] = path.split(\"/\")[-1].split(\".\")[0]\n    \n    return df\n\n\n# Read and concatenate all files of the train data from specified dataset\ndef create_full_data(dataset_name):\n    paths = glob.glob(data_dir + f'train/{dataset_name}/*')\n    final_df = pd.concat([read_data(p) for p in paths])\n    final_df['dataset'] = dataset_name\n    \n    return final_df","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:08:37.568027Z","iopub.execute_input":"2024-05-09T15:08:37.568308Z","iopub.status.idle":"2024-05-09T15:08:37.574526Z","shell.execute_reply.started":"2024-05-09T15:08:37.568285Z","shell.execute_reply":"2024-05-09T15:08:37.573639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a dataframe with all tdcsfog data\ntdcsfog_data = create_full_data('tdcsfog')\ntdcsfog_data","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:08:37.575621Z","iopub.execute_input":"2024-05-09T15:08:37.57588Z","iopub.status.idle":"2024-05-09T15:08:56.40892Z","shell.execute_reply.started":"2024-05-09T15:08:37.575852Z","shell.execute_reply":"2024-05-09T15:08:56.407953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a dataframe with all defog data\ndefog_data = create_full_data('defog')\ndefog_data","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:08:56.410093Z","iopub.execute_input":"2024-05-09T15:08:56.410411Z","iopub.status.idle":"2024-05-09T15:09:24.926284Z","shell.execute_reply.started":"2024-05-09T15:08:56.410387Z","shell.execute_reply":"2024-05-09T15:09:24.925306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_data_valid = defog_data.loc[ \\\n                        (defog_data.Valid == True) & (defog_data.Task == True)].copy()\n\ndefog_data_valid.reset_index(drop=True, inplace=True)\ndefog_data_valid.drop(['Valid', 'Task'], axis=1, inplace=True)\n\ndefog_data_valid","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:09:24.927692Z","iopub.execute_input":"2024-05-09T15:09:24.928063Z","iopub.status.idle":"2024-05-09T15:09:25.91606Z","shell.execute_reply.started":"2024-05-09T15:09:24.928031Z","shell.execute_reply":"2024-05-09T15:09:25.915156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_data = pd.concat([tdcsfog_data, defog_data_valid])\nfull_data","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:09:25.917949Z","iopub.execute_input":"2024-05-09T15:09:25.918772Z","iopub.status.idle":"2024-05-09T15:09:26.471337Z","shell.execute_reply.started":"2024-05-09T15:09:25.918736Z","shell.execute_reply":"2024-05-09T15:09:26.470338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tasks_data['Duration'] = tasks_data['End'] - tasks_data['Begin']\n\ntasks_data = pd.pivot_table(tasks_data, values=['Duration'],\n                                index=['Id'], columns=['Task'],\n                                aggfunc='sum', fill_value=0)\n\ntasks_data.columns = [c[-1] for c in tasks_data.columns]\ntasks_data = tasks_data.reset_index()\n\ntasks_data['t_kmeans'] = KMeans(n_clusters=10, random_state=42, n_init=10) \\\n                    .fit_predict(tasks_data[tasks_data.columns[1:]])\n\n\nsubjects_data = subjects_data.fillna(0).groupby('Subject') \\\n    [['Visit', 'Age', 'YearsSinceDx', 'UPDRSIII_On', 'UPDRSIII_Off', 'NFOGQ']].median()\nsubjects_data = subjects_data.reset_index()\n\nsubjects_data['s_kmeans'] = KMeans(n_clusters=10, random_state=42, n_init=10) \\\n                    .fit_predict(subjects_data[subjects_data.columns[1:]])","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:09:26.475993Z","iopub.execute_input":"2024-05-09T15:09:26.476373Z","iopub.status.idle":"2024-05-09T15:09:26.62671Z","shell.execute_reply.started":"2024-05-09T15:09:26.476347Z","shell.execute_reply":"2024-05-09T15:09:26.625853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_data = full_data.merge(full_metadata, on='Id', how='inner') \\\n                        .merge(tasks_data[['Id', 't_kmeans']], how='left', on='Id').fillna(-1) \\\n                        .merge(subjects_data.drop('Visit', axis=1),\n                                       on='Subject', how='left').fillna(-1)\n\nfull_data","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:09:26.628274Z","iopub.execute_input":"2024-05-09T15:09:26.628948Z","iopub.status.idle":"2024-05-09T15:09:52.130602Z","shell.execute_reply.started":"2024-05-09T15:09:26.628915Z","shell.execute_reply":"2024-05-09T15:09:52.129695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_data[full_data.dataset == 'defog'].index[0]","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:09:52.131906Z","iopub.execute_input":"2024-05-09T15:09:52.132283Z","iopub.status.idle":"2024-05-09T15:09:54.274932Z","shell.execute_reply.started":"2024-05-09T15:09:52.132249Z","shell.execute_reply":"2024-05-09T15:09:54.273999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's take the first one million rows of the full_data dataframe that will correcpond to tdscfog dataset rows, and another one million rows (starting from index 7,062,672 to 8,062,672) will be from the defog dataset. This will give us a balanced final dataset with equal number of samples from both datasets.","metadata":{}},{"cell_type":"code","source":"full_data_sample = pd.concat([full_data[:1000_000], full_data[7_062_672:8_062_672]])\nfull_data_sample","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:09:54.276044Z","iopub.execute_input":"2024-05-09T15:09:54.276342Z","iopub.status.idle":"2024-05-09T15:09:54.486879Z","shell.execute_reply.started":"2024-05-09T15:09:54.276318Z","shell.execute_reply":"2024-05-09T15:09:54.485966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# Example DataFrame\n# Assuming 'full_data_sample' is your existing DataFrame\n# full_data_sample = pd.read_csv('your_data.csv')  # If loading from a CSV file\n\n# Step 1: Handle missing values\n# Use SimpleImputer for numerical and categorical columns\nnum_imputer = SimpleImputer(strategy='mean')  # Fill missing values with mean for numerical columns\ncat_imputer = SimpleImputer(strategy='most_frequent')  # Fill missing values with most frequent for categorical columns\n\n# Step 2: Identify numerical and categorical columns\nnumerical_cols = full_data_sample.select_dtypes(include=[np.number]).columns.tolist()  # All numerical columns\ncategorical_cols = full_data_sample.select_dtypes(include=[object, 'category']).columns.tolist()  # All categorical columns\n\n# Step 3: Convert specific categorical data to numerical\n# Convert 'Medication' to 0 for 'off', 1 otherwise\nfull_data_sample['Medication'] = np.where(full_data_sample['Medication'] == 'off', 0, 1)\n\n# Step 4: Scale numerical data using StandardScaler\nscaler = StandardScaler()  # To standardize the numerical columns\n\n# Step 5: Create a preprocessing pipeline\n# Combine transformations for numerical and categorical columns\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', Pipeline([('impute', num_imputer), ('scale', scaler)]), numerical_cols),  # Numerical data: Impute then scale\n        ('cat', Pipeline([('impute', cat_imputer), ('encode', OneHotEncoder(drop='first'))]), categorical_cols)  # Categorical data: Impute then one-hot encode\n    ],\n    remainder='drop'  # Drop any other columns not specified\n)\n\n\n\n# Step 7: Apply preprocessing pipeline to the DataFrame\n# This will apply the transformations defined in 'preprocessor' to 'full_data_sample'\npreprocessed_data = preprocessor.fit_transform(full_data_sample)\n\n# Step 8: Print the preprocessed data\nprint(\"Preprocessed Data:\")\npreprocessed_data  # This shows the transformed data (likely as a numpy array)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:09:54.488024Z","iopub.execute_input":"2024-05-09T15:09:54.488329Z","iopub.status.idle":"2024-05-09T15:10:03.531697Z","shell.execute_reply.started":"2024-05-09T15:09:54.488304Z","shell.execute_reply":"2024-05-09T15:10:03.530801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_data_sample","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:10:03.532731Z","iopub.execute_input":"2024-05-09T15:10:03.533007Z","iopub.status.idle":"2024-05-09T15:10:03.564435Z","shell.execute_reply.started":"2024-05-09T15:10:03.532984Z","shell.execute_reply":"2024-05-09T15:10:03.56343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Feature Engineering \n\n\nBelow we are creating the function that takes a dataframe as input and generates new features based on the AccV, AccML, and AccAP columns. The function returns the original dataframe with the new features appended.","metadata":{}},{"cell_type":"code","source":"# Create new features based on accelerometer data columns and returns the updated dataframe\ndef create_features(df):\n    acc_cols = ['AccV', 'AccML', 'AccAP']\n\n    # Adjusted backfill operations\n    for acc in acc_cols:\n        df[f'{acc}_lag_2'] = df.groupby('Id')[acc].shift(2).bfill()\n        df[f'{acc}_lag_3'] = df.groupby('Id')[acc].shift(3).bfill()\n        df[f'{acc}_lag_4'] = df.groupby('Id')[acc].shift(4).bfill()\n        df[f'{acc}_lag_5'] = df.groupby('Id')[acc].shift(5).bfill()\n\n        df[f'{acc}_cumsum'] = (df[acc]).groupby(df['Id']).cumsum()\n\n        df[f'{acc}_first_value'] = df.groupby('Id')[acc].transform('first')\n        df[f'{acc}_last_value'] = df.groupby('Id')[acc].transform('last')\n\n        df[f'{acc}_mean'] = df.groupby('Id')[acc].transform('mean')\n        df[f'{acc}_median'] = df.groupby('Id')[acc].transform('median')\n        df[f'{acc}_std'] = df.groupby('Id')[acc].transform('std')\n\n        df[f'{acc}_min'] = df.groupby('Id')[acc].transform('min')\n        df[f'{acc}_max'] = df.groupby('Id')[acc].transform('max')\n\n        df[f'{acc}_delta'] = df[f'{acc}_max'] - df[f'{acc}_min']\n\n        # Adjusted backfill operations for rolling mean\n        for lag in [1, 2, 3]:\n            df[f'ma_{lag}_{acc}'] = df[acc].rolling(lag).mean().bfill()\n            df[f'ma{lag}_{acc}'] = df[acc].rolling(lag).mean().bfill()\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:10:03.565627Z","iopub.execute_input":"2024-05-09T15:10:03.565899Z","iopub.status.idle":"2024-05-09T15:10:03.577336Z","shell.execute_reply.started":"2024-05-09T15:10:03.565876Z","shell.execute_reply":"2024-05-09T15:10:03.576536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply the above function to our dataset\nfull_data_sample = create_features(full_data_sample)\nfull_data_sample","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:10:03.578401Z","iopub.execute_input":"2024-05-09T15:10:03.578673Z","iopub.status.idle":"2024-05-09T15:10:13.090242Z","shell.execute_reply.started":"2024-05-09T15:10:03.57865Z","shell.execute_reply":"2024-05-09T15:10:13.08916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_data_sample.info()","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:10:13.091688Z","iopub.execute_input":"2024-05-09T15:10:13.092013Z","iopub.status.idle":"2024-05-09T15:10:13.108261Z","shell.execute_reply.started":"2024-05-09T15:10:13.091985Z","shell.execute_reply":"2024-05-09T15:10:13.107167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Feature Selection ","metadata":{}},{"cell_type":"code","source":"# Create multiclass event type column\nfull_data_sample['target'] = 'Normal'\ntargets = ['StartHesitation', 'Turn', 'Walking']\nfor target in targets:\n    full_data_sample.loc[full_data_sample[target] == 1, 'target'] = target\n    ","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:10:13.109891Z","iopub.execute_input":"2024-05-09T15:10:13.110238Z","iopub.status.idle":"2024-05-09T15:10:13.149442Z","shell.execute_reply.started":"2024-05-09T15:10:13.110212Z","shell.execute_reply":"2024-05-09T15:10:13.148474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assign variables with predictors and target\nX = full_data_sample.drop(['Id', 'dataset', 'Test', 'Subject', 'UPDRSIII_Off',\n                               'StartHesitation', 'Turn', 'Walking'], axis=1).copy()\ny = X.pop('target')\n","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:10:13.150567Z","iopub.execute_input":"2024-05-09T15:10:13.150852Z","iopub.status.idle":"2024-05-09T15:10:15.067493Z","shell.execute_reply.started":"2024-05-09T15:10:13.150828Z","shell.execute_reply":"2024-05-09T15:10:15.06674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Calculating Mutual information scores","metadata":{}},{"cell_type":"code","source":"\ndiscrete_features = X.dtypes == int\n\n# Calculate MI scores for the features in X with respect to the target variable y\ndef make_mi_scores(X, y, discrete_features):\n    mi_scores = mutual_info_classif(X, y, discrete_features=discrete_features)\n    mi_scores = pd.Series(mi_scores, name=\"MI Scores\", index=X.columns)\n    mi_scores = mi_scores.sort_values(ascending=False)\n    return mi_scores\n\n# Apply this function to our data\nmi_scores = make_mi_scores(X, y, discrete_features)","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:10:15.068621Z","iopub.execute_input":"2024-05-09T15:10:15.068954Z","iopub.status.idle":"2024-05-09T15:27:09.433908Z","shell.execute_reply.started":"2024-05-09T15:10:15.068928Z","shell.execute_reply":"2024-05-09T15:27:09.432806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show 20 features with the highest MI score\nmi_scores.to_frame().head(20)","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:27:09.435394Z","iopub.execute_input":"2024-05-09T15:27:09.436191Z","iopub.status.idle":"2024-05-09T15:27:09.44666Z","shell.execute_reply.started":"2024-05-09T15:27:09.436157Z","shell.execute_reply":"2024-05-09T15:27:09.445844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_mi_scores(scores):\n    scores = scores.sort_values(ascending=False)\n    fig, ax = plt.subplots(figsize=(13, 11))\n    ax = sns.barplot(x=scores.values, y=scores.index, palette=\"coolwarm\", orient='h')\n    ax.set_title(\"Mutual Information Scores\")\n    ax.set_xlabel(\"MI Scores\")\n    ax.set_ylabel(\"Features\")\n    plt.show()\n\nplot_mi_scores(mi_scores)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:27:09.447943Z","iopub.execute_input":"2024-05-09T15:27:09.448241Z","iopub.status.idle":"2024-05-09T15:27:10.418772Z","shell.execute_reply.started":"2024-05-09T15:27:09.448217Z","shell.execute_reply":"2024-05-09T15:27:10.417844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Keep the top `n` features\nn_top_features = 20  # Or any other number you prefer\n\n# Get the top `n` features from MI scores\ntop_features = mi_scores.head(n_top_features).index.tolist()\n\n# Ensure 'Time' is in the list of features to keep\n# Add 'Time' if it's not already in the list\nif 'Time' not in top_features:\n     top_features = ['Time'] + top_features\n\n# Now you have your final list of features to keep, including 'Time'\nfiltered_df = X[top_features]  # Keep only the specified features\n\nfiltered_df","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:27:10.419872Z","iopub.execute_input":"2024-05-09T15:27:10.420132Z","iopub.status.idle":"2024-05-09T15:27:10.766099Z","shell.execute_reply.started":"2024-05-09T15:27:10.420099Z","shell.execute_reply":"2024-05-09T15:27:10.765182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:27:10.767412Z","iopub.execute_input":"2024-05-09T15:27:10.767796Z","iopub.status.idle":"2024-05-09T15:27:11.010257Z","shell.execute_reply.started":"2024-05-09T15:27:10.767761Z","shell.execute_reply":"2024-05-09T15:27:11.009278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Example with SMOTE for oversampling\nsmote = SMOTE()\nX_smote, y_smote = smote.fit_resample(filtered_df, y)\n\n# Print the counts for each class after SMOTE\nprint(\"SMOTE Class Distribution:\")\nprint(y_smote.value_counts())\n","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:27:11.011322Z","iopub.execute_input":"2024-05-09T15:27:11.011602Z","iopub.status.idle":"2024-05-09T15:34:21.215944Z","shell.execute_reply.started":"2024-05-09T15:27:11.011579Z","shell.execute_reply":"2024-05-09T15:34:21.215005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a function to reduce memory usage\ndef reduce_mem_usage(df):\n    \n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n\n    for col in df.columns:\n        col_type = df[col].dtype.name\n\n        if col_type not in ['object', 'category', 'datetime64[ns, UTC]']:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n   \n    return df ","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:34:21.217468Z","iopub.execute_input":"2024-05-09T15:34:21.217751Z","iopub.status.idle":"2024-05-09T15:34:21.229761Z","shell.execute_reply.started":"2024-05-09T15:34:21.217726Z","shell.execute_reply":"2024-05-09T15:34:21.228614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# Apply the above function to our dataset\nfull_data = reduce_mem_usage(full_data_sample)","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:34:21.23499Z","iopub.execute_input":"2024-05-09T15:34:21.235383Z","iopub.status.idle":"2024-05-09T15:34:22.510773Z","shell.execute_reply.started":"2024-05-09T15:34:21.235353Z","shell.execute_reply":"2024-05-09T15:34:22.509842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# Split the balanced data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X_smote, y_smote, test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:34:22.512014Z","iopub.execute_input":"2024-05-09T15:34:22.512321Z","iopub.status.idle":"2024-05-09T15:34:25.400717Z","shell.execute_reply.started":"2024-05-09T15:34:22.512295Z","shell.execute_reply":"2024-05-09T15:34:25.399933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a base LGBM classifier\nlgbm_clf = LGBMClassifier(random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:34:25.401881Z","iopub.execute_input":"2024-05-09T15:34:25.402179Z","iopub.status.idle":"2024-05-09T15:34:25.406838Z","shell.execute_reply.started":"2024-05-09T15:34:25.40214Z","shell.execute_reply":"2024-05-09T15:34:25.40595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Hyperparameter grid for GridSearchCV\nlgbm_params = {\n    'boosting_type': ['gbdt'],  # Gradient Boosting Decision Trees\n    'objective': ['multiclass'],  # Multiclass classification\n    'num_class': [3],  # Number of classes in the target variable\n    'metric': ['multi_logloss'], # metric for evaluation\n    'learning_rate': [0.01, 0.1],  # Tune different learning rates\n    'num_leaves': [31, 63],  # Tune number of leaves\n    'max_depth': [-1, 7],  # Tune tree depth (unlimited or 7)\n    'n_estimators': [100, 200],  # Tune number of boosting iterations\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:34:25.407942Z","iopub.execute_input":"2024-05-09T15:34:25.408239Z","iopub.status.idle":"2024-05-09T15:34:25.419914Z","shell.execute_reply.started":"2024-05-09T15:34:25.408215Z","shell.execute_reply":"2024-05-09T15:34:25.419164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize GridSearchCV with the parameter grid and k-fold cross-validation\ngrid_search = GridSearchCV(lgbm_clf, lgbm_params, cv = 5 , scoring='accuracy', n_jobs=-1, verbose=2)","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:34:25.422345Z","iopub.execute_input":"2024-05-09T15:34:25.423017Z","iopub.status.idle":"2024-05-09T15:34:25.435369Z","shell.execute_reply.started":"2024-05-09T15:34:25.422991Z","shell.execute_reply":"2024-05-09T15:34:25.434549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Perform hyperparameter tuning on the training set\ngrid_search.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-09T15:34:25.436407Z","iopub.execute_input":"2024-05-09T15:34:25.437019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Get the best model and its hyperparameters\nbest_lgbm_clf = grid_search.best_estimator_\nprint(\"Best hyperparameters:\", grid_search.best_params_)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Train the model with the best hyperparameters\nbest_lgbm_clf.fit(X_train, y_train)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_test = best_lgbm_clf.predict(X_test)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model Evaluation\n\ndef class_accuracy(class_name, y_true, y_pred):\n    class_indices = (y_true == class_name)\n    return accuracy_score(y_true[class_indices], y_pred[class_indices])\n\naccuracy_start_hesitation = class_accuracy(\"start hesitation\", y_test, y_pred_test)\naccuracy_turn = class_accuracy(\"turn\", y_test, y_pred_test)\naccuracy_walking = class_accuracy(\"walking\", y_test, y_pred_test)\n\nprecision_scores = precision_score(y_test, y_pred_test, average=None, labels=[\"start hesitation\", \"turn\", \"walking\"])\nrecall_scores = recall_score(y_test, y_pred_test, average=None, labels=[\"start hesitation\", \"turn\", \"walking\"])\nf1_scores = f1_score(y_test, y_pred_test, average=None, labels=[\"start hesitation\", \"turn\", \"walking\"])\n\nprint(\"Class-wise accuracy:\")\nprint(\"  Start Hesitation:\", accuracy_start_hesitation)\nprint(\"  Turn:\", accuracy_turn)\nprint(\"  Walking:\", accuracy_walking)\n\nprint(\"Class-wise precision:\")\nprint(\"  Start Hesitation:\", precision_scores[0])\nprint(\"  Turn:\", precision_scores[1])\nprint(\"  Walking:\", precision_scores[2])\n\nprint(\"Class-wise recall:\")\nprint(\"  Start Hesitation:\", recall_scores[0])\nprint(\"  Turn:\", recall_scores[1])\nprint(\"  Walking:\", recall_scores[2])\n\nprint(\"Class-wise F1-score:\")\nprint(\"  Start Hesitation:\", f1_scores[0])\nprint(\"  Turn:\", f1_scores[1])\nprint(\"  Walking:\", f1_scores[2])\n\n# Print confusion matrix for each event\nfor event in [\"start hesitation\", \"turn\", \"walking\"]:\n    y_true_event = y_test[y_test == event]\n    y_pred_event = y_pred_test[y_pred_test == event]\n    cm = confusion_matrix(y_true_event, y_pred_event)\n    print(f\"\\nConfusion matrix for {event}:\")\n    print(cm)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Classification and prediction\ndef predict_duration(model, X, feature_cols):\n    durations = {'StartHesitation': 0, 'Turn': 0, 'Walking': 0}\n    for idx, row in X.iterrows():\n        input_data = row[feature_cols].values.reshape(1, -1)\n        predicted_class = model.predict(input_data)\n        durations[predicted_class[0]] += 1\n    return durations\n\nfeature_cols = top_features\ndurations = predict_duration(best_lgbm_clf, X_test, feature_cols)\nprint(\"Class durations:\", durations)\n\nmax_duration_class = max(durations, key=durations.get)\nprint(f\"The class with the longest duration is: {max_duration_class}\")\n\nsubject_durations = X_test.groupby('Subject').apply(lambda x: predict_duration(best_lgbm_clf, x, feature_cols))\nmax_duration_subject = subject_durations[max_duration_class].idxmax()\nprint(f\"The subject with the longest duration for class {max_duration_class} is: {max_duration_subject}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}