{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.11"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":102335,"databundleVersionId":12518947,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":27991.192132,"end_time":"2025-05-27T01:12:27.803749","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-05-26T17:25:56.611617","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"01f35722","cell_type":"markdown","source":"# CMI - Detect Behavior\n---\n\n### This notebook will walk you through a complete exploratory data analysis (EDA) of the CMI Detect Behavior Kaggle challenge.  If you found this notebook useful, please upvote and/or leave a comment to make it better -- thanks for reviewing!","metadata":{"papermill":{"duration":0.006833,"end_time":"2025-05-26T17:26:01.976444","exception":false,"start_time":"2025-05-26T17:26:01.969611","status":"completed"},"tags":[]}},{"id":"d2f3edf8","cell_type":"markdown","source":"# Libraries","metadata":{"papermill":{"duration":0.005339,"end_time":"2025-05-26T17:26:01.988914","exception":false,"start_time":"2025-05-26T17:26:01.983575","status":"completed"},"tags":[]}},{"id":"edcea8cc","cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport gc\nimport warnings\nwarnings.filterwarnings('ignore', category=FutureWarning)\nfrom matplotlib import pyplot as plt\nfrom sklearn.model_selection import KFold","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2025-05-30T16:49:56.960112Z","iopub.execute_input":"2025-05-30T16:49:56.960431Z","iopub.status.idle":"2025-05-30T16:49:56.965432Z","shell.execute_reply.started":"2025-05-30T16:49:56.960375Z","shell.execute_reply":"2025-05-30T16:49:56.964588Z"},"papermill":{"duration":5.478671,"end_time":"2025-05-26T17:26:07.473784","exception":false,"start_time":"2025-05-26T17:26:01.995113","status":"completed"},"tags":[],"ExecuteTime":{"end_time":"2025-05-30T15:42:34.798909Z","start_time":"2025-05-30T15:42:34.694766Z"},"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"id":"232cced1","cell_type":"markdown","source":"# Reading competition material","metadata":{"papermill":{"duration":0.00525,"end_time":"2025-05-26T17:26:07.484716","exception":false,"start_time":"2025-05-26T17:26:07.479466","status":"completed"},"tags":[]}},{"id":"e2717ee3","cell_type":"code","source":"train = pd.read_csv('/kaggle/input/cmi-detect-behavior-with-sensor-data/train.csv')\ntrain_demographics = pd.read_csv('/kaggle/input/cmi-detect-behavior-with-sensor-data/train_demographics.csv')\ntest = pd.read_csv('/kaggle/input/cmi-detect-behavior-with-sensor-data/test.csv')\ntest_demographics = pd.read_csv('/kaggle/input/cmi-detect-behavior-with-sensor-data/test_demographics.csv')","metadata":{"execution":{"iopub.status.busy":"2025-05-30T16:42:00.875713Z","iopub.execute_input":"2025-05-30T16:42:00.876206Z","iopub.status.idle":"2025-05-30T16:42:35.599273Z","shell.execute_reply.started":"2025-05-30T16:42:00.876178Z","shell.execute_reply":"2025-05-30T16:42:35.598621Z"},"papermill":{"duration":58.569029,"end_time":"2025-05-26T17:27:06.059156","exception":false,"start_time":"2025-05-26T17:26:07.490127","status":"completed"},"tags":[],"ExecuteTime":{"end_time":"2025-05-30T15:43:00.433702Z","start_time":"2025-05-30T15:42:54.034923Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"623414ee","cell_type":"markdown","source":"# Exploratory Data Analysis","metadata":{"papermill":{"duration":0.00502,"end_time":"2025-05-26T17:27:06.069940","exception":false,"start_time":"2025-05-26T17:27:06.064920","status":"completed"},"tags":[]}},{"id":"8115bf23","cell_type":"markdown","source":"### train.csv","metadata":{"papermill":{"duration":0.005386,"end_time":"2025-05-26T17:27:06.080559","exception":false,"start_time":"2025-05-26T17:27:06.075173","status":"completed"},"tags":[]}},{"id":"ac5b65f0","cell_type":"code","source":"print(f\"This dataset has {train.shape[0]} rows and {train.shape[1]} columns.\")\nprint(f\"There are {train.isna().sum().sum()} NA's in the dataset.\")\nprint(f\"There are {str(train.duplicated().sum())} duplicates in the dataset.\")\n\n# quick look at the data\ntrain.head(3)\n\n# There are quite a few NA's in this dataset, amounting to ~ 1.8% of the dataset.","metadata":{"execution":{"iopub.status.busy":"2025-05-30T16:43:44.374207Z","iopub.execute_input":"2025-05-30T16:43:44.374545Z","iopub.status.idle":"2025-05-30T16:43:50.402159Z","shell.execute_reply.started":"2025-05-30T16:43:44.374521Z","shell.execute_reply":"2025-05-30T16:43:50.401444Z"},"papermill":{"duration":40.506759,"end_time":"2025-05-26T17:27:46.592633","exception":false,"start_time":"2025-05-26T17:27:06.085874","status":"completed"},"tags":[],"ExecuteTime":{"end_time":"2025-05-30T15:43:07.669571Z","start_time":"2025-05-30T15:43:05.618854Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"ef50fdb2-19a8-4fb9-bb40-db8b66261e1e","cell_type":"markdown","source":"### test.csv","metadata":{}},{"id":"7b5bf2fd58f89949","cell_type":"code","source":"print(f\"This dataset has {test.shape[0]} rows and {test.shape[1]} columns.\")\nprint(f\"There are {test.isna().sum().sum()} NA's in the dataset.\")\nprint(f\"There are {str(test.duplicated().sum())} duplicates in the dataset.\")\n\n\n# quick look at the data\ntest.head(3)\n\n# There are no NA's in the test, but according to the competition material, we can expect them in test as well.","metadata":{"ExecuteTime":{"end_time":"2025-05-30T16:27:06.777314Z","start_time":"2025-05-30T16:27:06.757701Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T16:43:53.822828Z","iopub.execute_input":"2025-05-30T16:43:53.823575Z","iopub.status.idle":"2025-05-30T16:43:53.881936Z","shell.execute_reply.started":"2025-05-30T16:43:53.823544Z","shell.execute_reply":"2025-05-30T16:43:53.881157Z"}},"outputs":[],"execution_count":null},{"id":"1275b69f","cell_type":"markdown","source":"### To start, we will identify our unique_id & target.  We will also initially group our features into categorical / numerical groups by dtype as follows:","metadata":{"papermill":{"duration":0.006351,"end_time":"2025-05-26T17:28:17.101083","exception":false,"start_time":"2025-05-26T17:28:17.094732","status":"completed"},"tags":[]}},{"id":"39be57ae","cell_type":"code","source":"target = 'gesture'\nunique_id = 'sequence_id'\ncategorical_columns = [col for col in train.columns if col not in [target,unique_id] if train[col].dtype in ['object', 'category']]\nnumerical_columns = [col for col in train.columns if col not in [target,unique_id]  if train[col].dtype not in ['object','category']]\n\nprint(f'There are {len(categorical_columns)} categorical columns: {categorical_columns}')\nprint(f'There are {len(numerical_columns)} numerical columns: {numerical_columns}')","metadata":{"execution":{"iopub.status.busy":"2025-05-30T16:44:33.348975Z","iopub.execute_input":"2025-05-30T16:44:33.349269Z","iopub.status.idle":"2025-05-30T16:44:33.359576Z","shell.execute_reply.started":"2025-05-30T16:44:33.349250Z","shell.execute_reply":"2025-05-30T16:44:33.358418Z"},"papermill":{"duration":0.015367,"end_time":"2025-05-26T17:28:17.122453","exception":false,"start_time":"2025-05-26T17:28:17.107086","status":"completed"},"tags":[],"ExecuteTime":{"end_time":"2025-05-30T15:43:14.088845Z","start_time":"2025-05-30T15:43:14.081418Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"47d4ca72","cell_type":"markdown","source":"#### train.csv skewed numerical features:","metadata":{"papermill":{"duration":0.005842,"end_time":"2025-05-26T17:28:17.134699","exception":false,"start_time":"2025-05-26T17:28:17.128857","status":"completed"},"tags":[]}},{"id":"e7a4c796","cell_type":"code","source":"skewness_threshold = .5 # can tune / experiment with this value\nskewed_cols = [col for col in numerical_columns if train[col].skew() > skewness_threshold]\n\nprint(f'There are {len(skewed_cols)} skewed columns: {str(skewed_cols)}')\n\n# According to our threshold of .5, the majority of numerical columns are skewed.","metadata":{"execution":{"iopub.status.busy":"2025-05-30T16:44:49.203641Z","iopub.execute_input":"2025-05-30T16:44:49.203960Z","iopub.status.idle":"2025-05-30T16:44:51.199643Z","shell.execute_reply.started":"2025-05-30T16:44:49.203936Z","shell.execute_reply":"2025-05-30T16:44:51.198696Z"},"papermill":{"duration":6.678288,"end_time":"2025-05-26T17:28:23.819168","exception":false,"start_time":"2025-05-26T17:28:17.140880","status":"completed"},"tags":[],"ExecuteTime":{"end_time":"2025-05-30T15:43:17.152093Z","start_time":"2025-05-30T15:43:16.445528Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"23e2cbb9aef50d79","cell_type":"markdown","source":"#### train.csv imbalanced categorical features:","metadata":{}},{"id":"83132fc7cee89573","cell_type":"code","source":"imbalance_threshold = 3.33\nimbalanced_cols = [col for col in categorical_columns if (train[col].count() / train[col].value_counts().values.min()) > imbalance_threshold]\n\nprint(f'There are {len(imbalanced_cols)} imbalanced columns: {str(imbalanced_cols)}')\n\n# According to our imbalance_threshold of 3.33, of our 6 categorical features, 4 are imbalanced.  We note row_id is probably not going to be used as a feature for modelling.","metadata":{"ExecuteTime":{"end_time":"2025-05-30T16:03:35.019425Z","start_time":"2025-05-30T16:03:34.731144Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T16:46:24.222062Z","iopub.execute_input":"2025-05-30T16:46:24.222530Z","iopub.status.idle":"2025-05-30T16:46:25.077614Z","shell.execute_reply.started":"2025-05-30T16:46:24.222498Z","shell.execute_reply":"2025-05-30T16:46:25.076735Z"}},"outputs":[],"execution_count":null},{"id":"9b28d8a0afef27f8","cell_type":"markdown","source":"#### train.csv high cardinality categorical features:","metadata":{}},{"id":"30fda20072cab23c","cell_type":"code","source":"high_cardinality_threshold = 8\nhigh_cardinality_columns = [x for x in categorical_columns if train[x].nunique() > high_cardinality_threshold]\n\nprint(f'There are {len(high_cardinality_columns)} high cardinality columns: {str(high_cardinality_columns)}')\n\n# According to our high cardinality threshold, row_id & subject are high cardinality columns.  Again, row_id is probably not going to be used as a feature for modelling.","metadata":{"ExecuteTime":{"end_time":"2025-05-30T16:03:41.793375Z","start_time":"2025-05-30T16:03:41.680346Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T16:47:12.495656Z","iopub.execute_input":"2025-05-30T16:47:12.495968Z","iopub.status.idle":"2025-05-30T16:47:12.791700Z","shell.execute_reply.started":"2025-05-30T16:47:12.495942Z","shell.execute_reply":"2025-05-30T16:47:12.790800Z"}},"outputs":[],"execution_count":null},{"id":"f4f8efdb","cell_type":"markdown","source":"### Let's make sure all the columns in train exist in test, and vice versa...except for the target...for both sets of training files.","metadata":{"papermill":{"duration":0.009382,"end_time":"2025-05-26T17:28:23.840387","exception":false,"start_time":"2025-05-26T17:28:23.831005","status":"completed"},"tags":[]}},{"id":"6d5556f6","cell_type":"code","source":"train_unique_cols = [x for x in train.drop([target],axis=1).columns if x not in test.columns]\nprint('All columns in train exist in test.') if not train_unique_cols else print(f'The following train columns are not in test: {train_unique_cols}')\n\ntest_unique_cols = [x for x in test.columns if x not in train.columns]\nprint('All columns in test exist in train.') if not test_unique_cols else print(f'The following test columns are not in train: {test_unique_cols}')\n\ntrain_unique_cols = [x for x in train_demographics.columns if x not in test_demographics.columns]\nprint('All columns in train_demographics exist in test_demographics.') if not train_unique_cols else print(f'The following train columns are not in test: {train_unique_cols}')\n\ntest_unique_cols = [x for x in test_demographics.columns if x not in train_demographics.columns]\nprint('All columns in test_demographics exist in train_demographics.') if not test_unique_cols else print(f'The following test columns are not in train: {test_unique_cols}')\n\n\n# According to the competition material, sequence_type, orientation, & behavior are train only.  We discover 'phase' is also only train, which is not unexpected given what we are trying to predict in this competition.","metadata":{"execution":{"iopub.status.busy":"2025-05-30T16:48:11.068979Z","iopub.execute_input":"2025-05-30T16:48:11.069540Z","iopub.status.idle":"2025-05-30T16:48:11.578372Z","shell.execute_reply.started":"2025-05-30T16:48:11.069514Z","shell.execute_reply":"2025-05-30T16:48:11.577477Z"},"papermill":{"duration":0.620483,"end_time":"2025-05-26T17:28:24.467989","exception":false,"start_time":"2025-05-26T17:28:23.847506","status":"completed"},"tags":[],"ExecuteTime":{"end_time":"2025-05-30T15:44:50.803436Z","start_time":"2025-05-30T15:44:50.535763Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"f7eba87d","cell_type":"markdown","source":"#### Histplot & Boxplots for numerical features. We will sample 10% of the data of the columns that are not skewed.","metadata":{"papermill":{"duration":0.00672,"end_time":"2025-05-26T17:28:24.481415","exception":false,"start_time":"2025-05-26T17:28:24.474695","status":"completed"},"tags":[]}},{"id":"c615d638","cell_type":"code","source":"not_skewed_num = ['acc_x', 'acc_y', 'acc_z', 'rot_x', 'rot_y', 'rot_z', 'thm_1', 'thm_3', 'thm_4']\ntrain_samp = train[not_skewed_num].sample(frac=0.1, replace=False, random_state=1)\n\nfor col in not_skewed_num:\n    _, axes = plt.subplots(1,2,figsize=(8,4),sharex=False,sharey=False)\n   \n    sns.histplot(data=train_samp.dropna(), x=col,bins=10,ax=axes[0],log_scale=False, color = 'lightblue')\n    axes[0].set_title(f'{col}')\n\n    sns.boxplot(data=train_samp.dropna(),x=col,ax=axes[1],showfliers=False, color = 'honeydew')\n    axes[1].set_title(f'{col} (no outliers)')\n\n    plt.tight_layout()\n    plt.show()\n\ndel train_samp\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2025-05-30T16:50:09.832830Z","iopub.execute_input":"2025-05-30T16:50:09.833747Z","iopub.status.idle":"2025-05-30T16:50:12.812122Z","shell.execute_reply.started":"2025-05-30T16:50:09.833716Z","shell.execute_reply":"2025-05-30T16:50:12.811079Z"},"papermill":{"duration":3.656037,"end_time":"2025-05-26T17:28:28.145126","exception":false,"start_time":"2025-05-26T17:28:24.489089","status":"completed"},"tags":[],"ExecuteTime":{"end_time":"2025-05-30T15:54:56.532274Z","start_time":"2025-05-30T15:54:55.197964Z"},"trusted":true,"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"id":"ef7759b5-8c07-4691-b369-53736bccb4c8","cell_type":"markdown","source":"#### Countplot for categorical features. We will sample 10% of the data of the columns that do not have a high cardinality.","metadata":{}},{"id":"fb3fd600d60a6501","cell_type":"code","source":"cat_plot_cols = [col for col in categorical_columns if col not in high_cardinality_columns]\ntrain_samp = train[cat_plot_cols].sample(frac=0.1, replace=False, random_state=1)\n\nfor col in cat_plot_cols:\n    _, ax = plt.subplots(1,1,figsize=(6,3))\n\n    sns.countplot(data=train_samp.dropna(), y = col, color = 'lightblue')\n    ax.set_title(f'{col}')\n\n    plt.tight_layout()\n    plt.show()\n\ndel train_samp\ngc.collect()","metadata":{"ExecuteTime":{"end_time":"2025-05-30T16:13:14.067045Z","start_time":"2025-05-30T16:13:09.875049Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T16:50:38.962968Z","iopub.execute_input":"2025-05-30T16:50:38.963280Z","iopub.status.idle":"2025-05-30T16:50:39.813548Z","shell.execute_reply.started":"2025-05-30T16:50:38.963256Z","shell.execute_reply":"2025-05-30T16:50:39.812645Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"id":"95ee5f0dd0c73393","cell_type":"markdown","source":"#### train_demographics","metadata":{}},{"id":"4b30feed8573dba3","cell_type":"code","source":"print(f\"This dataset has {train_demographics.shape[0]} rows and {train_demographics.shape[1]} columns.\")\nprint(f\"There are {train_demographics.isna().sum().sum()} NA's in the dataset.\")\nprint(f\"There are {str(train_demographics.duplicated().sum())} duplicates in the dataset.\")\n\n# quick look at the data\ntrain_demographics.head(3)\n\n# There are 0 NA's in this dataset.","metadata":{"ExecuteTime":{"end_time":"2025-05-30T16:14:38.098042Z","start_time":"2025-05-30T16:14:38.082324Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T16:51:45.917389Z","iopub.execute_input":"2025-05-30T16:51:45.918062Z","iopub.status.idle":"2025-05-30T16:51:45.930995Z","shell.execute_reply.started":"2025-05-30T16:51:45.918034Z","shell.execute_reply":"2025-05-30T16:51:45.930113Z"}},"outputs":[],"execution_count":null},{"id":"619938ae-ab17-433a-978e-77a81b825b87","cell_type":"markdown","source":"#### test_demographics","metadata":{}},{"id":"f461b1da6db7aab8","cell_type":"code","source":"print(f\"This dataset has {test_demographics.shape[0]} rows and {test_demographics.shape[1]} columns.\")\nprint(f\"There are {test_demographics.isna().sum().sum()} NA's in the dataset.\")\nprint(f\"There are {str(test_demographics.duplicated().sum())} duplicates in the dataset.\")\n\n# quick look at the data\ntest_demographics.head(3)\n\n# There are 0 NA's in this dataset, but according to the competition material, we can expect some.","metadata":{"ExecuteTime":{"end_time":"2025-05-30T16:28:20.094052Z","start_time":"2025-05-30T16:28:20.076951Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T16:52:13.468314Z","iopub.execute_input":"2025-05-30T16:52:13.468688Z","iopub.status.idle":"2025-05-30T16:52:13.482364Z","shell.execute_reply.started":"2025-05-30T16:52:13.468661Z","shell.execute_reply":"2025-05-30T16:52:13.481461Z"}},"outputs":[],"execution_count":null},{"id":"e503d03b-ae77-42da-b4f9-d0a81952b6c5","cell_type":"markdown","source":"### Grouping our features into categorical / numerical groups by dtype as follows:","metadata":{}},{"id":"110e5855e4c5751b","cell_type":"code","source":"demo_categorical_columns = [col for col in train_demographics.columns if col not in [target,unique_id] if train_demographics[col].dtype in ['object', 'category']]\ndemo_numerical_columns = [col for col in train_demographics.columns if col not in [target,unique_id]  if train_demographics[col].dtype not in ['object','category']]\n\nprint(f'There are {len(demo_categorical_columns)} categorical columns: {demo_categorical_columns}')\nprint(f'There are {len(demo_numerical_columns)} numerical columns: {demo_numerical_columns}')","metadata":{"ExecuteTime":{"end_time":"2025-05-30T16:17:04.763222Z","start_time":"2025-05-30T16:17:04.759623Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T16:52:27.491520Z","iopub.execute_input":"2025-05-30T16:52:27.491830Z","iopub.status.idle":"2025-05-30T16:52:27.497915Z","shell.execute_reply.started":"2025-05-30T16:52:27.491808Z","shell.execute_reply":"2025-05-30T16:52:27.496687Z"}},"outputs":[],"execution_count":null},{"id":"d175d160-4c42-43f9-a4c7-b958b97c06c2","cell_type":"markdown","source":"#### train_demographics.csv skewed numerical features:","metadata":{}},{"id":"415e2f577d3ffed4","cell_type":"code","source":"skewness_threshold = .5 # can tune / experiment with this value\ndemo_skewed_cols = [col for col in demo_numerical_columns if train_demographics[col].skew() > skewness_threshold]\n\nprint(f'There are {len(demo_skewed_cols)} skewed columns: {str(demo_skewed_cols)}')\n\n# According to our threshold of .5, 2 of 7 numerical columns in train_demographics are skewed.","metadata":{"ExecuteTime":{"end_time":"2025-05-30T16:18:46.626777Z","start_time":"2025-05-30T16:18:46.622264Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T16:54:19.773382Z","iopub.execute_input":"2025-05-30T16:54:19.774367Z","iopub.status.idle":"2025-05-30T16:54:19.784232Z","shell.execute_reply.started":"2025-05-30T16:54:19.774329Z","shell.execute_reply":"2025-05-30T16:54:19.783070Z"}},"outputs":[],"execution_count":null},{"id":"c88b4093-89cc-4735-949c-8f66d5f3ed78","cell_type":"markdown","source":"#### train_demographics.csv imbalanced features:","metadata":{}},{"id":"d3fae1e582ce9e1","cell_type":"code","source":"imbalance_threshold = 3.33\ndemo_imbalanced_cols = [col for col in demo_categorical_columns if (train_demographics[col].count() / train_demographics[col].value_counts().values.min()) > imbalance_threshold]\n\nprint(f'There are {len(demo_imbalanced_cols)} imbalanced columns: {str(demo_imbalanced_cols)}')\n\n# According to our imbalance_threshold of 3.33, subject is imbalanced. 'subject' is used as an identifier, so it will not be used in modelling efforts.","metadata":{"ExecuteTime":{"end_time":"2025-05-30T16:19:54.063347Z","start_time":"2025-05-30T16:19:54.058150Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T16:55:21.709698Z","iopub.execute_input":"2025-05-30T16:55:21.710013Z","iopub.status.idle":"2025-05-30T16:55:21.717986Z","shell.execute_reply.started":"2025-05-30T16:55:21.709987Z","shell.execute_reply":"2025-05-30T16:55:21.717026Z"}},"outputs":[],"execution_count":null},{"id":"863aa8ed-c949-48a5-a0f3-8af52f11102b","cell_type":"markdown","source":"#### train_demographics.csv high cardinality features:","metadata":{}},{"id":"374bd7cc5f1c91fb","cell_type":"code","source":"high_cardinality_threshold = 8\ndemo_high_cardinality_columns = [x for x in demo_categorical_columns if train_demographics[x].nunique() > high_cardinality_threshold]\n\nprint(f'There are {len(demo_high_cardinality_columns)} high cardinality columns: {str(demo_high_cardinality_columns)}')\n\n# According to our high cardinality threshold, subject has a high cardinality.  'subject' is used as an identifier, so it will not be used in modelling efforts.","metadata":{"ExecuteTime":{"end_time":"2025-05-30T16:20:48.341697Z","start_time":"2025-05-30T16:20:48.336763Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T16:55:34.556954Z","iopub.execute_input":"2025-05-30T16:55:34.557288Z","iopub.status.idle":"2025-05-30T16:55:34.563114Z","shell.execute_reply.started":"2025-05-30T16:55:34.557268Z","shell.execute_reply":"2025-05-30T16:55:34.562082Z"}},"outputs":[],"execution_count":null},{"id":"7fb3669e9c92d12","cell_type":"code","source":"train_samp = train_demographics[demo_numerical_columns].sample(frac=0.1, replace=False, random_state=1)\n\nfor col in demo_numerical_columns:\n    _, axes = plt.subplots(1,2,figsize=(8,4),sharex=False,sharey=False)\n\n    sns.histplot(data=train_samp.dropna(), x=col,bins=10,ax=axes[0],log_scale=False, color = 'lightblue')\n    axes[0].set_title(f'{col}')\n\n    sns.boxplot(data=train_samp.dropna(),x=col,ax=axes[1],showfliers=True, color = 'honeydew')\n    axes[1].set_title(f'{col} (no outliers)')\n\n    plt.tight_layout()\n    plt.show()\n\ndel train_samp\ngc.collect()\n\n# Clearly, some of the numerical features in train_demographics would be better treated as categorical, such as adult_child, sex, & handedness.","metadata":{"ExecuteTime":{"end_time":"2025-05-30T16:24:38.719231Z","start_time":"2025-05-30T16:24:37.798311Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T16:57:07.326845Z","iopub.execute_input":"2025-05-30T16:57:07.327503Z","iopub.status.idle":"2025-05-30T16:57:09.588647Z","shell.execute_reply.started":"2025-05-30T16:57:07.327474Z","shell.execute_reply":"2025-05-30T16:57:09.587752Z"}},"outputs":[],"execution_count":null},{"id":"75fc2af08bb860eb","cell_type":"markdown","source":"### The target: gesture","metadata":{}},{"id":"354d7912","cell_type":"code","source":"train_samp = pd.DataFrame(train[target]).sample(frac=0.1, replace=False, random_state=1).sort_index()\n\n_, ax = plt.subplots(1,1,figsize=(8,4))\nsns.countplot(data=train_samp.dropna(), y = target, color = 'lightblue')\nax.set_title(f'{target}')\n\nplt.tight_layout()\nplt.show()\n\ndel train_samp\ngc.collect()\n\n# The most prevalent gestures are 'Neck - scratch' & 'Text on phone'.  The least prevalent are 'Write name on leg' & 'Pinch knee/leg skin'.","metadata":{"execution":{"iopub.status.busy":"2025-05-30T16:57:25.272654Z","iopub.execute_input":"2025-05-30T16:57:25.272959Z","iopub.status.idle":"2025-05-30T16:57:25.725232Z","shell.execute_reply.started":"2025-05-30T16:57:25.272937Z","shell.execute_reply":"2025-05-30T16:57:25.724307Z"},"papermill":{"duration":1.876257,"end_time":"2025-05-26T17:29:09.444058","exception":false,"start_time":"2025-05-26T17:29:07.567801","status":"completed"},"tags":[],"ExecuteTime":{"end_time":"2025-05-30T16:31:53.116304Z","start_time":"2025-05-30T16:31:52.785306Z"},"trusted":true},"outputs":[],"execution_count":null},{"id":"8cfbf2ec","cell_type":"markdown","source":"# If you found this notebook useful please vote and/or leave a suggestion for improvements -- thanks for reviewing!","metadata":{"papermill":{"duration":0.01413,"end_time":"2025-05-27T01:12:23.837435","exception":false,"start_time":"2025-05-27T01:12:23.823305","status":"completed"},"tags":[]}}]}