{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport polars as pl\nimport seaborn as sns\nimport xgboost as xgb\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.model_selection import train_test_split\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\n\n\n\ntrain = pl.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pl.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-07T09:53:41.098677Z","iopub.execute_input":"2024-10-07T09:53:41.099191Z","iopub.status.idle":"2024-10-07T09:53:41.120733Z","shell.execute_reply.started":"2024-10-07T09:53:41.099145Z","shell.execute_reply":"2024-10-07T09:53:41.119705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This dataset contains parquet files as well. We've been using pandas for data handling and analysis.\n\nThis seemed like a good oppurtunity to explore polaris : https://pola.rs/\n\nWhich is claimed to be faster than pandas. ","metadata":{}},{"cell_type":"code","source":"# creating a helper class here for methods that are re-usable:\nclass Utils:\n    def __init__(self):\n        pass\n    \n    \n    def plot_bp(self, graph_data : dict, data_frame, config:dict):\n        if config['plot_type'] == 'line_plot':\n            sns.set_style('whitegrid')\n            plt.figure(figsize=(10, 6))\n            sns.lineplot(data=data_frame, x=graph_specs['x'], y='Physical-Systolic_BP', marker='o', label='systolic')\n            sns.lineplot(data=data_frame, x=graph_specs['x'], y='Physical-Diastolic_BP',marker='o', label='diastolic')\n            plt.title(graph_data['title'])\n            plt.xlabel(graph_data['x_label'])\n            plt.ylabel('Blood Pressure')\n            plt.legend(title=graph_data['legend'])\n            plt.show()\n        elif config['plot_type'] == 'scatter_plot':\n            print('coming soon')\n        else:\n            print('utils : todo-> explore and write code for pair plots')\n            \n            \n    def corr_matrix_heatmap(self, correlation_matrix, config:dict={}):\n        sns.set(rc={'figure.figsize': (20, 10)})\n        column_names = correlation_matrix.columns\n        sns.heatmap(correlation_matrix, annot=True, cmap='magma', linewidths=.5, xticklabels=column_names, yticklabels=column_names)\n        plt.title('Correlation Matrix')\n        plt.show()\n\n        \n    def percent_formatter(self, x, pos):\n        return f'{x:.0f}%' \n    \n    \n    \nutils = Utils()\nconfig = dict()","metadata":{"execution":{"iopub.status.busy":"2024-10-07T09:48:26.589924Z","iopub.execute_input":"2024-10-07T09:48:26.591075Z","iopub.status.idle":"2024-10-07T09:48:26.604441Z","shell.execute_reply.started":"2024-10-07T09:48:26.591025Z","shell.execute_reply":"2024-10-07T09:48:26.602860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Basic EDA","metadata":{}},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:55:58.595151Z","iopub.execute_input":"2024-10-07T07:55:58.596250Z","iopub.status.idle":"2024-10-07T07:55:58.619625Z","shell.execute_reply.started":"2024-10-07T07:55:58.596196Z","shell.execute_reply":"2024-10-07T07:55:58.618291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:55:58.622041Z","iopub.execute_input":"2024-10-07T07:55:58.622593Z","iopub.status.idle":"2024-10-07T07:55:58.643569Z","shell.execute_reply.started":"2024-10-07T07:55:58.622548Z","shell.execute_reply":"2024-10-07T07:55:58.642159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"train : {train.shape}\")\nprint(f\"test : {test.shape}\") ","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:55:58.646088Z","iopub.execute_input":"2024-10-07T07:55:58.647027Z","iopub.status.idle":"2024-10-07T07:55:58.656453Z","shell.execute_reply.started":"2024-10-07T07:55:58.646961Z","shell.execute_reply":"2024-10-07T07:55:58.655177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"According to the documentation for the data : the actual test set is hidden. In this public version, we give some sample data in the correct format to help you author your solutions. The full test set comprises about 3800 instances.","metadata":{}},{"cell_type":"code","source":"# columns not available in test\nmissing_test_cols = []\nfor column in train.columns:\n    if column not in test.columns:\n        missing_test_cols.append(column)\n    else:\n        continue\nprint(f\"Missing columns from test : {missing_test_cols}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:55:58.658869Z","iopub.execute_input":"2024-10-07T07:55:58.659424Z","iopub.status.idle":"2024-10-07T07:55:58.670873Z","shell.execute_reply.started":"2024-10-07T07:55:58.659366Z","shell.execute_reply":"2024-10-07T07:55:58.669627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.drop(\"id\")\ntrain\n","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:55:58.676107Z","iopub.execute_input":"2024-10-07T07:55:58.677070Z","iopub.status.idle":"2024-10-07T07:55:58.702883Z","shell.execute_reply.started":"2024-10-07T07:55:58.677021Z","shell.execute_reply":"2024-10-07T07:55:58.701617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_values = train.select([pl.col(col).is_null().sum() for col in train.columns])\nnull_values","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:55:58.705420Z","iopub.execute_input":"2024-10-07T07:55:58.706484Z","iopub.status.idle":"2024-10-07T07:55:58.718578Z","shell.execute_reply.started":"2024-10-07T07:55:58.706418Z","shell.execute_reply":"2024-10-07T07:55:58.717297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Almost every column except the first three has null valuesThere are considerable null values in every column except id, basic season, age and sex.\n\nSamples with non null target values can be used for supervised learning.\n\nThe target sii is available exactly for those participants for whom we have results of the Parent-Child Internet Addiction Test (PCIAT).","metadata":{}},{"cell_type":"code","source":"non_null_values = train.select([pl.col(col).is_not_null().sum() for col in train.columns])\nnon_null_values","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:55:58.721560Z","iopub.execute_input":"2024-10-07T07:55:58.721977Z","iopub.status.idle":"2024-10-07T07:55:58.734435Z","shell.execute_reply.started":"2024-10-07T07:55:58.721934Z","shell.execute_reply":"2024-10-07T07:55:58.733036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n\n","metadata":{}},{"cell_type":"markdown","source":"## Null vs non null values","metadata":{}},{"cell_type":"markdown","source":"Checking the frequency of null vs non null values in the data","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 15))  # Width, Height in inches\n\nnon_null = non_null_values.to_numpy().flatten()\nnull = null_values.to_numpy().flatten()\n\n\nplt.barh(non_null_values.columns, non_null, color='lightblue', label=\"Non null values\")\nplt.barh(null_values.columns, null, color='pink', label=\"Null values\")\n\nplt.xlabel('Values')\nplt.title('Null Values vs the non null values')\nplt.legend()\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:55:58.739544Z","iopub.execute_input":"2024-10-07T07:55:58.740023Z","iopub.status.idle":"2024-10-07T07:56:00.343416Z","shell.execute_reply.started":"2024-10-07T07:55:58.739978Z","shell.execute_reply":"2024-10-07T07:56:00.342135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sample with null/non null targets\n","metadata":{}},{"cell_type":"markdown","source":"The target variable 'sii' is also null for some rows","metadata":{}},{"cell_type":"code","source":"# This sample of data can be used for supervised learning\nnon_null_target_features = train.filter(pl.col(\"sii\").is_not_null())\nnon_null_target_features","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:56:00.345940Z","iopub.execute_input":"2024-10-07T07:56:00.346740Z","iopub.status.idle":"2024-10-07T07:56:00.372847Z","shell.execute_reply.started":"2024-10-07T07:56:00.346646Z","shell.execute_reply":"2024-10-07T07:56:00.371637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This sample of data can be used for unsupervised learning\n\nnull_target_features = train.filter(pl.col(\"sii\").is_null())\nnull_target_features","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:56:00.374328Z","iopub.execute_input":"2024-10-07T07:56:00.374690Z","iopub.status.idle":"2024-10-07T07:56:00.398924Z","shell.execute_reply.started":"2024-10-07T07:56:00.374652Z","shell.execute_reply":"2024-10-07T07:56:00.397784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Seprating numerical and categorical samples\n","metadata":{}},{"cell_type":"code","source":"# seprating numerical and categorical columns:\n\nnumeric_features = train.select(pl.col(pl.Int64) | pl.col(pl.Float64))\nnumeric_features","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:56:00.401911Z","iopub.execute_input":"2024-10-07T07:56:00.402327Z","iopub.status.idle":"2024-10-07T07:56:00.421424Z","shell.execute_reply.started":"2024-10-07T07:56:00.402284Z","shell.execute_reply":"2024-10-07T07:56:00.420248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_features =  train.select(pl.col(pl.Utf8))\ncategorical_features\n","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:56:00.422958Z","iopub.execute_input":"2024-10-07T07:56:00.423443Z","iopub.status.idle":"2024-10-07T07:56:00.437196Z","shell.execute_reply.started":"2024-10-07T07:56:00.423402Z","shell.execute_reply":"2024-10-07T07:56:00.435900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Handling null and categorical values\nThere are significant null values in the data set. \n\n1. Since the data contains categorical data, we need to encode it first\n2. Then, using KNNImputer from scikit, will have to impute,","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\n\ncategories_np = train.select(categorical_features.columns).to_numpy()\n\nonehot_encoder = OneHotEncoder(sparse=False)  \ncategories_encoded = onehot_encoder.fit_transform(categories_np)\nencoded_columns = onehot_encoder.get_feature_names_out(categorical_features.columns)\ntrain_encoded = pl.DataFrame(categories_encoded, schema=encoded_columns.tolist())\ntrain_encoded\n\ntrain_encoded = pl.concat([train.drop(categorical_features.columns), train_encoded], how='horizontal')\ntrain_encoded","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:56:00.438708Z","iopub.execute_input":"2024-10-07T07:56:00.439170Z","iopub.status.idle":"2024-10-07T07:56:00.498141Z","shell.execute_reply.started":"2024-10-07T07:56:00.439097Z","shell.execute_reply":"2024-10-07T07:56:00.496925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.impute import KNNImputer\n\n\ndf_np = train_encoded.to_numpy()\nimputer = KNNImputer(n_neighbors=2)\ndf_imputed_np = imputer.fit_transform(df_np)\ntrain_imputed = pl.DataFrame(df_imputed_np, schema=train_encoded.columns)\ntrain_imputed","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:56:23.704418Z","iopub.execute_input":"2024-10-07T07:56:23.704943Z","iopub.status.idle":"2024-10-07T07:56:31.812470Z","shell.execute_reply.started":"2024-10-07T07:56:23.704892Z","shell.execute_reply":"2024-10-07T07:56:31.811178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Exploring some Metadata","metadata":{}},{"cell_type":"markdown","source":"The tabular data in train.csv and test.csv consists of measurements from various instruments, with field descriptions provided in data_dictionary.csv.","metadata":{}},{"cell_type":"code","source":"data_dict = pl.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')\ndata_dict.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:56:49.507760Z","iopub.execute_input":"2024-10-07T07:56:49.508238Z","iopub.status.idle":"2024-10-07T07:56:49.521679Z","shell.execute_reply.started":"2024-10-07T07:56:49.508192Z","shell.execute_reply":"2024-10-07T07:56:49.520527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"physical_measures = data_dict.filter(pl.col('Instrument') == 'Physical Measures')\ndemographics = data_dict.filter(pl.col('Instrument') == 'Demographics')\nfitness_gram_child = data_dict.filter(pl.col('Instrument') == 'FitnessGram Child')\nbiometric_impedance_analysis = data_dict.filter(pl.col('Instrument') == 'Bio-electric Impedance Analysis')\n\nfitness_gram_child","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:56:56.114299Z","iopub.execute_input":"2024-10-07T07:56:56.115421Z","iopub.status.idle":"2024-10-07T07:56:56.129934Z","shell.execute_reply.started":"2024-10-07T07:56:56.115371Z","shell.execute_reply":"2024-10-07T07:56:56.128454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Distribution\n","metadata":{}},{"cell_type":"markdown","source":"## Age distribution between boys and girls","metadata":{}},{"cell_type":"code","source":"colors = ['blue', 'green']\nlabels = ['boys', 'girls']\n\nfig, axs = plt.subplots(2, 1)\nfor i, sex in enumerate(train['Basic_Demos-Sex'].unique()):\n    age_counts = train.filter(pl.col('Basic_Demos-Sex') == sex)['Basic_Demos-Age'].value_counts()\n    axs[i].bar(age_counts['Basic_Demos-Age'], age_counts['count'], color=colors[i], label=labels[i])\n    axs[i].set_ylabel('count')\n    axs[i].legend()\n    plt.suptitle('Age Distribution')\n    axs[i].set_xlabel('years')\n\n\naxs[1].set_xlabel('years')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:57:00.712183Z","iopub.execute_input":"2024-10-07T07:57:00.713392Z","iopub.status.idle":"2024-10-07T07:57:01.186425Z","shell.execute_reply.started":"2024-10-07T07:57:00.713340Z","shell.execute_reply":"2024-10-07T07:57:01.185208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The ages range from 5 to 22. Number of boys is twice the number of girls. ","metadata":{}},{"cell_type":"markdown","source":"## Distribution of enrollment throughout the seasons.\n","metadata":{}},{"cell_type":"code","source":"from matplotlib.ticker import FuncFormatter\n\n\nenrollment_season = train.group_by('Basic_Demos-Enroll_Season').len()\ntotal_counts = len(train)\npercentage = (enrollment_season['len'] / total_counts) * 100\n\n\nplt.bar(enrollment_season['Basic_Demos-Enroll_Season'], percentage, color='lightblue')\nplt.gca().yaxis.set_major_formatter(FuncFormatter(utils.percent_formatter))\n\n\n\nplt.xlabel('Enrollment season')\nplt.ylabel('Count')\nplt.title('Bar Chart Example')\nplt.xticks()  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:57:03.601994Z","iopub.execute_input":"2024-10-07T07:57:03.602484Z","iopub.status.idle":"2024-10-07T07:57:03.860416Z","shell.execute_reply.started":"2024-10-07T07:57:03.602439Z","shell.execute_reply":"2024-10-07T07:57:03.859367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Enrollments seem farily distributed with highest in spring and lowest in fall","metadata":{}},{"cell_type":"code","source":"data_dict.head(20)","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:57:06.818753Z","iopub.execute_input":"2024-10-07T07:57:06.819235Z","iopub.status.idle":"2024-10-07T07:57:06.831697Z","shell.execute_reply.started":"2024-10-07T07:57:06.819188Z","shell.execute_reply":"2024-10-07T07:57:06.829075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature distribution for physical measures","metadata":{}},{"cell_type":"markdown","source":"checking the istrubution of physical features across the dataset:","metadata":{}},{"cell_type":"code","source":"total_columns=len( physical_measures['Field'])\nnum_cols=3\nnrows = (total_columns + num_cols - 1) // num_cols\n\n\nfig, axes = plt.subplots(nrows=nrows, ncols=num_cols, figsize=(15, 5 * nrows))\naxes = axes.flatten()\n\nfor i, col in enumerate(physical_measures['Field']):\n    sns.histplot(data=train, x=col, bins=30, kde=True, alpha=0.5, ax=axes[i])\n    axes[i].set_title(f'Count of {col}')\n    axes[i].set_xlabel(f'{col}')\n    axes[i].set_ylabel('Frequency')\n    \nfor j in range(i + 1, len(axes)):\n    fig.delaxes(axes[j])","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:57:16.518803Z","iopub.execute_input":"2024-10-07T07:57:16.519288Z","iopub.status.idle":"2024-10-07T07:57:19.479064Z","shell.execute_reply.started":"2024-10-07T07:57:16.519245Z","shell.execute_reply":"2024-10-07T07:57:19.477806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ps = train.select(\n\"Physical-BMI\",\n\"Physical-Height\",\n\"Physical-Weight\",\n\"Physical-Waist_Circumference\",\n\"Physical-Diastolic_BP\",\n\"Physical-HeartRate\",\n\"Physical-Systolic_BP\")\n\nps","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:57:21.346049Z","iopub.execute_input":"2024-10-07T07:57:21.347021Z","iopub.status.idle":"2024-10-07T07:57:21.357017Z","shell.execute_reply.started":"2024-10-07T07:57:21.346967Z","shell.execute_reply":"2024-10-07T07:57:21.355837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Creating boxplots for identifying outliers","metadata":{}},{"cell_type":"code","source":"total_columns=len( ps.columns)\nnum_cols=2\nnrows = (total_columns + num_cols - 1) // num_cols\n\n\nfig, axes = plt.subplots(nrows=nrows, ncols=num_cols, figsize=(20, 5 * nrows))\naxes = axes.flatten()\n\nfor i, col in enumerate(ps.columns):\n    sns.boxplot(data=train[col], ax=axes[i])\n    axes[i].set_title(f'Count of {col}')\n    axes[i].set_xlabel(f'{col}')\n    axes[i].set_ylabel('Frequency')\n    plt.xticks(rotation=45)\n    plt.xticks(fontsize=5)\n    \nfor j in range(i + 1, len(axes)):\n    fig.delaxes(axes[j])","metadata":{"execution":{"iopub.status.busy":"2024-10-07T07:57:24.130315Z","iopub.execute_input":"2024-10-07T07:57:24.131323Z","iopub.status.idle":"2024-10-07T08:01:11.369815Z","shell.execute_reply.started":"2024-10-07T07:57:24.131277Z","shell.execute_reply":"2024-10-07T08:01:11.368321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Box plots show some physical features to have considerable outliers : waist circumference , blood pressure and heart rate","metadata":{}},{"cell_type":"markdown","source":"### Analysing blood pressure data","metadata":{}},{"cell_type":"markdown","source":"From the box plots, min value of blood pressure is 0 which is obviously an outlier. So is the max value. Systolic and diastolic are very spread out. ","metadata":{}},{"cell_type":"code","source":"bp_age = train['Physical-Systolic_BP', 'Physical-Diastolic_BP', 'Basic_Demos-Age']\nconfig['plot_type'] = 'line_plot'    \ngraph_specs = {\n    \"fig_size\" : (10, 6),\n    \"x\"        : 'Basic_Demos-Age',\n    \"title\"    : \"Blood Pressure vs Age\",\n    \"x_label\"  : \"Age\",\n    \"legend\"   : \"Blood Pressure Type\"\n}\nutils.plot_bp(graph_specs, bp_age, config)","metadata":{"execution":{"iopub.status.busy":"2024-10-07T08:01:11.372052Z","iopub.execute_input":"2024-10-07T08:01:11.372514Z","iopub.status.idle":"2024-10-07T08:01:13.000979Z","shell.execute_reply.started":"2024-10-07T08:01:11.372469Z","shell.execute_reply":"2024-10-07T08:01:12.999658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"for understanding subplots see https://dev.to/thalesbruno/subplotting-with-matplotlib-and-seaborn-5ei8","metadata":{}},{"cell_type":"code","source":"grouped_data = train.group_by('sii')\ngrouped_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-07T08:04:59.638560Z","iopub.execute_input":"2024-10-07T08:04:59.639043Z","iopub.status.idle":"2024-10-07T08:04:59.672336Z","shell.execute_reply.started":"2024-10-07T08:04:59.638999Z","shell.execute_reply":"2024-10-07T08:04:59.671193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Comparing the physical measures across different levels of sii","metadata":{}},{"cell_type":"code","source":"train_pandas = train_imputed.to_pandas()\n\ngrouped_data = train_pandas.groupby('sii').mean().reset_index()\n\nbmi = 'Physical-BMI'\nheart_rate = 'Physical-HeartRate'\nsii = 'sii'\n\n### a. Box plots to compare distributions:\nsns.boxplot(x='sii', y=bmi, data=train_pandas)\nplt.title('BMI Distribution by SII')\nplt.show()\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-07T08:27:56.826330Z","iopub.execute_input":"2024-10-07T08:27:56.826865Z","iopub.status.idle":"2024-10-07T08:27:57.342251Z","shell.execute_reply.started":"2024-10-07T08:27:56.826820Z","shell.execute_reply":"2024-10-07T08:27:57.340961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nsns.boxplot(x='sii', y=heart_rate, data=train_pandas)\nplt.title('Heart Rate Distribution by SII')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-07T08:27:58.693638Z","iopub.execute_input":"2024-10-07T08:27:58.694086Z","iopub.status.idle":"2024-10-07T08:27:59.176295Z","shell.execute_reply.started":"2024-10-07T08:27:58.694043Z","shell.execute_reply":"2024-10-07T08:27:59.175035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### b. Bar plots to compare means:\nmeans = grouped_data[[sii, bmi, heart_rate]]\nmeans.set_index(sii).plot(kind='bar')\nplt.title('Mean BMI and Heart Rate by SII')\nplt.ylabel('Mean Value')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-07T08:28:02.309102Z","iopub.execute_input":"2024-10-07T08:28:02.309600Z","iopub.status.idle":"2024-10-07T08:28:02.797471Z","shell.execute_reply.started":"2024-10-07T08:28:02.309551Z","shell.execute_reply":"2024-10-07T08:28:02.796203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n### c. Scatter plots to visualize relationships:\nsns.scatterplot(x=train_pandas[sii], y=train_pandas[bmi])\nplt.title('BMI vs. SII')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-07T08:28:04.757000Z","iopub.execute_input":"2024-10-07T08:28:04.757492Z","iopub.status.idle":"2024-10-07T08:28:05.272199Z","shell.execute_reply.started":"2024-10-07T08:28:04.757443Z","shell.execute_reply":"2024-10-07T08:28:05.270928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(x=train_pandas[sii], y=train_pandas[heart_rate])\nplt.title('Heart Rate vs. SII')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-07T08:28:07.525644Z","iopub.execute_input":"2024-10-07T08:28:07.526732Z","iopub.status.idle":"2024-10-07T08:28:08.047626Z","shell.execute_reply.started":"2024-10-07T08:28:07.526684Z","shell.execute_reply":"2024-10-07T08:28:08.046266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"correlation_matrix  = train_imputed.select(\"Physical-BMI\",\n\"Physical-Height\",\n\"Physical-Weight\",\n\"Physical-Waist_Circumference\",\n\"Physical-Diastolic_BP\",\n\"Physical-HeartRate\",\n\"Physical-Systolic_BP\", \n\"sii\", \n\"Basic_Demos-Age\",\n\"Basic_Demos-Sex\").corr()\ncorrelation_matrix","metadata":{"execution":{"iopub.status.busy":"2024-10-07T08:52:30.367745Z","iopub.execute_input":"2024-10-07T08:52:30.368222Z","iopub.status.idle":"2024-10-07T08:52:30.381842Z","shell.execute_reply.started":"2024-10-07T08:52:30.368176Z","shell.execute_reply":"2024-10-07T08:52:30.380532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"utils.corr_matrix_heatmap(correlation_matrix)","metadata":{"execution":{"iopub.status.busy":"2024-10-07T09:36:54.657185Z","iopub.execute_input":"2024-10-07T09:36:54.657661Z","iopub.status.idle":"2024-10-07T09:36:55.622286Z","shell.execute_reply.started":"2024-10-07T09:36:54.657617Z","shell.execute_reply":"2024-10-07T09:36:55.620981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Physical heart and gender rate seem to have a low correlation with the sii","metadata":{}},{"cell_type":"markdown","source":"Using XGBoost regressor:","metadata":{}},{"cell_type":"code","source":"X = train_imputed.drop('sii')\ny = train_imputed['sii']\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\nxgb_reg = xgb.XGBRegressor(n_estimators=100, learning_rate=0.1, max_depth=3, random_state=42)\n\nxgb_reg.fit(X_train, y_train)\n\ny_pred = xgb_reg.predict(X_test)\n\nmse = mean_squared_error(y_test, y_pred)\nprint(\"Mean Squared Error:\", mse)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-07T09:13:57.574360Z","iopub.execute_input":"2024-10-07T09:13:57.574843Z","iopub.status.idle":"2024-10-07T09:13:58.112569Z","shell.execute_reply.started":"2024-10-07T09:13:57.574795Z","shell.execute_reply":"2024-10-07T09:13:58.111652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred= xgb_reg.predict(X_test)\npred","metadata":{"execution":{"iopub.status.busy":"2024-10-07T10:01:33.060314Z","iopub.execute_input":"2024-10-07T10:01:33.060913Z","iopub.status.idle":"2024-10-07T10:01:33.089098Z","shell.execute_reply.started":"2024-10-07T10:01:33.060851Z","shell.execute_reply":"2024-10-07T10:01:33.087843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n\n","metadata":{}}]}