{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:32:56.348659Z","iopub.execute_input":"2024-11-28T17:32:56.349109Z","iopub.status.idle":"2024-11-28T17:32:56.376661Z","shell.execute_reply.started":"2024-11-28T17:32:56.349046Z","shell.execute_reply":"2024-11-28T17:32:56.375473Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy import stats\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom IPython.display import display, HTML\nimport matplotlib.gridspec as gridspec\nfrom matplotlib.ticker import PercentFormatter\n\nimport statsmodels.api as sm\nfrom sklearn.preprocessing import PowerTransformer\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\ndef display_html(size=3, content=\"content\"):\n    display(HTML(f\"<h{size}>{content}</h{size}>\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:32:56.382846Z","iopub.execute_input":"2024-11-28T17:32:56.383240Z","iopub.status.idle":"2024-11-28T17:33:00.921422Z","shell.execute_reply.started":"2024-11-28T17:32:56.383207Z","shell.execute_reply":"2024-11-28T17:33:00.920287Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **1. Data Preview**","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ndata_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:00.923654Z","iopub.execute_input":"2024-11-28T17:33:00.924763Z","iopub.status.idle":"2024-11-28T17:33:01.027027Z","shell.execute_reply.started":"2024-11-28T17:33:00.924704Z","shell.execute_reply":"2024-11-28T17:33:01.025902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train data \ndisplay(train_data.head())\nprint('Shape of the train data: ', train_data.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:01.028508Z","iopub.execute_input":"2024-11-28T17:33:01.028986Z","iopub.status.idle":"2024-11-28T17:33:01.071144Z","shell.execute_reply.started":"2024-11-28T17:33:01.028940Z","shell.execute_reply":"2024-11-28T17:33:01.069929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display(train_data.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:01.074156Z","iopub.execute_input":"2024-11-28T17:33:01.074930Z","iopub.status.idle":"2024-11-28T17:33:01.081982Z","shell.execute_reply.started":"2024-11-28T17:33:01.074875Z","shell.execute_reply":"2024-11-28T17:33:01.080766Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# test data \ndisplay(test_data.head())\nprint('Shape of the test data: ', test_data.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:01.083304Z","iopub.execute_input":"2024-11-28T17:33:01.083659Z","iopub.status.idle":"2024-11-28T17:33:01.115149Z","shell.execute_reply.started":"2024-11-28T17:33:01.083628Z","shell.execute_reply":"2024-11-28T17:33:01.114067Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# data dict \ndisplay(data_dict.head())\nprint('Shape of the data dict: ', data_dict.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:01.116729Z","iopub.execute_input":"2024-11-28T17:33:01.117173Z","iopub.status.idle":"2024-11-28T17:33:01.132197Z","shell.execute_reply.started":"2024-11-28T17:33:01.117127Z","shell.execute_reply":"2024-11-28T17:33:01.131100Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_labels = ['None', 'Mild', 'Moderate', 'Severe']\nseason_dtype =  pd.CategoricalDtype(categories=['Spring', 'Summer', 'Autmon', 'Winter'], ordered=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:01.133837Z","iopub.execute_input":"2024-11-28T17:33:01.134625Z","iopub.status.idle":"2024-11-28T17:33:01.150604Z","shell.execute_reply.started":"2024-11-28T17:33:01.134569Z","shell.execute_reply":"2024-11-28T17:33:01.149307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = (\n    pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n    .assign(**{col: lambda df: train_data[col].astype(season_dtype) for col in train_data.filter(regex='Season').columns})\n)\n\ntest = (\n    pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n    .assign(**{col: lambda df: test_data[col].astype(season_dtype) for col in test_data.filter(regex='Season').columns})\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:01.152015Z","iopub.execute_input":"2024-11-28T17:33:01.152443Z","iopub.status.idle":"2024-11-28T17:33:01.230289Z","shell.execute_reply.started":"2024-11-28T17:33:01.152394Z","shell.execute_reply":"2024-11-28T17:33:01.229127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['Basic_Demos-Enroll_Season'].dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:01.231977Z","iopub.execute_input":"2024-11-28T17:33:01.232320Z","iopub.status.idle":"2024-11-28T17:33:01.239790Z","shell.execute_reply.started":"2024-11-28T17:33:01.232288Z","shell.execute_reply":"2024-11-28T17:33:01.238584Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Helper Functions","metadata":{}},{"cell_type":"code","source":"def calc_stats(data, columns):\n    if isinstance(columns, str):\n        columns = [columns]\n\n    stats = []\n    for col in columns:\n        if data[col].dtype == 'object' or data[col].dtype=='category':\n            counts = data[col].value_counts(dropna=False, sort=False)\n            percents=data[col].value_counts(dropna=False, normalize=True, sort=False)*100\n            formate = counts.astype(str)+\" (\" + percents.round(2).astype(str)+\"%)\"\n            stats_col = pd.DataFrame({'count (%)': formate})\n            stats.append(stats_col)\n        else:\n            stats_col = data[col].describe().to_frame().transpose()\n            stats_col['missing'] = data[col].isnull().sum()\n            stats_col.index.name = col\n            stats.append(stats_col)\n    return pd.concat(stats, axis=0)\n\n\n# univariate plots for numeric variable\ndef num_univar_plots(data, var, bins=10, figsize=(15, 7)):\n    display_html(3, f\"Univariate Analysis of {var}\")\n    display_html(content=\"\")\n    col = data.loc[:, var].copy()\n\n    fig, ax = plt.subplots(2, 3, figsize=figsize)\n    ax = ax.ravel()\n\n    # histogram\n    sns.histplot(data=data, x=var, bins=bins, ax=ax[0], kde=True, color='#1973bd')\n    sns.rugplot(data=data, x=var, ax=ax[0], color='black')\n    ax[0].set(title='Histogram')\n\n    # kdeplot\n    sns.ecdfplot(data=data, x=var, ax=ax[1], color='red')\n    ax[1].set(title='CDF')\n\n    # powertransformer\n    data = data.assign(**{\n        f\"{var}_pwt\":(\n            PowerTransformer(method='yeo-johnson')\n            .fit_transform(data.loc[:, [var]]).ravel()\n        )\n    })\n    skew = data[f\"{var}_pwt\"].skew()\n    sns.kdeplot(data=data, x=f\"{var}_pwt\",fill=True, ax=ax[2], color='green')\n    sns.rugplot(data=data, x=f\"{var}_pwt\", ax=ax[2], color='black')\n    ax[2].set(title='Power Transformed')\n\n    # box-plot\n    sns.boxplot(data=data, x=var, color='orange', ax=ax[3])\n    ax[3].set(title='Boxplot')\n\n    # violin plot\n    sns.violinplot(data=data, x=var, ax=ax[4], color='orange')\n    ax[4].set(title='violin plot')\n\n    # qq plot \n    sm.qqplot(col.dropna(), line='45', fit=True, ax=ax[5])\n    ax[5].set(title='QQ Plot')\n\n    plt.tight_layout()\n    plt.show()\n\n# Univar plots for categorical variable \ndef cat_univar_plots(data,var, order=None, figsize=(15,5)):\n    display_html(2, f\"Univariate Analysis of {var}\")\n    display_html(content=\"\")\n\n    fig, axes = plt.subplots(1, 2, figsize=figsize)\n    ax = axes.ravel()\n\n    counts = data[var].value_counts()\n    colors = [tuple(np.random.choice(256, size=3)/255) for _ in range(len(counts))]\n\n    barplot = ax[0].bar(x=range(len(counts)),height=counts.values, tick_label=counts.index, color=colors, edgecolor='black', alpha=0.7)\n    ax[0].bar_label(barplot, color='black')\n    ax[0].set(title='Bar Chart', xlabel='Categories', ylabel='Count')\n    ax[0].set_xticklabels(ax[0].get_xticklabels(), rotation=45, ha='right')\n\n    ax1=ax[1]\n    ax1.set_title('Pie Chart')\n    pie = ax1.pie(counts.values, autopct='%1.2f%%', labels=counts.index, colors=colors, wedgeprops=dict(alpha=0.7, edgecolor='black'))\n    ax1.legend(loc='upper left', bbox_to_anchor=(1.02, 1), title='Categories', title_fontproperties=dict(weight=\"bold\", size=10))\n    plt.setp(pie[2],weight=\"bold\",color=\"white\")\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:01.244367Z","iopub.execute_input":"2024-11-28T17:33:01.244875Z","iopub.status.idle":"2024-11-28T17:33:01.265931Z","shell.execute_reply.started":"2024-11-28T17:33:01.244794Z","shell.execute_reply":"2024-11-28T17:33:01.264701Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['sii'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:01.267401Z","iopub.execute_input":"2024-11-28T17:33:01.267743Z","iopub.status.idle":"2024-11-28T17:33:01.291434Z","shell.execute_reply.started":"2024-11-28T17:33:01.267710Z","shell.execute_reply":"2024-11-28T17:33:01.290268Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **2. Missing Value**","metadata":{}},{"cell_type":"code","source":"missing_count = train.isnull().sum().reset_index()\nmissing_count.columns = ['feature', 'null_count']\nmissing_count = missing_count.sort_values('null_count', ascending=False)\nmissing_count['null_ratio'] = missing_count['null_count'] / len(train)\n\nplt.figure(figsize=(6, 15))\nplt.title('Missing values over the whole training dataset')\n\n# Plotting the missing values\nplt.barh(np.arange(len(missing_count)), missing_count['null_ratio'], color='coral', label='missing')\nplt.barh(np.arange(len(missing_count)), \n         1 - missing_count['null_ratio'],\n         left=missing_count['null_ratio'],\n         color='darkseagreen', label='available')\n\nplt.yticks(np.arange(len(missing_count)), missing_count['feature'])\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0, 1)\nplt.legend()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:01.293223Z","iopub.execute_input":"2024-11-28T17:33:01.294026Z","iopub.status.idle":"2024-11-28T17:33:02.484687Z","shell.execute_reply.started":"2024-11-28T17:33:01.293988Z","shell.execute_reply":"2024-11-28T17:33:02.483395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# filter the dataframe to keep only rows where 'ssi' is not null\nsupervised_usable = train[train['sii'].notnull()]\n\n# count of missing value where there are all target value present\nmissing_count = supervised_usable.isnull().sum().reset_index()\nmissing_count.columns = ['feature', 'null_count']\nmissing_count = missing_count.sort_values('null_count', ascending=False)\nmissing_count['null_ratio'] = missing_count['null_count']/len(supervised_usable)\n\nplt.figure(figsize=(6,20))\nplt.title(f'Missing values over the {len(supervised_usable)} sample which have not null target')\nplt.barh(np.arange(len(missing_count)), missing_count['null_ratio'], color='coral', label='missing')\nplt.barh(np.arange(len(missing_count)), 1-missing_count['null_ratio'],left=missing_count['null_ratio'], color='darkseagreen', label='available')\nplt.yticks(np.arange(len(missing_count)), missing_count['feature'])\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0, 1)\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:02.486242Z","iopub.execute_input":"2024-11-28T17:33:02.486616Z","iopub.status.idle":"2024-11-28T17:33:03.523004Z","shell.execute_reply.started":"2024-11-28T17:33:02.486583Z","shell.execute_reply":"2024-11-28T17:33:03.521776Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- The target is available for those participants for whom we have results of the Parent-Child Internet Addiction Test(PCIAT), and it is function of the PCIAT total score","metadata":{}},{"cell_type":"code","source":"# Group by 'sii' column and aggregate to get min, max, and count for 'PCIAT-PCIAT_Total'\nresult = (\n    train.groupby('sii')['PCIAT-PCIAT_Total']\n    .agg([\n        ('PCIAT-PCIAT_Total min', 'min'),\n        ('PCIAT-PCIAT_Total max', 'max'),\n        ('count', 'size')\n    ])\n    .reset_index()\n    .sort_values(by='sii')\n)\nresult","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:03.524189Z","iopub.execute_input":"2024-11-28T17:33:03.524521Z","iopub.status.idle":"2024-11-28T17:33:03.549986Z","shell.execute_reply.started":"2024-11-28T17:33:03.524477Z","shell.execute_reply":"2024-11-28T17:33:03.548714Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Test dataset doesn't have any PCIAT columns (otherwise prediction would be trivial).\n\n**Insight:**\n1. We should focus on predicting the target from all other features except the PCIAT results.\n2. We know the target only present for approximately for two third samples. The samples without can perhaps be used for semi-supervised learning.\n3. We know `sii` is derived from `PCIAT-PCIAT_Total`. then we predict `PCIAT-PCIAT_Total` and then transform this prediction to a `sii` prediction for submission. As `PCIAT-PCIAT_Total` is more granular and informative then `sii`, training to predict `PCIAT-PCIAT_Total` has the potential to produce a better model. ","metadata":{}},{"cell_type":"code","source":"data_dict[data_dict['Field']=='PCIAT-PCIAT_Total']['Value Labels'].iloc[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:03.551676Z","iopub.execute_input":"2024-11-28T17:33:03.552264Z","iopub.status.idle":"2024-11-28T17:33:03.564567Z","shell.execute_reply.started":"2024-11-28T17:33:03.552226Z","shell.execute_reply":"2024-11-28T17:33:03.563290Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_cols = set(train.columns)\ntest_cols = set(test.columns)\ncols_not_in_test = sorted(list(train_cols-test_cols))\ntrain_with_sii = train[train['sii'].notna()][cols_not_in_test]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:03.566202Z","iopub.execute_input":"2024-11-28T17:33:03.566640Z","iopub.status.idle":"2024-11-28T17:33:03.581421Z","shell.execute_reply.started":"2024-11-28T17:33:03.566603Z","shell.execute_reply":"2024-11-28T17:33:03.580287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ntrain_with_sii[train_with_sii.isna().sum(axis=1)>1].sample(5).style.applymap(\n    lambda x: 'background-color: #FFC0CB' if pd.isna(x) else ''\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:03.583087Z","iopub.execute_input":"2024-11-28T17:33:03.583437Z","iopub.status.idle":"2024-11-28T17:33:03.682639Z","shell.execute_reply.started":"2024-11-28T17:33:03.583404Z","shell.execute_reply":"2024-11-28T17:33:03.681539Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- We know that the PCIAT-PCIAT_Total is the sum of PCIAT-PCIAT_01 to PCIAT-PCIAT_20 and it have the value b/w 0 to 5.\n- From row 2 and 3 where atleast one PCIAT question is not responded by child. pciat total for row 2 is 27 with two missing value. but this is incorrect becoze the maximum sum of all possible missing value is `5+5=10` then pciat total would be 37 and in that case `SII` should be 1 means we have two possible target for this one is 0 and other is 1.\n- finally calucating PCIAT total using non-na value is not correct. thats why `SII` column have some issue. ","metadata":{}},{"cell_type":"markdown","source":"- Calucation of `SII` value based on `PCIAT_total` and in that case our final `SII` value maintain all the max and min value of `pciat-01` to `pciat-20`","metadata":{}},{"cell_type":"code","source":"pciat_cols = [f'PCIAT-PCIAT_{i+1:02d}' for i in np.arange(20)]\n\ndef new_SII(row):\n    if pd.isna(row['PCIAT-PCIAT_Total']):\n        return np.nan\n    max_value = row['PCIAT-PCIAT_Total']+row[pciat_cols].isna().sum() * 5\n\n    if row['PCIAT-PCIAT_Total']<=30 and max_value<=30:\n        return 0\n    elif 30<row['PCIAT-PCIAT_Total']<=49 and max_value<=49:\n        return 1\n    elif 49<row['PCIAT-PCIAT_Total']<=79 and max_value<=79:\n        return 2\n    elif 79<row['PCIAT-PCIAT_Total']<=93 and max_value<=93:\n        return 3\n    return np.nan\n\ntrain['new_sii'] = train.apply(new_SII, axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:03.684031Z","iopub.execute_input":"2024-11-28T17:33:03.684493Z","iopub.status.idle":"2024-11-28T17:33:05.203133Z","shell.execute_reply.started":"2024-11-28T17:33:03.684458Z","shell.execute_reply":"2024-11-28T17:33:05.201941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:05.204422Z","iopub.execute_input":"2024-11-28T17:33:05.204761Z","iopub.status.idle":"2024-11-28T17:33:05.236115Z","shell.execute_reply.started":"2024-11-28T17:33:05.204729Z","shell.execute_reply":"2024-11-28T17:33:05.234962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mismatch_rows = train[\n    (train['new_sii'] != train['sii']) & train['sii'].notna()\n]\n\nmismatch_rows[pciat_cols + [\n    'PCIAT-PCIAT_Total', 'sii', 'new_sii'\n]].style.applymap(\n    lambda x: 'background-color: #FFC0CB' if pd.isna(x) else ''\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:05.237218Z","iopub.execute_input":"2024-11-28T17:33:05.237526Z","iopub.status.idle":"2024-11-28T17:33:05.263006Z","shell.execute_reply.started":"2024-11-28T17:33:05.237495Z","shell.execute_reply":"2024-11-28T17:33:05.261915Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['sii'] = train['new_sii']\n\ntrain['complete_resp_total'] = train['PCIAT-PCIAT_Total'].where(train[pciat_cols].notna().all(axis=1), np.nan)\nsii_map = {0: '0 (None)', 1: '1 (Mild)', 2: '2 (Moderate)', 3: '3 (Severe)'}\ntrain['sii'] = train['sii'].map(sii_map)\ntrain['sii'] = train['sii'].fillna('Missing')\n\nlabel_order = ['Missing', '0 (None)', '1 (Mild)', '2 (Moderate)', '3 (Severe)']\ntrain['sii'] = pd.Categorical(train['sii'], categories=label_order, ordered=True)\ntrain.drop(columns='new_sii', inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:05.264811Z","iopub.execute_input":"2024-11-28T17:33:05.265219Z","iopub.status.idle":"2024-11-28T17:33:05.281642Z","shell.execute_reply.started":"2024-11-28T17:33:05.265185Z","shell.execute_reply":"2024-11-28T17:33:05.280349Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['sii'] = train['sii'].fillna('Missing')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:05.283144Z","iopub.execute_input":"2024-11-28T17:33:05.283469Z","iopub.status.idle":"2024-11-28T17:33:05.296047Z","shell.execute_reply.started":"2024-11-28T17:33:05.283438Z","shell.execute_reply":"2024-11-28T17:33:05.294753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cat_univar_plots(train, 'sii')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:05.297124Z","iopub.execute_input":"2024-11-28T17:33:05.297442Z","iopub.status.idle":"2024-11-28T17:33:05.309665Z","shell.execute_reply.started":"2024-11-28T17:33:05.297411Z","shell.execute_reply":"2024-11-28T17:33:05.308594Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Note : Apparently 40% of the participants were not affected by internet use, 31% were not Known, and only 10% participants were moderately get effected by internet use.","metadata":{"execution":{"iopub.status.busy":"2024-11-27T11:19:02.200254Z","iopub.execute_input":"2024-11-27T11:19:02.200924Z","iopub.status.idle":"2024-11-27T11:19:02.207996Z","shell.execute_reply.started":"2024-11-27T11:19:02.200875Z","shell.execute_reply":"2024-11-27T11:19:02.206475Z"}}},{"cell_type":"code","source":"train['sii'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:05.311047Z","iopub.execute_input":"2024-11-28T17:33:05.311354Z","iopub.status.idle":"2024-11-28T17:33:05.329964Z","shell.execute_reply.started":"2024-11-28T17:33:05.311326Z","shell.execute_reply":"2024-11-28T17:33:05.328654Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"assert train['Basic_Demos-Age'].isna().sum()==0\nassert train['Basic_Demos-Sex'].isna().sum()==0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:05.331686Z","iopub.execute_input":"2024-11-28T17:33:05.332042Z","iopub.status.idle":"2024-11-28T17:33:05.344128Z","shell.execute_reply.started":"2024-11-28T17:33:05.332009Z","shell.execute_reply":"2024-11-28T17:33:05.342934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:05.345520Z","iopub.execute_input":"2024-11-28T17:33:05.345895Z","iopub.status.idle":"2024-11-28T17:33:05.384159Z","shell.execute_reply.started":"2024-11-28T17:33:05.345833Z","shell.execute_reply":"2024-11-28T17:33:05.380800Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Demographics**","metadata":{}},{"cell_type":"code","source":"num_univar_plots(train, 'Basic_Demos-Age')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:05.385744Z","iopub.execute_input":"2024-11-28T17:33:05.387081Z","iopub.status.idle":"2024-11-28T17:33:07.001640Z","shell.execute_reply.started":"2024-11-28T17:33:05.387021Z","shell.execute_reply":"2024-11-28T17:33:07.000420Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['Age Group'] = pd.cut(\n    train['Basic_Demos-Age'],\n    bins=[4, 12, 18, 22],\n    labels=['Children (5-12)', 'Adolescents (13-18)', 'Adults (19-22)']\n)\ncalc_stats(train, 'Age Group')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:07.011129Z","iopub.execute_input":"2024-11-28T17:33:07.011546Z","iopub.status.idle":"2024-11-28T17:33:07.031449Z","shell.execute_reply.started":"2024-11-28T17:33:07.011509Z","shell.execute_reply":"2024-11-28T17:33:07.030219Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sex_map = {0: 'Male', 1: 'Female'}\ntrain['Basic_Demos-Sex'] = train['Basic_Demos-Sex'].map(sex_map)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:07.032850Z","iopub.execute_input":"2024-11-28T17:33:07.033225Z","iopub.status.idle":"2024-11-28T17:33:07.040561Z","shell.execute_reply.started":"2024-11-28T17:33:07.033191Z","shell.execute_reply":"2024-11-28T17:33:07.039207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calc_stats(train, 'Basic_Demos-Sex')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:07.042295Z","iopub.execute_input":"2024-11-28T17:33:07.042776Z","iopub.status.idle":"2024-11-28T17:33:07.058405Z","shell.execute_reply.started":"2024-11-28T17:33:07.042725Z","shell.execute_reply":"2024-11-28T17:33:07.057256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 4, figsize=(19, 5))\nax = axes.ravel()\n# SII by Age \nsns.boxplot(data=train, x='sii', y='Basic_Demos-Age', ax=ax[0], palette='Set2')\nax[0].set_title('SII by Age')\nax[0].set_ylabel('Age')\nax[0].set_xlabel('SII')\n\n# voiline plot\nsns.violinplot(data=train, x='sii', y='Basic_Demos-Age', ax=ax[1], palette='Set2')\nax[1].set_title('SII by Age')\nax[1].set_ylabel('Age')\nax[1].set_xlabel('SII')\n\n# complete PCIAT-total response by different Age Group\nsns.boxplot(data=train, x='Age Group', y='complete_resp_total', ax=ax[2], palette='Set2')\nax[2].set_title('Complete PCIAT responses by Different Age Group')\nax[2].set_ylabel('PCIAT Total for complete responses')\nax[2].set_xlabel('Age Group')\n\n# PCIAT-PCIAT_Total for different sex\nsns.histplot(data=train, hue='Basic_Demos-Sex', x='complete_resp_total', ax=ax[3], palette='Set2', multiple='stack')\nax[3].set_title('PCIAT_Total Distribution by Sex')\nax[3].set_xlabel('PCIAT_Total for complete response')\nax[3].set_ylabel('Frequency')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:07.059777Z","iopub.execute_input":"2024-11-28T17:33:07.060139Z","iopub.status.idle":"2024-11-28T17:33:08.239926Z","shell.execute_reply.started":"2024-11-28T17:33:07.060105Z","shell.execute_reply":"2024-11-28T17:33:08.238727Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- From the multibar plot we look the freq of pciat_total complete response for male is approximately double the female. and its only due the double amount of data present for male as compare to female. and both distribution have nearly same pattern but subtle.\n- from the box plot and violinplot we figure out only older age group people have severe and moderate impact of internet use.\n- from violinplt we get most of the data are distributed in the lower age group mainly 5-10.\n","metadata":{}},{"cell_type":"code","source":"stats = train.groupby(['Age Group', 'sii']).size().unstack(fill_value=0)\n\nfig, ax = plt.subplots(1, len(stats), figsize=(19, 5))\n\nfor i, age_group in enumerate(stats.index):\n    group_counts = stats.loc[age_group]/stats.loc[age_group].sum()\n    ax[i].pie(group_counts, labels=group_counts.index, autopct='%1.1f%%', startangle=90,labeldistance=1.06, pctdistance=0.8, colors=sns.color_palette('Set2'))\n    ax[i].set_title(f'SII distribution for {age_group}')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:08.241449Z","iopub.execute_input":"2024-11-28T17:33:08.241831Z","iopub.status.idle":"2024-11-28T17:33:08.754641Z","shell.execute_reply.started":"2024-11-28T17:33:08.241796Z","shell.execute_reply":"2024-11-28T17:33:08.753457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = train.groupby(['Age Group', 'sii']).size().unstack(fill_value=0)\nstats","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:08.756073Z","iopub.execute_input":"2024-11-28T17:33:08.756419Z","iopub.status.idle":"2024-11-28T17:33:08.775264Z","shell.execute_reply.started":"2024-11-28T17:33:08.756382Z","shell.execute_reply":"2024-11-28T17:33:08.773984Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- SII distribution for children and adult is similar nearly 80% of the child are either not being impacted by internet use. but from the above table we get the data for Adolescents and Adults are very smaller than Children. ","metadata":{"execution":{"iopub.status.busy":"2024-11-27T11:21:00.549902Z","iopub.execute_input":"2024-11-27T11:21:00.550271Z","iopub.status.idle":"2024-11-27T11:21:00.556883Z","shell.execute_reply.started":"2024-11-27T11:21:00.550237Z","shell.execute_reply":"2024-11-27T11:21:00.555458Z"}}},{"cell_type":"markdown","source":"## **PreInt_EduHx (Internet use)**","metadata":{}},{"cell_type":"code","source":"calc_stats(train, 'PreInt_EduHx-Season')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:08.776759Z","iopub.execute_input":"2024-11-28T17:33:08.777301Z","iopub.status.idle":"2024-11-28T17:33:08.793095Z","shell.execute_reply.started":"2024-11-28T17:33:08.777240Z","shell.execute_reply":"2024-11-28T17:33:08.791766Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# minimum and max age of children that they use internet\ndata = train[train['PreInt_EduHx-computerinternet_hoursday'].notna()]\nage_range = data['Basic_Demos-Age']\nprint(f'Age range for participants with measured PreInt_EduHx-computerinternet_hoursday : {age_range.min()}-{age_range.max()}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:08.794496Z","iopub.execute_input":"2024-11-28T17:33:08.794929Z","iopub.status.idle":"2024-11-28T17:33:08.804882Z","shell.execute_reply.started":"2024-11-28T17:33:08.794845Z","shell.execute_reply":"2024-11-28T17:33:08.803678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['PreInt_EduHx-computerinternet_hoursday'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:08.806387Z","iopub.execute_input":"2024-11-28T17:33:08.807248Z","iopub.status.idle":"2024-11-28T17:33:08.818353Z","shell.execute_reply.started":"2024-11-28T17:33:08.807213Z","shell.execute_reply":"2024-11-28T17:33:08.816968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hrs_map = {0: '<1hr/day', 1: '~1hr/day', 2: '~2hrs/day', 3: '>=3hrs/day'}\ntrain['encoded_internet_use'] = train['PreInt_EduHx-computerinternet_hoursday'].map(hrs_map).fillna('Missing')\nhrs_order = ['Missing', '<1hr/day', '~1hr/day', '~2hrs/day', '>=3hrs/day']\ntrain['encoded_internet_use'] = pd.Categorical(train['encoded_internet_use'], categories=hrs_order, ordered=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:08.819948Z","iopub.execute_input":"2024-11-28T17:33:08.820339Z","iopub.status.idle":"2024-11-28T17:33:08.833115Z","shell.execute_reply.started":"2024-11-28T17:33:08.820293Z","shell.execute_reply":"2024-11-28T17:33:08.831850Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 4, figsize=(19, 5))\nax = axes.ravel()\n\n# PCIAT-PCIAT_Total for different sex\nsns.countplot(data=train, x='encoded_internet_use', ax=ax[0], palette='Set2')\nax[0].set_title('Distribution of Internet Use (In Hours)')\nax[0].set_xlabel('Internet Usage (In Hours)')\nax[0].set_ylabel('Frequency')\n\ntotal = len(train['encoded_internet_use'])\nfor p in ax[0].patches:\n    count = int(p.get_height())\n    percentage = '{:.1f}%'.format(100 * count / total)\n    ax[0].annotate(f'{count} ({percentage})', (p.get_x() + p.get_width() / 2., p.get_height()), \n                 ha='center', va='baseline', fontsize=10, color='black', xytext=(0, 5), \n                 textcoords='offset points')\n\n# SII by Age \nsns.boxplot(data=train, x='encoded_internet_use', y='Basic_Demos-Age', ax=ax[1], palette='Set2')\nax[1].set_title('Internet Usage by Age')\nax[1].set_ylabel('Age')\nax[1].set_xlabel('Internet Usage(In Hours)')\n\n# voiline plot\nsns.violinplot(data=train, x='encoded_internet_use', y='Basic_Demos-Age', ax=ax[2], palette='Set2')\nax[2].set_title('Internet Usage by Age')\nax[2].set_ylabel('Age')\nax[2].set_xlabel('Internet Usage(In Hours)')\n\n# complete PCIAT-total response by different Age Group\nsns.boxplot(data=train, x='Age Group', y='PreInt_EduHx-computerinternet_hoursday', ax=ax[3], palette='Set2')\nax[3].set_title('Internet Usage by Different Age Group')\nax[3].set_ylabel('Internet Usage Per Day (In numbers)')\nax[3].set_xlabel('Age Group')\n\n\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:08.834761Z","iopub.execute_input":"2024-11-28T17:33:08.835125Z","iopub.status.idle":"2024-11-28T17:33:09.884644Z","shell.execute_reply.started":"2024-11-28T17:33:08.835090Z","shell.execute_reply":"2024-11-28T17:33:09.883534Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- As like Previous box plot, Older age group usage maximum hours of internet per day\n- ~17 % of data are missing and 38 % people usage internet 1hr and less than that","metadata":{}},{"cell_type":"code","source":"stats = train.groupby(['Age Group', 'encoded_internet_use']).size().unstack(fill_value=0)\nfig, ax = plt.subplots(1, len(stats), figsize=(19, 5))\n\nfor i, age_grp in enumerate(stats.index):\n    grp_counts = stats.loc[age_grp]/stats.loc[age_grp].sum()\n    ax[i].pie(grp_counts, labels=grp_counts.index, autopct='%1.1f%%', startangle=90, labeldistance=1.06, colors=sns.color_palette('Set2'))\n    ax[i].set_title(f'Distribution of Internet use in Hours by {age_grp}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:09.886003Z","iopub.execute_input":"2024-11-28T17:33:09.886325Z","iopub.status.idle":"2024-11-28T17:33:10.257356Z","shell.execute_reply.started":"2024-11-28T17:33:09.886293Z","shell.execute_reply":"2024-11-28T17:33:10.255921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = train.groupby(['Age Group', 'encoded_internet_use']).size().unstack(fill_value=0)\nstats","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:10.259117Z","iopub.execute_input":"2024-11-28T17:33:10.260118Z","iopub.status.idle":"2024-11-28T17:33:10.282744Z","shell.execute_reply.started":"2024-11-28T17:33:10.259815Z","shell.execute_reply":"2024-11-28T17:33:10.281513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = train.groupby(['Basic_Demos-Sex', 'encoded_internet_use']\n).size().unstack(fill_value=0)\nstats_prop = stats.div(stats.sum(axis=1), axis=0) * 100\n\nstats = stats.astype(str) +' (' + stats_prop.round(1).astype(str) + '%)'\nstats","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:10.284166Z","iopub.execute_input":"2024-11-28T17:33:10.284487Z","iopub.status.idle":"2024-11-28T17:33:10.306230Z","shell.execute_reply.started":"2024-11-28T17:33:10.284442Z","shell.execute_reply":"2024-11-28T17:33:10.304671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# total percentage of children the gives more than 3hrs time a day on internet\ntrain_non_na = train[train['PreInt_EduHx-computerinternet_hoursday'].notna()]\nrows = (train_non_na['PreInt_EduHx-computerinternet_hoursday']==3).sum()\npercentage = (f'{rows/len(train_non_na)*100:.2f} %')\nprint(f'Internet use 3hrs and more: {percentage} ')\nrows = (train_non_na['PreInt_EduHx-computerinternet_hoursday']<1).sum()\npercentage = (f'{(rows/len(train_non_na))*100:.2f}')\nprint(f'Internet use <1hrs per day : {percentage}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:10.307432Z","iopub.execute_input":"2024-11-28T17:33:10.307752Z","iopub.status.idle":"2024-11-28T17:33:10.319156Z","shell.execute_reply.started":"2024-11-28T17:33:10.307720Z","shell.execute_reply":"2024-11-28T17:33:10.317977Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Internet by SII","metadata":{}},{"cell_type":"code","source":"sii_report = train[train['sii']!='Missing']\nsii_report.loc[: , 'sii'] = sii_report['sii'].cat.remove_unused_categories()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:10.320636Z","iopub.execute_input":"2024-11-28T17:33:10.321059Z","iopub.status.idle":"2024-11-28T17:33:10.331765Z","shell.execute_reply.started":"2024-11-28T17:33:10.321017Z","shell.execute_reply":"2024-11-28T17:33:10.330594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = sii_report.groupby(['encoded_internet_use', 'sii']).size().unstack()\nstats_prop = stats.div(stats.sum(axis=1), axis=0)*100\n\nstats = stats.astype(str)+'(' + stats_prop.round(1).astype(str)+'%)'\nstats","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:10.333883Z","iopub.execute_input":"2024-11-28T17:33:10.334380Z","iopub.status.idle":"2024-11-28T17:33:10.358381Z","shell.execute_reply.started":"2024-11-28T17:33:10.334329Z","shell.execute_reply":"2024-11-28T17:33:10.357316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig = plt.figure(figsize=(12,10))\ngs = fig.add_gridspec(2, 2, height_ratios=[1, 1.5])\n\n# SII vs Hours spend ON Internet\nax1 = fig.add_subplot(gs[0, 0])\nsns.boxplot(data=sii_report, x='sii', y='PreInt_EduHx-computerinternet_hoursday', ax=ax1, palette='Set3')\nax1.set_title('SII vs Hours spend On Internet')\nax1.set_xlabel('SII')\nax1.set_ylabel('Hours Spend On Internet Per Day')\n\n# PCIAT complete response after using Internet(In Hours) per Day\nax2 = fig.add_subplot(gs[0,1])\nsns.boxplot(data=sii_report, x='encoded_internet_use', y='complete_resp_total', palette='Set2')\nax2.set_title('PCIAT_Total by Hours of Internet use')\nax2.set_xlabel('Internet use(In Hours) per day')\nax2.set_ylabel('PCIAT_Total for complete PCIAT responses')\n\n# SII vs Hours of Internet usage by Age Group \nax3 = fig.add_subplot(gs[1, :])\nsns.boxplot(data=sii_report, x='encoded_internet_use', y='complete_resp_total', palette='Set3', hue='Age Group')\nax3.set_title('PCIAT complete Response after using hours of Internet by Age Group')\nax3.set_ylabel('PCIAT complete Response')\nax3.set_xlabel('Internet use (In Hours) per day')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:10.359901Z","iopub.execute_input":"2024-11-28T17:33:10.360259Z","iopub.status.idle":"2024-11-28T17:33:11.459508Z","shell.execute_reply.started":"2024-11-28T17:33:10.360226Z","shell.execute_reply":"2024-11-28T17:33:11.458243Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- From 2nd plot, as the hours of internet usage increases per day PCIAT_Total response also increase. `<1hr/day`, `~1hr/day`, `~2hrs/day` have some potentail outliers. and also have broader spread of PCIAT complete response for `~2hr/day`, `>=3hrs/day` indicating that not everyone with high internet usage exhibits problematic behaivious. likely missing data have lower `PCIAT complete response`, means that they not use internet or less invovlement.\n- Children exhibit broader spread of PCIAT completeness response as compare to adults and adolescents.suggesting they are more sensitive to overuse of internet.`adolescents` pciat response increases steadly but they maintain moderate levels compared to children.`adults` have lowest pciat reponse even they use internet `>=3hrs/day`.","metadata":{}},{"cell_type":"code","source":"stats = sii_report.groupby(\n    ['sii', 'encoded_internet_use']\n).size().unstack(fill_value=0)\nfig, axes = plt.subplots(1, len(stats), figsize=(18, 5))\n\nfor i, sii_group in enumerate(stats.index):\n    group_counts = stats.loc[sii_group] / stats.loc[sii_group].sum()\n    axes[i].pie(\n        group_counts, labels=group_counts.index, autopct='%1.1f%%',\n        startangle=90, colors=sns.color_palette(\"Set2\"), labeldistance=1.1,pctdistance=0.9\n    )\n    axes[i].set_title(f'Hours of using computer/internet\\n for SII = {sii_group}')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:11.460836Z","iopub.execute_input":"2024-11-28T17:33:11.461205Z","iopub.status.idle":"2024-11-28T17:33:11.925874Z","shell.execute_reply.started":"2024-11-28T17:33:11.461171Z","shell.execute_reply":"2024-11-28T17:33:11.924582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = sii_report.groupby(['sii', 'encoded_internet_use']).size().unstack(fill_value=0)\nstats_prop = stats.div(stats.sum(axis=1), axis=0)*100\nstats =stats.astype(str) + \"(\" + stats_prop.round(1).astype(str)+\"%)\"\nstats","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:11.927566Z","iopub.execute_input":"2024-11-28T17:33:11.928023Z","iopub.status.idle":"2024-11-28T17:33:11.952457Z","shell.execute_reply.started":"2024-11-28T17:33:11.927972Z","shell.execute_reply":"2024-11-28T17:33:11.951397Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- From the pie chart, total 83 people they use internet `<1hrs` have internet problemetic issues of all ages. because they have sii score higher(20.7% sii socre is moderate while 14.7% have severe imacat","metadata":{}},{"cell_type":"markdown","source":"<div style=\"border: 2px solid #c9c9c9; padding: 15px; border-radius: 5px; background-color: #f7f7f7;\">\n    <h3>Summary of Findings</h3>\n    <ol>\n        <li>The SII scores tend to increase with age but show a U-shaped relationship, with adolescents having the highest median PCIAT scores.</li>\n        <li>The higher the age, the more hours participants spent online (clear linear trend).</li>\n        <li>People with higher SII scores generally spend more time online, but adolescents stand out as the most affected age group across all categories of internet use.</li>\n        <li>There are participants of almost all ages (5 to 21) who spend less than an hour a day online and have high SII scores.</li>\n    </ol>\n    <p><em>Note:</em> These results should be interpreted with caution, as there is considerable overlap between the different SII and internet use categories, and severe cases and adults are under-represented in the data.</p>\n</div>","metadata":{}},{"cell_type":"markdown","source":"## **Children's Global Assessment Scale**","metadata":{}},{"cell_type":"code","source":"calc_stats(train, 'CGAS-CGAS_Score')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:11.953659Z","iopub.execute_input":"2024-11-28T17:33:11.953998Z","iopub.status.idle":"2024-11-28T17:33:11.973559Z","shell.execute_reply.started":"2024-11-28T17:33:11.953964Z","shell.execute_reply":"2024-11-28T17:33:11.972552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[train['CGAS-CGAS_Score']>100]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:11.974704Z","iopub.execute_input":"2024-11-28T17:33:11.975034Z","iopub.status.idle":"2024-11-28T17:33:12.002754Z","shell.execute_reply.started":"2024-11-28T17:33:11.975004Z","shell.execute_reply":"2024-11-28T17:33:12.001676Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- mean and median is approximately equal that implies this score is normaly distributed.\n- we know the max value of the max value `CGAS SCORE` is 100. 999.0 is only one outlier here. if we remove it max value is 95.0. ","metadata":{}},{"cell_type":"code","source":"train.loc[train['CGAS-CGAS_Score'] == 999.0, 'CGAS-CGAS_Score']=np.nan","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:12.004031Z","iopub.execute_input":"2024-11-28T17:33:12.004349Z","iopub.status.idle":"2024-11-28T17:33:12.018130Z","shell.execute_reply.started":"2024-11-28T17:33:12.004319Z","shell.execute_reply":"2024-11-28T17:33:12.016936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_univar_plots(train, 'CGAS-CGAS_Score')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:12.019446Z","iopub.execute_input":"2024-11-28T17:33:12.019793Z","iopub.status.idle":"2024-11-28T17:33:13.371882Z","shell.execute_reply.started":"2024-11-28T17:33:12.019753Z","shell.execute_reply":"2024-11-28T17:33:13.370473Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- `CGAS-CGAS_Score` is unimodel. and the majority of data have score b/w 50 to 80. According to **CDF** approximately 50% of data have score 65 and lower than that. **QQ-Plot** is says `CGAS_Score` is not perfectly distributed becoze at the end it deviate from the original path.","metadata":{"execution":{"iopub.status.busy":"2024-11-27T11:25:56.080336Z","iopub.execute_input":"2024-11-27T11:25:56.081227Z","iopub.status.idle":"2024-11-27T11:25:56.087955Z","shell.execute_reply.started":"2024-11-27T11:25:56.081184Z","shell.execute_reply":"2024-11-27T11:25:56.086673Z"}}},{"cell_type":"markdown","source":"- `CGAS-CGAS_Score` is a rating score for the general function of patient. `CGAS` purpose the clinicains to rate the children from 1 to 100 based on their lowest level of functioning without any treatment and prognosis.\n- Let's bin the `CGAS-CGAS_Score` column based on the established score categories and draw counts.","metadata":{}},{"cell_type":"code","source":"bins = np.arange(0, 101, 10)\nlabels = [\n    '1-10 : Denotes critical mental health that demand urgent care',\n    '11-20 : Needs considirable supervision',\n    '21-30: Unable to function in almost all areas',\n    '31-40: Major imparement in functioning in Several areas',\n    '41-50: Modarate degree of interference in functioning',\n    '51-60: Variable functioning with sporadic functioning',\n    '61-70: Some difficulty in single area',\n    '71-80: Indicates general mental stability',\n    '81-90: strong mental resilience with minor concerns.',\n    '91-100:outstanding mental clarity' \n]\n\ntrain['CGAS_Score_Bin'] = pd.cut(train['CGAS-CGAS_Score'], bins=bins, labels=labels)\ncounts = train['CGAS_Score_Bin'].value_counts().reindex(labels).to_frame()\ncounts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:13.373587Z","iopub.execute_input":"2024-11-28T17:33:13.374094Z","iopub.status.idle":"2024-11-28T17:33:13.393904Z","shell.execute_reply.started":"2024-11-28T17:33:13.374030Z","shell.execute_reply":"2024-11-28T17:33:13.392741Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_filt = train.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:13.395449Z","iopub.execute_input":"2024-11-28T17:33:13.395828Z","iopub.status.idle":"2024-11-28T17:33:13.409204Z","shell.execute_reply.started":"2024-11-28T17:33:13.395792Z","shell.execute_reply":"2024-11-28T17:33:13.408050Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_filt = train_filt.dropna(subset=['CGAS_Score_Bin', 'complete_resp_total'])\ntrain_filt.loc[:, 'CGAS_Score_Bin'] = train_filt['CGAS_Score_Bin'].cat.remove_unused_categories()\ntrain_filt.loc[:, 'sii'] = train_filt['sii'].cat.remove_unused_categories()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:13.410692Z","iopub.execute_input":"2024-11-28T17:33:13.411068Z","iopub.status.idle":"2024-11-28T17:33:13.428757Z","shell.execute_reply.started":"2024-11-28T17:33:13.411034Z","shell.execute_reply":"2024-11-28T17:33:13.427413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(train_filt)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:13.430128Z","iopub.execute_input":"2024-11-28T17:33:13.430422Z","iopub.status.idle":"2024-11-28T17:33:13.439311Z","shell.execute_reply.started":"2024-11-28T17:33:13.430392Z","shell.execute_reply":"2024-11-28T17:33:13.438210Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CGAS-CGAS_Score v/s SII\nfig, ax=plt.subplots(1, 2, figsize=(15, 5))\nsns.boxplot(data=train_filt, x='sii', y='CGAS-CGAS_Score', palette='Set3', ax=ax[0])\nax[0].set_title('Distribution of CGAS-CGAS_Score by SII')\nax[0].set_xlabel('SII Score ')\nax[0].set_ylabel('CGAS-CGAS_Score ')\n\n# CGAS-CGAS label bins V/S Complete_resp_totalrange_labels=[label.split(':')[0] for label in train_filt['CGAS_Score_Bin'].cat.categories]\nrange_labels=[label.split(':')[0] for label in train_filt['CGAS_Score_Bin'].cat.categories]\nsns.boxplot(data=train_filt, x='CGAS_Score_Bin', y='complete_resp_total', palette='Set3', ax=ax[1])\nax[1].set_xticklabels(range_labels)\nax[1].set_title('Distribution of PCIAT response total by Different CGAS_Score_Bin')\nax[1].set_xlabel('CGAS Score Bin')\nax[1].set_ylabel('pciat complete response')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:13.440981Z","iopub.execute_input":"2024-11-28T17:33:13.441426Z","iopub.status.idle":"2024-11-28T17:33:14.037952Z","shell.execute_reply.started":"2024-11-28T17:33:13.441376Z","shell.execute_reply":"2024-11-28T17:33:14.036803Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Higher the SII SCORE lower the median of CGAS-CGAS_Score and the decrement is very small\n","metadata":{}},{"cell_type":"code","source":"score_min_max = train.groupby('sii')['CGAS-CGAS_Score'].agg(['min', 'max'])\nscore_min_max = score_min_max.rename(columns={'min': 'Minimum CGAS Score', 'max':'Maximum CGAS Score'})\nscore_min_max","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:14.039515Z","iopub.execute_input":"2024-11-28T17:33:14.040013Z","iopub.status.idle":"2024-11-28T17:33:14.056144Z","shell.execute_reply.started":"2024-11-28T17:33:14.039964Z","shell.execute_reply":"2024-11-28T17:33:14.054906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"score_min_max","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:14.057423Z","iopub.execute_input":"2024-11-28T17:33:14.058018Z","iopub.status.idle":"2024-11-28T17:33:14.075071Z","shell.execute_reply.started":"2024-11-28T17:33:14.057977Z","shell.execute_reply":"2024-11-28T17:33:14.073807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Distribution of WORST CGAS-CGAS_Score\ntrain_filt[train_filt['CGAS-CGAS_Score']<35][['Basic_Demos-Age','Basic_Demos-Sex', 'sii', 'CGAS-CGAS_Score', 'PreInt_EduHx-computerinternet_hoursday']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:14.076400Z","iopub.execute_input":"2024-11-28T17:33:14.076726Z","iopub.status.idle":"2024-11-28T17:33:14.097661Z","shell.execute_reply.started":"2024-11-28T17:33:14.076693Z","shell.execute_reply":"2024-11-28T17:33:14.096523Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Distribution of Good CGAS-CGAS_Score\ntrain_filt[train_filt['CGAS-CGAS_Score']>90][['Basic_Demos-Age', 'Basic_Demos-Sex', 'sii', 'CGAS-CGAS_Score', 'PreInt_EduHx-computerinternet_hoursday']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:14.099195Z","iopub.execute_input":"2024-11-28T17:33:14.099632Z","iopub.status.idle":"2024-11-28T17:33:14.122579Z","shell.execute_reply.started":"2024-11-28T17:33:14.099584Z","shell.execute_reply":"2024-11-28T17:33:14.121224Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- There are no participants with highest`CGAS-CGAS_Score` having higher score of SII. this suggest the parental responses to the PCIAT QUESTIONAIRE \nmay reflect some effects of PIU on global health and functioning.\n- The particiant with best and worst cgas score all have sii score either 0 or 1 with different hours of internet usage. this means in the train data participant already have other significant issue that not related with PIU.\n- The higher variability makes it hard to draw clear relationship between `SII` and `CGAS-CGAS_Score`.\n- Small sample size of certain cgas-score categories may lead to baised interpretation.","metadata":{}},{"cell_type":"markdown","source":"## **Physical Measure**","metadata":{}},{"cell_type":"code","source":"phy_columns = [col for col in train.columns if 'Physical' in col]\nphy_num_cols = [col for col in phy_columns if train[col].dtypes!='category']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:14.124118Z","iopub.execute_input":"2024-11-28T17:33:14.124577Z","iopub.status.idle":"2024-11-28T17:33:14.134749Z","shell.execute_reply.started":"2024-11-28T17:33:14.124520Z","shell.execute_reply":"2024-11-28T17:33:14.133588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_univar_plots(train, 'Physical-BMI')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:14.136211Z","iopub.execute_input":"2024-11-28T17:33:14.136636Z","iopub.status.idle":"2024-11-28T17:33:15.772851Z","shell.execute_reply.started":"2024-11-28T17:33:14.136589Z","shell.execute_reply":"2024-11-28T17:33:15.771427Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_univar_plots(train, 'Physical-Height')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:15.774267Z","iopub.execute_input":"2024-11-28T17:33:15.774619Z","iopub.status.idle":"2024-11-28T17:33:17.100390Z","shell.execute_reply.started":"2024-11-28T17:33:15.774584Z","shell.execute_reply":"2024-11-28T17:33:17.099247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_univar_plots(train, 'Physical-Weight')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:17.102093Z","iopub.execute_input":"2024-11-28T17:33:17.102532Z","iopub.status.idle":"2024-11-28T17:33:18.481832Z","shell.execute_reply.started":"2024-11-28T17:33:17.102486Z","shell.execute_reply":"2024-11-28T17:33:18.480595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_univar_plots(train, 'Physical-Waist_Circumference')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:18.483497Z","iopub.execute_input":"2024-11-28T17:33:18.483810Z","iopub.status.idle":"2024-11-28T17:33:19.951605Z","shell.execute_reply.started":"2024-11-28T17:33:18.483780Z","shell.execute_reply":"2024-11-28T17:33:19.950510Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_univar_plots(train, 'Physical-Diastolic_BP')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:19.952692Z","iopub.execute_input":"2024-11-28T17:33:19.953017Z","iopub.status.idle":"2024-11-28T17:33:21.330715Z","shell.execute_reply.started":"2024-11-28T17:33:19.952986Z","shell.execute_reply":"2024-11-28T17:33:21.329427Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_univar_plots(train, 'Physical-HeartRate')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:21.332255Z","iopub.execute_input":"2024-11-28T17:33:21.332665Z","iopub.status.idle":"2024-11-28T17:33:22.844304Z","shell.execute_reply.started":"2024-11-28T17:33:21.332621Z","shell.execute_reply":"2024-11-28T17:33:22.843311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_univar_plots(train, 'Physical-Systolic_BP')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:22.845846Z","iopub.execute_input":"2024-11-28T17:33:22.846289Z","iopub.status.idle":"2024-11-28T17:33:24.587189Z","shell.execute_reply.started":"2024-11-28T17:33:22.846244Z","shell.execute_reply":"2024-11-28T17:33:24.585945Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calc_stats(train, phy_num_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:24.588736Z","iopub.execute_input":"2024-11-28T17:33:24.589198Z","iopub.status.idle":"2024-11-28T17:33:24.628656Z","shell.execute_reply.started":"2024-11-28T17:33:24.589162Z","shell.execute_reply":"2024-11-28T17:33:24.627484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# weight height cols \nwt_ht_cols = ['Physical-BMI', 'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:24.630009Z","iopub.execute_input":"2024-11-28T17:33:24.630318Z","iopub.status.idle":"2024-11-28T17:33:24.635478Z","shell.execute_reply.started":"2024-11-28T17:33:24.630286Z","shell.execute_reply":"2024-11-28T17:33:24.634301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(train[wt_ht_cols]==0).sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:24.636825Z","iopub.execute_input":"2024-11-28T17:33:24.637159Z","iopub.status.idle":"2024-11-28T17:33:24.658181Z","shell.execute_reply.started":"2024-11-28T17:33:24.637129Z","shell.execute_reply":"2024-11-28T17:33:24.657006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[wt_ht_cols] = train[wt_ht_cols].replace(0, np.nan)\ncalc_stats(train, wt_ht_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:24.659771Z","iopub.execute_input":"2024-11-28T17:33:24.660178Z","iopub.status.idle":"2024-11-28T17:33:24.693401Z","shell.execute_reply.started":"2024-11-28T17:33:24.660145Z","shell.execute_reply":"2024-11-28T17:33:24.692266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# conversion of weight in kg and height in cm to recalculate BMI\ninch_to_cm = 2.54\nlbs_to_kg = 0.453592\n\ntrain['Physical-Weight'] = lbs_to_kg * train['Physical-Weight']\ntrain['Physical-Height'] = inch_to_cm* train['Physical-Height']\ntrain['Physical-Waist_Circumference'] = inch_to_cm * train['Physical-Waist_Circumference']\n\n# BMI = weight(in kg)/height(m*m)\ntrain['Physical-BMI'] = np.where(train['Physical-Weight'].notna()&train['Physical-Height'].notna(), train['Physical-Weight']/((train['Physical-Height']/100)**2), np.nan)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:24.694489Z","iopub.execute_input":"2024-11-28T17:33:24.694828Z","iopub.status.idle":"2024-11-28T17:33:24.704317Z","shell.execute_reply.started":"2024-11-28T17:33:24.694796Z","shell.execute_reply":"2024-11-28T17:33:24.703106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calc_stats(train, wt_ht_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:24.705969Z","iopub.execute_input":"2024-11-28T17:33:24.706431Z","iopub.status.idle":"2024-11-28T17:33:24.743777Z","shell.execute_reply.started":"2024-11-28T17:33:24.706379Z","shell.execute_reply":"2024-11-28T17:33:24.742436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 3, figsize=(15, 5))\n# Physical-Weight v/s Age \nsns.scatterplot(data=train, x='Basic_Demos-Age', y='Physical-Weight', ax=ax[0])\nax[0].set_title('Physical Weight by Age')\nax[0].set_xlabel('Age of Participants')\nax[0].set_ylabel('Weight of Participants')\n\n# Physical-Height v/s Age\nsns.scatterplot(data=train, x='Basic_Demos-Age', y='Physical-Height', ax=ax[1])\nax[1].set_title('Physical Height by Age')\nax[1].set_xlabel('Age of Participants')\nax[1].set_ylabel('Height of Participants')\n\n# Physical-Waist_Cicumference\nsns.scatterplot(data=train, x='Physical-Weight', y='Physical-Waist_Circumference', ax=ax[2])\nax[2].set_title('Waist circumference by weight')\nax[2].set_xlabel('Weight of Participants')\nax[2].set_ylabel('Waist_Circumference of Participants')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:24.745350Z","iopub.execute_input":"2024-11-28T17:33:24.745885Z","iopub.status.idle":"2024-11-28T17:33:25.476381Z","shell.execute_reply.started":"2024-11-28T17:33:24.745811Z","shell.execute_reply":"2024-11-28T17:33:25.475296Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- As the Age increases weight and height of participants also increases along with that as the weight increases waist of circumference will also increases. this all show correct relation that we prefer\n- However, there are indivisuals who are unusually taller and overweight according to their Age.\n- There are also few outliers in waist circumference when weight is 40 and waist circumference is 100 cm. and many more in other two plots.\n- the problem with data cleaning is that we don't say that data is incorrect or not for example at the age of 7.5 years height is 175cm it's possible the harmon growth of indivisual is high OR gignatism OR incorrect value is entered.","metadata":{}},{"cell_type":"markdown","source":"#### Blood Pressure & Heart Rate","metadata":{}},{"cell_type":"code","source":"# blood pressure and heartrate cols \nbp_hr_cols = ['Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:25.477741Z","iopub.execute_input":"2024-11-28T17:33:25.478180Z","iopub.status.idle":"2024-11-28T17:33:25.483543Z","shell.execute_reply.started":"2024-11-28T17:33:25.478134Z","shell.execute_reply":"2024-11-28T17:33:25.482440Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(train[bp_hr_cols]<50).sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:25.485089Z","iopub.execute_input":"2024-11-28T17:33:25.485740Z","iopub.status.idle":"2024-11-28T17:33:25.502035Z","shell.execute_reply.started":"2024-11-28T17:33:25.485704Z","shell.execute_reply":"2024-11-28T17:33:25.500752Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Fact** \n- We know that systolic BP is always greater than Diastolic BP","metadata":{}},{"cell_type":"code","source":"train[train['Physical-Diastolic_BP'] >= train['Physical-Systolic_BP']][bp_hr_cols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:25.503766Z","iopub.execute_input":"2024-11-28T17:33:25.504239Z","iopub.status.idle":"2024-11-28T17:33:25.522146Z","shell.execute_reply.started":"2024-11-28T17:33:25.504188Z","shell.execute_reply":"2024-11-28T17:33:25.520813Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Again we not comment on the accuracy of data, But we can flag these rows for further manual inspection one by one. or replace the with NaN values. ","metadata":{}},{"cell_type":"code","source":"train[bp_hr_cols] = train[bp_hr_cols].replace(0, np.nan)\ntrain.loc[train['Physical-Diastolic_BP'] >= train['Physical-Systolic_BP'], bp_hr_cols] = np.nan","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:25.524076Z","iopub.execute_input":"2024-11-28T17:33:25.525022Z","iopub.status.idle":"2024-11-28T17:33:25.535578Z","shell.execute_reply.started":"2024-11-28T17:33:25.524977Z","shell.execute_reply":"2024-11-28T17:33:25.534354Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2,sharey=True, figsize=(10,5))\nsns.scatterplot(data=train, y='Physical-HeartRate', x='Physical-Diastolic_BP', ax=ax[0])\nax[0].set_title('HeartRate by Diastolic BP')\nax[0].set_xlabel('Diastolic BP (mmHg)')\nax[0].set_ylabel('HeartRate (beats/min)')\n\nsns.scatterplot(data=train, x='Physical-HeartRate', y='Physical-Systolic_BP', ax=ax[1])\nax[1].set_title('HeartRate by Systolic BP')\nax[1].set_xlabel('Systolic BP (mmHg)')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:25.537216Z","iopub.execute_input":"2024-11-28T17:33:25.537841Z","iopub.status.idle":"2024-11-28T17:33:25.992419Z","shell.execute_reply.started":"2024-11-28T17:33:25.537766Z","shell.execute_reply":"2024-11-28T17:33:25.991230Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> There is no relationship is found b/w heartrate and blood pressure. i think measurement of this feature will happened in resting state or under non-stressfull condition","metadata":{}},{"cell_type":"markdown","source":"#### Blood pressure vs Body Mass Index (BMI)¶\n>  Typically, systolic (SBP) and diastolic (DBP) blood pressure are positively correlated, as they both reflect the functioning of the cardiovascular system. However, there can be deviations:\n\n- Isolated Systolic Hypertension: High SBP with normal DBP\n- Isolated Diastolic Hypertension: Normal SBP with high DBP\n- General Hypertension: Both SBP and DBP are elevated\n\n> BMI is often used as an indicator of overall body fat and can correlate with blood pressure (e.g. higher BMI values indicating overweight or obesity are commonly associated with elevated blood pressure). Let's see if this is true for the study participants.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(12,5))\nsns.scatterplot(data=train, x='Physical-BMI', y='Physical-Systolic_BP', ax=ax[0])\nax[0].set_title('BMI by Diastolic BP')\nax[0].set_xlabel('Systolic BP (mmHg)')\nax[0].set_ylabel('BMI (in weight/m^2)')\n\nsns.scatterplot(data=train, y='Physical-Diastolic_BP', x='Physical-Systolic_BP', ax=ax[1], color='red')\nax[1].set_title('Diastolic BP by Systolic BP')\nax[1].set_xlabel('Systolic BP (mmHg)')\nax[1].set_ylabel('Diastolic BP (mmHg)')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:25.994056Z","iopub.execute_input":"2024-11-28T17:33:25.994506Z","iopub.status.idle":"2024-11-28T17:33:26.486253Z","shell.execute_reply.started":"2024-11-28T17:33:25.994449Z","shell.execute_reply":"2024-11-28T17:33:26.485078Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- There is no correlation b/w systolic BP and BMI as expected\n- Systolic BP show strong correlation with Diastolic BP. along with that Isolated Systolic Hypertension and Isolated Diastolic Hypertension is present or it might be instead of ISH, IDH outilier is present in our dataset.","metadata":{}},{"cell_type":"code","source":"normal_ranges = {\n    'Physical-BMI': (18.5, 24.9),\n    'Physical-Height' : (100, 190),\n    'Physical-Weight' : (20, 120),\n    'Physical-Waist_Circumference': (50, 90),\n    'Physical-Diastolic_BP': (60, 80),\n    'Physical-HeartRate' : (60, 100),\n    'Physical-Systolic_BP': (90, 120)\n}\n\ndef total_out_of_range(data, column, low, high):\n    return ((data[column]<low) | (data[column]>high)).sum()\n\ntotal_count = {\n    col : total_out_of_range(train, col, *normal_ranges[col]) for col in normal_ranges\n}\n\nprint('Number of rows with values that outside of ranges: ')\n\nfor col, count in total_count.items():\n    total_valid = train[col].notna().sum()\n    percentage = (count/total_valid)*100\n    print(f'{col} : {count} ({percentage:.2f}%)')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:26.496618Z","iopub.execute_input":"2024-11-28T17:33:26.497051Z","iopub.status.idle":"2024-11-28T17:33:26.512603Z","shell.execute_reply.started":"2024-11-28T17:33:26.497013Z","shell.execute_reply":"2024-11-28T17:33:26.511429Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bmi_categories = [\n    ('Underweight', train['Physical-BMI']<18.5),\n    ('Normal weight', (train['Physical-BMI']>=18.5) & (train['Physical-BMI']<=24.9)),\n    ('Overweight', (train['Physical-BMI']>24.9) & (train['Physical-BMI']<=29.9)),\n    ('Obesity', (train['Physical-BMI']>=30))\n]\n\nbmi_categories_count = {label: condition.sum() for label, condition in bmi_categories}\n\nplt.figure(figsize=(5, 6))\nplt.pie(bmi_categories_count.values(), labels=bmi_categories_count.keys(), autopct='%1.1f%%', startangle=90, colors=plt.cm.Set3.colors)\nplt.title('BMI Distribution by Category')\nplt.axis('equal')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:26.514145Z","iopub.execute_input":"2024-11-28T17:33:26.514548Z","iopub.status.idle":"2024-11-28T17:33:26.707645Z","shell.execute_reply.started":"2024-11-28T17:33:26.514512Z","shell.execute_reply":"2024-11-28T17:33:26.705934Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Check Extreme Deviation Cases**","metadata":{"execution":{"iopub.status.busy":"2024-11-27T11:36:49.907482Z","iopub.execute_input":"2024-11-27T11:36:49.907982Z","iopub.status.idle":"2024-11-27T11:36:49.915258Z","shell.execute_reply.started":"2024-11-27T11:36:49.907937Z","shell.execute_reply":"2024-11-27T11:36:49.913538Z"}}},{"cell_type":"code","source":"train[train['Physical-BMI']<12][phy_columns+['Basic_Demos-Age']].sort_values(by='Physical-BMI')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:26.709429Z","iopub.execute_input":"2024-11-28T17:33:26.711475Z","iopub.status.idle":"2024-11-28T17:33:26.751549Z","shell.execute_reply.started":"2024-11-28T17:33:26.711412Z","shell.execute_reply":"2024-11-28T17:33:26.750425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[train['Physical-Systolic_BP']>160][phy_columns+['Basic_Demos-Age']].sort_values('Physical-Systolic_BP')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:26.753258Z","iopub.execute_input":"2024-11-28T17:33:26.753693Z","iopub.status.idle":"2024-11-28T17:33:26.782177Z","shell.execute_reply.started":"2024-11-28T17:33:26.753648Z","shell.execute_reply":"2024-11-28T17:33:26.780919Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Significant number of `BMI` and `Blood Pressure` values gone out of range corresponding to normal values suggesting that many participant have disproportionate body proportions or it might be possible incorrect values are filled. while `Height` and `Weight` of Participants are within range.\n","metadata":{}},{"cell_type":"markdown","source":"## **FitnessGrame Vitals and Treadmill**","metadata":{"execution":{"iopub.status.busy":"2024-11-27T11:37:35.877617Z","iopub.execute_input":"2024-11-27T11:37:35.878007Z","iopub.status.idle":"2024-11-27T11:37:35.882780Z","shell.execute_reply.started":"2024-11-27T11:37:35.877975Z","shell.execute_reply":"2024-11-27T11:37:35.881504Z"}}},{"cell_type":"code","source":"phy_columns = [col for col in train.columns if 'Physical' in col]\nfvt_columns = [col for col in train.columns if 'Fitness' in col]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:26.783810Z","iopub.execute_input":"2024-11-28T17:33:26.784284Z","iopub.status.idle":"2024-11-28T17:33:26.791957Z","shell.execute_reply.started":"2024-11-28T17:33:26.784234Z","shell.execute_reply":"2024-11-28T17:33:26.790641Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* Fitness_Endurance-Max_Stage : highest endurance level a person can achieve during a fitness test before stopping due to exhaustion. For a treadmill-based exercise test, the minimum time is typically 3 minutes, which is the duration of the first stage.\n* Fitness_Endurance-Time_Mins : total time taken in minutes to achieve fitness endurance max stage.\n* total time to achieve max stage is obtain by concatenation of `Fitness_Endurance-Time_Mins` and `Fitness_Endurance-Time_Sec`","metadata":{}},{"cell_type":"code","source":"data = train[train['Fitness_Endurance-Max_Stage'].notnull()]\nage_range = data['Basic_Demos-Age']\nprint(f'Age range for participants with Fitness_Endurance-Max_Stage data: {age_range.min()} - {age_range.max()} years')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:26.793440Z","iopub.execute_input":"2024-11-28T17:33:26.793808Z","iopub.status.idle":"2024-11-28T17:33:26.807222Z","shell.execute_reply.started":"2024-11-28T17:33:26.793774Z","shell.execute_reply":"2024-11-28T17:33:26.805955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_univar_plots(train, 'Fitness_Endurance-Max_Stage')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:26.808759Z","iopub.execute_input":"2024-11-28T17:33:26.809213Z","iopub.status.idle":"2024-11-28T17:33:28.341119Z","shell.execute_reply.started":"2024-11-28T17:33:26.809164Z","shell.execute_reply":"2024-11-28T17:33:28.339927Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- the majority of data have max stage b/w 4-8 with highly skewed towards right. median for the data is nearly about stage 5.\n- in `QQ-PLOT`, data is deviated from the line which directly indected that distribution of data is not normal.","metadata":{}},{"cell_type":"code","source":"fvt_columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:28.342647Z","iopub.execute_input":"2024-11-28T17:33:28.343042Z","iopub.status.idle":"2024-11-28T17:33:28.349967Z","shell.execute_reply.started":"2024-11-28T17:33:28.343005Z","shell.execute_reply":"2024-11-28T17:33:28.348613Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 4, figsize=(15, 5))\n\n# fiteness endurance season\ntrain['Fitness_Endurance-Season'].value_counts(normalize=True, dropna=False).plot.pie(autopct='%1.1f%%', startangle=90, colors=plt.cm.Set3.colors, ax=ax[0])\naxes[0].set_title('Fitness Endurance Season')\naxes[0].axis('equal')\n\n# fitness endurance season by max stage\nsns.violinplot(data=train, y='Fitness_Endurance-Max_Stage', x='Fitness_Endurance-Season', ax=ax[1])\nax[1].set_title('Max Fitness Stage by Season')\n\n# histplot for time in minutes\nsns.histplot(data=train, x='Fitness_Endurance-Time_Mins', kde=True, bins=20, ax=ax[2])\nax[2].set_xlabel('Time (in mins)')\nax[2].set_title('Fitness Endurance Time (in Mins)')\n\nsns.histplot(data=train, x='Fitness_Endurance-Time_Sec', kde=True, bins=20, ax=ax[3])\nax[3].set_xlabel('Time (in Sec)')\nax[3].set_title('Fitness Endurance Time (in Sec)')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:28.351315Z","iopub.execute_input":"2024-11-28T17:33:28.351759Z","iopub.status.idle":"2024-11-28T17:33:29.381436Z","shell.execute_reply.started":"2024-11-28T17:33:28.351715Z","shell.execute_reply":"2024-11-28T17:33:29.380373Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Endurance By Age**","metadata":{}},{"cell_type":"code","source":"plt.subplots(figsize=(10,4))\nsns.violinplot(data=train, y='Fitness_Endurance-Max_Stage', x='Basic_Demos-Age', palette='Set3')\nplt.title('Fitness Endurance Max Stage by Age')\nplt.ylabel('Max Stage')\nplt.xlabel('Age')\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:29.383032Z","iopub.execute_input":"2024-11-28T17:33:29.383388Z","iopub.status.idle":"2024-11-28T17:33:29.800015Z","shell.execute_reply.started":"2024-11-28T17:33:29.383354Z","shell.execute_reply":"2024-11-28T17:33:29.798883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calc_stats(train, fvt_columns[1:])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:29.801496Z","iopub.execute_input":"2024-11-28T17:33:29.801843Z","iopub.status.idle":"2024-11-28T17:33:29.830043Z","shell.execute_reply.started":"2024-11-28T17:33:29.801810Z","shell.execute_reply":"2024-11-28T17:33:29.828949Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> for the treadmile test we need minimum three minutes to complete 1 stage. let's see the minutes and seconds are missing for non missing value of max stage","metadata":{}},{"cell_type":"code","source":"train[(train['Fitness_Endurance-Max_Stage'].notna()) &  (train['Fitness_Endurance-Time_Mins'].isna() | train['Fitness_Endurance-Time_Sec'].isna())][fvt_columns[1:]]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:29.831415Z","iopub.execute_input":"2024-11-28T17:33:29.831882Z","iopub.status.idle":"2024-11-28T17:33:29.847274Z","shell.execute_reply.started":"2024-11-28T17:33:29.831809Z","shell.execute_reply":"2024-11-28T17:33:29.846126Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- we know for any non Na value of max stage time in mins >= 3 . it might be possible during data entry minutes or seconds were blank \n(endered as NaN) when they should have been recorded as 0 mins.While the missing seconds are not as important,\nthe missing minutes may actually be missing and treating them as 0 would give\nan incorrect test result. I think it's better to just remove these suspicious cases.","metadata":{}},{"cell_type":"code","source":"train.loc[\n    (train['Fitness_Endurance-Max_Stage'].notna()) &  \n    (train['Fitness_Endurance-Time_Mins'].isna() | \n     train['Fitness_Endurance-Time_Sec'].isna()), fvt_columns[1:]] = np.nan","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:29.848992Z","iopub.execute_input":"2024-11-28T17:33:29.849511Z","iopub.status.idle":"2024-11-28T17:33:29.863442Z","shell.execute_reply.started":"2024-11-28T17:33:29.849457Z","shell.execute_reply":"2024-11-28T17:33:29.862377Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['Fitness_Endurance-Total_Time_Sec'] = train['Fitness_Endurance-Time_Mins']*60+train['Fitness_Endurance-Time_Sec']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:29.864760Z","iopub.execute_input":"2024-11-28T17:33:29.865166Z","iopub.status.idle":"2024-11-28T17:33:29.877649Z","shell.execute_reply.started":"2024-11-28T17:33:29.865115Z","shell.execute_reply":"2024-11-28T17:33:29.876372Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calc_stats(train, ['Fitness_Endurance-Max_Stage','Fitness_Endurance-Total_Time_Sec'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:29.878974Z","iopub.execute_input":"2024-11-28T17:33:29.879355Z","iopub.status.idle":"2024-11-28T17:33:29.908304Z","shell.execute_reply.started":"2024-11-28T17:33:29.879315Z","shell.execute_reply":"2024-11-28T17:33:29.907175Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- On average participant reached at stage 5 in the endurance test.\n- Some of the participants are not able to pass single stage(min=0) or there is error in data again\n- There is small number of participants(like age 7 and 8 years) with exceptionally endurance stage.\n- There is substantial amount of data(over 80% of the dataset lacks this informations).","metadata":{}},{"cell_type":"markdown","source":"## **FitnessGram Child**","metadata":{}},{"cell_type":"code","source":"fgc_data_dict = data_dict[data_dict['Instrument'] == 'FitnessGram Child']\nfgc_columns = []\n\nfor index, row in fgc_data_dict.iterrows():\n    if '_Zone' not in row['Field']:\n        measure_field = row['Field']\n        measure_desc = row['Description']\n        zone_field = measure_field + '_Zone'\n        zone_row = fgc_data_dict[fgc_data_dict['Field'] == zone_field]\n\n        if not zone_row.empty:\n            zone_desc = zone_row['Description'].values[0]\n            fgc_columns.append((measure_field, zone_field, measure_desc, zone_desc))\n\n\nfig, axes = plt.subplots(2, 4, figsize=(25, 11))\n\n# Loop through fgc_columns and plot histograms\nfor idx, (measure, zone, measure_desc, zone_desc) in enumerate(fgc_columns):\n    row = idx // 4  \n    col = idx % 4  \n\n    sns.histplot(data=train, x=measure, hue=zone, bins=20, palette='Set1', ax=axes[row, col], kde=True)\n    axes[row, col].set_title(measure_desc)  # Add titles for subplots\n\nseason_counts = train['FGC-Season'].value_counts(dropna=False, normalize=True)\naxes[1, 3].pie(season_counts,labels=season_counts.index, autopct='%1.1f%%', startangle=90, colors=sns.color_palette('Set3'))\naxes[1, 3].set_title('Season of Participation')\naxes[1, 3].axis('equal')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:29.909668Z","iopub.execute_input":"2024-11-28T17:33:29.910048Z","iopub.status.idle":"2024-11-28T17:33:33.293996Z","shell.execute_reply.started":"2024-11-28T17:33:29.910015Z","shell.execute_reply":"2024-11-28T17:33:33.292645Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Most of the distribution skewed towards lower performance. but in most of the participants achieved a healthy fitness zone for `Sit & Reach total` and `Trunk lift Total`.\n- But the value of different zones overlap significantly. this may be because the zone ranges are different for different age group","metadata":{}},{"cell_type":"code","source":"measurment_columns = [measure for measure, _, _, _ in fgc_columns]\ncalc_stats(train, measurment_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:33.295580Z","iopub.execute_input":"2024-11-28T17:33:33.296001Z","iopub.status.idle":"2024-11-28T17:33:33.342045Z","shell.execute_reply.started":"2024-11-28T17:33:33.295963Z","shell.execute_reply":"2024-11-28T17:33:33.340657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def compute_min_max_by_sex(train, sex, fgc_columns):\n    results = []\n\n    for measure, zone, _, _ in fgc_columns:\n        sorted_zones = sorted(train[zone].dropna().unique())\n\n        for zone_val in sorted_zones:\n            data = train[(train['Basic_Demos-Sex']==sex) & (train[zone]==zone_val)][measure]\n\n            if not data.empty:\n                min, max = data.min(), data.max()\n\n                results.append({\n                    'Zone' : int(zone_val),\n                    'Measure': measure,\n                    'Min-Max' : f'{min}-{max}',\n                })\n\n    df = pd.DataFrame(results).pivot_table(\n        index='Zone', columns='Measure', values='Min-Max', aggfunc='first'\n    )\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:33.343625Z","iopub.execute_input":"2024-11-28T17:33:33.344067Z","iopub.status.idle":"2024-11-28T17:33:33.352824Z","shell.execute_reply.started":"2024-11-28T17:33:33.343992Z","shell.execute_reply":"2024-11-28T17:33:33.351513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"compute_min_max_by_sex(train, 'Male', fgc_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:33.354212Z","iopub.execute_input":"2024-11-28T17:33:33.354735Z","iopub.status.idle":"2024-11-28T17:33:33.440352Z","shell.execute_reply.started":"2024-11-28T17:33:33.354671Z","shell.execute_reply":"2024-11-28T17:33:33.439204Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> The ranges for each measure and zone by age (only for males, just to check if the overlap still exists):","metadata":{}},{"cell_type":"code","source":"result = []\n\nfor measure, zone, _, _ in fgc_columns:\n    sorted_data = sorted(train[zone].dropna().unique())\n    for zone_val in sorted_data:\n        zone_age_sex_data = train[train[zone]==zone_val][['Basic_Demos-Age', 'Basic_Demos-Sex', measure]]\n\n        age_agg = zone_age_sex_data['Basic_Demos-Age'].dropna().unique()\n    \n        for age in sorted(age_agg):\n            data = zone_age_sex_data[(zone_age_sex_data['Basic_Demos-Age']==age) & (zone_age_sex_data['Basic_Demos-Sex']=='Male')][measure]\n    \n            if not data.empty:\n                min_val, max_val = data.min(), data.max()\n                result.append({\n                    'Age': age,\n                    'Sex':'Male',\n                    'Zone': zone_val,\n                    'Min-Max': f'{min_val}-{max_val}',\n                    'Measure': measure\n                })\n\ndf = pd.DataFrame(result).pivot_table(\n    index=['Age','Sex','Zone'], columns='Measure', values='Min-Max', aggfunc='first'\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:33.442061Z","iopub.execute_input":"2024-11-28T17:33:33.442484Z","iopub.status.idle":"2024-11-28T17:33:33.672675Z","shell.execute_reply.started":"2024-11-28T17:33:33.442447Z","shell.execute_reply":"2024-11-28T17:33:33.671292Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:33.674531Z","iopub.execute_input":"2024-11-28T17:33:33.674914Z","iopub.status.idle":"2024-11-28T17:33:33.704512Z","shell.execute_reply.started":"2024-11-28T17:33:33.674879Z","shell.execute_reply":"2024-11-28T17:33:33.703215Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- This table show the various fitness zone(like: curl-up total , push-up total, etc) across different age and zone.\n- There is significant overlaps b/w min max value of fitness zone like at the age of 9 `FGC-FGC_CU` have the value 6.0-10 that may corresponds to Zone0(needs to improvement) or  Zone1(Healthy Fitness Zone)\n- This overlap indicate that the criteria for each zone are not sharply defined by specific ranges.even for the same age. and these zone columns appear to be as features that add extra noise. i would not use them in modeling.","metadata":{}},{"cell_type":"markdown","source":"*Age Range for each measure columns*","metadata":{"execution":{"iopub.status.busy":"2024-11-27T13:31:11.570731Z","iopub.execute_input":"2024-11-27T13:31:11.571483Z","iopub.status.idle":"2024-11-27T13:31:11.577796Z","shell.execute_reply.started":"2024-11-27T13:31:11.571446Z","shell.execute_reply":"2024-11-27T13:31:11.576659Z"}}},{"cell_type":"code","source":"range = []\nfor measure in measurment_columns:\n    valid_data = train[~train[measure].isna()]\n    min_val = valid_data['Basic_Demos-Age'].min()\n    max_val = valid_data['Basic_Demos-Age'].max()\n\n    range.append({\n        'Measure': measure,\n        'Min Age': min_val,\n        'Max Age': max_val\n    })\n\ndf = pd.DataFrame(range)\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:33.706379Z","iopub.execute_input":"2024-11-28T17:33:33.706849Z","iopub.status.idle":"2024-11-28T17:33:33.737896Z","shell.execute_reply.started":"2024-11-28T17:33:33.706797Z","shell.execute_reply":"2024-11-28T17:33:33.736748Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**RelationShip with target variable (PCIAT_Total for complete PCIAT Response)**","metadata":{}},{"cell_type":"code","source":"train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:33.739114Z","iopub.execute_input":"2024-11-28T17:33:33.739442Z","iopub.status.idle":"2024-11-28T17:33:33.746921Z","shell.execute_reply.started":"2024-11-28T17:33:33.739406Z","shell.execute_reply":"2024-11-28T17:33:33.745772Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cols = [col for col in train.columns if col.startswith('FGC-') and 'Zone' not in col and 'Season' not in col]\ncols.extend(['Fitness_Endurance-Max_Stage','Fitness_Endurance-Total_Time_Sec'])\n\ndata_subset = train[cols + ['complete_resp_total']]\n\ncm = data_subset.corr()\nplt.figure(figsize=(10,8))\nsns.heatmap(cm, vmin=-1, vmax=1, annot=True, fmt='.2f', cmap='coolwarm')\nplt.title('Correlation HeatMap')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:33.748277Z","iopub.execute_input":"2024-11-28T17:33:33.748589Z","iopub.status.idle":"2024-11-28T17:33:34.363658Z","shell.execute_reply.started":"2024-11-28T17:33:33.748546Z","shell.execute_reply":"2024-11-28T17:33:34.362519Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- `Grip Strength(dominant)` is highly correlated with `Grip Strength(Non dominant)` , `Seat & Reach left` is highly correlated with `Seat & Reach right`, and also `Fitness Endurance max stage` is highly correlated with `Fitness Endurance total time sec`\n- `Curl-Up` is partially correlated with `Push-Up`.\n- Better performance in physical test doesn't neccessarily indicate a higher level of physical activity.\n- the main thing to rememeber here is that physical performance also improve with age, So the positive correlation between physical performance and PIU Severity is driven by age.","metadata":{}},{"cell_type":"markdown","source":"#### Let's see how the picture changes when we plots the same thing by Age group, and Add age to see if the measure still correlated with age","metadata":{}},{"cell_type":"code","source":"age_group = train['Age Group'].unique()\nfig, axes = plt.subplots(1, 3, figsize=(18, 6))\nfor i, age_grp in enumerate(age_group):\n    data = train[train['Age Group']==age_grp]\n    cm = data[cols + ['Basic_Demos-Age', 'complete_resp_total']].corr()\n    sns.heatmap(cm,annot=True, vmin=-1, vmax=1, cmap='coolwarm', fmt='.1f', ax=axes[i], cbar=i==0)\n    axes[i].set_title(f'{age_grp}')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:34.365020Z","iopub.execute_input":"2024-11-28T17:33:34.365341Z","iopub.status.idle":"2024-11-28T17:33:36.063078Z","shell.execute_reply.started":"2024-11-28T17:33:34.365310Z","shell.execute_reply":"2024-11-28T17:33:36.061932Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Note**\n\n- In each age group we see that age correlates with all the physical fitness positively mainly in case of adults and child.\n- The correlation b/w age and complete resp total (PIU severity) mainly for child shows +ve correlation.\n- for `Adeloscent` , very week correlation or null b/w PIU Sevirity with physical fitness and in case of `adults` there is no data.\n- In overall PIU severity do not show noticable correlation with physical fitness, and it appear that age may be driving both increased fitness performance and higher PIU Severity","metadata":{}},{"cell_type":"markdown","source":"## **Bio-electric Impedence Analysis**","metadata":{}},{"cell_type":"code","source":"bia = [col for col in train.columns if 'BIA' in col]\nbia","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:36.064929Z","iopub.execute_input":"2024-11-28T17:33:36.065363Z","iopub.status.idle":"2024-11-28T17:33:36.073674Z","shell.execute_reply.started":"2024-11-28T17:33:36.065313Z","shell.execute_reply":"2024-11-28T17:33:36.072478Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bia_col = bia[1:]\nbia_col.extend(['Basic_Demos-Age', 'complete_resp_total'])\ncm = train[bia_col].corr()\nplt.figure(figsize=(10,8))\nsns.heatmap(cm, vmin=-1, vmax=1, annot=True, fmt='.2f', cmap='coolwarm')\nplt.title('Correlation HeatMap')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:36.075245Z","iopub.execute_input":"2024-11-28T17:33:36.075688Z","iopub.status.idle":"2024-11-28T17:33:37.298027Z","shell.execute_reply.started":"2024-11-28T17:33:36.075640Z","shell.execute_reply":"2024-11-28T17:33:37.296811Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Highly correlated to each other ","metadata":{"execution":{"iopub.status.busy":"2024-11-28T16:27:08.595445Z","iopub.execute_input":"2024-11-28T16:27:08.595844Z","iopub.status.idle":"2024-11-28T16:27:08.604029Z","shell.execute_reply.started":"2024-11-28T16:27:08.595807Z","shell.execute_reply":"2024-11-28T16:27:08.602259Z"}}},{"cell_type":"code","source":"from sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\n\n# Select the features for PCA\nbia_col = bia[1:]\nX = train[bia_col].dropna()\n\n# Standardize the data\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)\n\n# Perform PCA\npca = PCA()\nX_pca = pca.fit_transform(X_scaled)\n\n# Explained Variance Ratio\nexplained_variance = pca.explained_variance_ratio_\n\n# Plotting the explained variance ratio\nplt.figure(figsize=(10, 6))\nplt.bar(\n    x=np.arange(len(explained_variance)),  # Use numpy's arange to avoid issues\n    height=explained_variance, \n    alpha=0.7, \n    align='center', \n    label='Individual Explained Variance'\n)\nplt.step(\n    x=np.arange(len(explained_variance)), \n    y=np.cumsum(explained_variance), \n    where='mid', \n    label='Cumulative Explained Variance'\n)\nplt.ylabel('Explained Variance Ratio')\nplt.xlabel('Principal Component Index')\nplt.legend(loc='best')\nplt.title('Explained Variance by Principal Components')\nplt.show()\n\n# Calculate the number of components explaining 99% variance\ncumulative_variance = np.cumsum(explained_variance)\nn_components_99 = np.argmax(cumulative_variance >= 0.99) + 1\nprint(f\"Number of components explaining 99% variance: {n_components_99}\")\n\n# Create a DataFrame of the principal components\npca_df = pd.DataFrame(X_pca, columns=[f\"PC{i+1}\" for i in np.arange(X_pca.shape[1])])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:37.299608Z","iopub.execute_input":"2024-11-28T17:33:37.299979Z","iopub.status.idle":"2024-11-28T17:33:37.944614Z","shell.execute_reply.started":"2024-11-28T17:33:37.299944Z","shell.execute_reply":"2024-11-28T17:33:37.943528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bia_data_dict = data_dict[data_dict['Instrument'] == 'Bio-electric Impedance Analysis']\nnumeric_bia = bia_data_dict[bia_data_dict['Type']=='float']['Field'].to_list()\ncategoric_bia=bia_data_dict[bia_data_dict['Type']!='float']['Field'].to_list()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:37.946215Z","iopub.execute_input":"2024-11-28T17:33:37.946527Z","iopub.status.idle":"2024-11-28T17:33:37.953625Z","shell.execute_reply.started":"2024-11-28T17:33:37.946496Z","shell.execute_reply":"2024-11-28T17:33:37.952515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize=(17,5))\n\nseason_counts = train['BIA-Season'].value_counts(normalize=True, dropna=False)\naxes[0].pie(season_counts, labels=season_counts.index, autopct='%1.1f%%', startangle=90, colors=sns.color_palette('Set3'))\naxes[0].set_title('Season of Participation')\naxes[0].axis('equal')\n\nfor i, col in enumerate(categoric_bia[1:]):\n    sns.countplot(x=col, data=train, palette=\"Set3\", ax=axes[i+1])\n    axes[i+1].set_title(data_dict[data_dict['Field'] == col]['Description'].values[0])\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:37.954997Z","iopub.execute_input":"2024-11-28T17:33:37.955333Z","iopub.status.idle":"2024-11-28T17:33:38.508308Z","shell.execute_reply.started":"2024-11-28T17:33:37.955301Z","shell.execute_reply":"2024-11-28T17:33:38.507039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bmi_data = train[['BIA-BIA_BMI', 'Physical-BMI']].dropna()\n\nplt.figure(figsize=(8,5))\nsns.scatterplot(x='BIA-BIA_BMI', y='Physical-BMI',data=bmi_data, color='b')\nplt.xlabel('BIA-BIA_BMI')\nplt.ylabel('Physical-BMI')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-28T17:33:38.509754Z","iopub.execute_input":"2024-11-28T17:33:38.510100Z","iopub.status.idle":"2024-11-28T17:33:38.821188Z","shell.execute_reply.started":"2024-11-28T17:33:38.510068Z","shell.execute_reply":"2024-11-28T17:33:38.819911Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Physical Activity Questionnaire**","metadata":{"execution":{"iopub.status.busy":"2024-11-28T16:54:17.067393Z","iopub.execute_input":"2024-11-28T16:54:17.068483Z","iopub.status.idle":"2024-11-28T16:54:17.076699Z","shell.execute_reply.started":"2024-11-28T16:54:17.068418Z","shell.execute_reply":"2024-11-28T16:54:17.075431Z"}}},{"cell_type":"markdown","source":"#### Adolescents","metadata":{}},{"cell_type":"code","source":"paq = [col for col in train.columns if 'PAQ_A' in col]\npaq","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:38.823087Z","iopub.execute_input":"2024-11-28T17:33:38.823569Z","iopub.status.idle":"2024-11-28T17:33:38.831598Z","shell.execute_reply.started":"2024-11-28T17:33:38.823519Z","shell.execute_reply":"2024-11-28T17:33:38.830410Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = train[train['PAQ_A-PAQ_A_Total'].notnull()]\nage_range = data['Basic_Demos-Age']\nprint(f'Age range for Adolescents (in PAQ_A-PAQ_A_Total data): {age_range.min()}-{age_range.max()} years')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:38.833013Z","iopub.execute_input":"2024-11-28T17:33:38.833358Z","iopub.status.idle":"2024-11-28T17:33:38.849110Z","shell.execute_reply.started":"2024-11-28T17:33:38.833326Z","shell.execute_reply":"2024-11-28T17:33:38.847957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_univar_plots(data, 'PAQ_A-PAQ_A_Total')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:38.850513Z","iopub.execute_input":"2024-11-28T17:33:38.851010Z","iopub.status.idle":"2024-11-28T17:33:40.208251Z","shell.execute_reply.started":"2024-11-28T17:33:38.850974Z","shell.execute_reply":"2024-11-28T17:33:40.207039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calc_stats(train, 'PAQ_A-PAQ_A_Total')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:33:40.209658Z","iopub.execute_input":"2024-11-28T17:33:40.210010Z","iopub.status.idle":"2024-11-28T17:33:40.229931Z","shell.execute_reply.started":"2024-11-28T17:33:40.209977Z","shell.execute_reply":"2024-11-28T17:33:40.228951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"season_count = train['PAQ_A-Season'].value_counts(normalize=True, dropna=False)\nplt.pie(season_count, labels=season_count.index, colors=plt.cm.Set3.colors, autopct='%1.1f%%')\nplt.title('PAQ_A Season (Adolescents)')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:34:03.747385Z","iopub.execute_input":"2024-11-28T17:34:03.747803Z","iopub.status.idle":"2024-11-28T17:34:03.895090Z","shell.execute_reply.started":"2024-11-28T17:34:03.747769Z","shell.execute_reply":"2024-11-28T17:34:03.893573Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Children","metadata":{}},{"cell_type":"code","source":"paq_c= [col for col in train.columns if 'PAQ_C' in col]\npaq_c","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:35:19.319546Z","iopub.execute_input":"2024-11-28T17:35:19.319986Z","iopub.status.idle":"2024-11-28T17:35:19.327522Z","shell.execute_reply.started":"2024-11-28T17:35:19.319948Z","shell.execute_reply":"2024-11-28T17:35:19.326335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = train[train['PAQ_C-PAQ_C_Total'].notnull()]\nage_range = data['Basic_Demos-Age']\nprint(f'Age range for Adolescents (in PAQ_A-PAQ_A_Total data): {age_range.min()}-{age_range.max()} years')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:36:35.510640Z","iopub.execute_input":"2024-11-28T17:36:35.511099Z","iopub.status.idle":"2024-11-28T17:36:35.523397Z","shell.execute_reply.started":"2024-11-28T17:36:35.511060Z","shell.execute_reply":"2024-11-28T17:36:35.521997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_univar_plots(data, 'PAQ_C-PAQ_C_Total')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:37:18.548082Z","iopub.execute_input":"2024-11-28T17:37:18.548541Z","iopub.status.idle":"2024-11-28T17:37:20.241480Z","shell.execute_reply.started":"2024-11-28T17:37:18.548496Z","shell.execute_reply":"2024-11-28T17:37:20.240254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calc_stats(train, 'PAQ_C-PAQ_C_Total')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:38:21.126803Z","iopub.execute_input":"2024-11-28T17:38:21.127239Z","iopub.status.idle":"2024-11-28T17:38:21.148840Z","shell.execute_reply.started":"2024-11-28T17:38:21.127200Z","shell.execute_reply":"2024-11-28T17:38:21.147700Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"season_count = train['PAQ_C-Season'].value_counts(normalize=True, dropna=False)\nplt.pie(season_count, labels=season_count.index, colors=plt.cm.Set3.colors, autopct='%1.1f%%')\nplt.title('PAQ_C Season (Children)')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:43:52.310187Z","iopub.execute_input":"2024-11-28T17:43:52.310702Z","iopub.status.idle":"2024-11-28T17:43:52.469915Z","shell.execute_reply.started":"2024-11-28T17:43:52.310662Z","shell.execute_reply":"2024-11-28T17:43:52.468147Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Sleep Disturbance","metadata":{}},{"cell_type":"code","source":"sds_col = [col for col in train.columns if 'SDS-' in col]\nsds_col","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:52:07.983850Z","iopub.execute_input":"2024-11-28T17:52:07.984276Z","iopub.status.idle":"2024-11-28T17:52:07.992244Z","shell.execute_reply.started":"2024-11-28T17:52:07.984242Z","shell.execute_reply":"2024-11-28T17:52:07.990964Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = train[train['SDS-SDS_Total_Raw'].notnull()] \nage_range = data['Basic_Demos-Age']\nprint(f'Age of participants for SDS-SDS_Total_Raw: {age_range.min()}-{age_range.max()}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T18:06:43.225048Z","iopub.execute_input":"2024-11-28T18:06:43.225466Z","iopub.status.idle":"2024-11-28T18:06:43.236643Z","shell.execute_reply.started":"2024-11-28T18:06:43.225434Z","shell.execute_reply":"2024-11-28T18:06:43.235424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calc_stats(train, sds_col[1:])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T18:07:48.926200Z","iopub.execute_input":"2024-11-28T18:07:48.926644Z","iopub.status.idle":"2024-11-28T18:07:48.950462Z","shell.execute_reply.started":"2024-11-28T18:07:48.926605Z","shell.execute_reply":"2024-11-28T18:07:48.949391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# SDS-SDS_Total_Raw\nfig, ax = plt.subplots(1,2,figsize=(15,5))\nsns.histplot(train['SDS-SDS_Total_Raw'].dropna(), bins=20, kde=True, ax=ax[0])\nax[0].set_title('SDS-SDS_Total_Raw')\nax[0].set_xlabel('Value')\n\n\nsns.histplot(train['SDS-SDS_Total_T'].dropna(), bins=20, kde=True, ax=ax[1])\nax[1].set_title('SDS-SDS_Total_T')\nax[1].set_xlabel('Value')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T18:10:49.666871Z","iopub.execute_input":"2024-11-28T18:10:49.667300Z","iopub.status.idle":"2024-11-28T18:10:50.285780Z","shell.execute_reply.started":"2024-11-28T18:10:49.667263Z","shell.execute_reply":"2024-11-28T18:10:50.284595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}