{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":7453542,"sourceType":"datasetVersion","datasetId":921302}],"dockerImageVersionId":30840,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import PercentFormatter\nimport os\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\npd.options.display.max_columns = None\nfrom sklearn.cluster import KMeans\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.manifold import TSNE\nimport plotly.express as px\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.pipeline import Pipeline\nfrom keras.models import Model\nfrom keras.layers import Input, Dense, Dropout\nfrom keras.optimizers import Adam\nfrom sklearn.preprocessing import StandardScaler\n\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import cohen_kappa_score, ConfusionMatrixDisplay, confusion_matrix\n\nimport re\nfrom sklearn.base import clone\nimport ydf\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\nimport polars as pl\n\n\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:07.716467Z","iopub.execute_input":"2025-01-23T08:27:07.716723Z","iopub.status.idle":"2025-01-23T08:27:25.935228Z","shell.execute_reply.started":"2025-01-23T08:27:07.716702Z","shell.execute_reply":"2025-01-23T08:27:25.934480Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import cohen_kappa_score, ConfusionMatrixDisplay, confusion_matrix\n\nimport re\nfrom sklearn.base import clone\nimport ydf\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.pipeline import Pipeline\nfrom keras.models import Model\nfrom keras.layers import Input, Dense, Dropout\nfrom keras.optimizers import Adam\nfrom sklearn.preprocessing import StandardScaler\n\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:25.936004Z","iopub.execute_input":"2025-01-23T08:27:25.936697Z","iopub.status.idle":"2025-01-23T08:27:25.941967Z","shell.execute_reply.started":"2025-01-23T08:27:25.936666Z","shell.execute_reply":"2025-01-23T08:27:25.941115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\nimport lightgbm as lgb\nfrom colorama import Fore, Style\nfrom scipy.optimize import minimize\nimport os","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:25.942783Z","iopub.execute_input":"2025-01-23T08:27:25.943084Z","iopub.status.idle":"2025-01-23T08:27:26.190698Z","shell.execute_reply.started":"2025-01-23T08:27:25.943057Z","shell.execute_reply":"2025-01-23T08:27:26.190088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_labels = ['None', 'Mild', 'Moderate', 'Severe']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:26.192561Z","iopub.execute_input":"2025-01-23T08:27:26.193108Z","iopub.status.idle":"2025-01-23T08:27:26.196289Z","shell.execute_reply.started":"2025-01-23T08:27:26.193086Z","shell.execute_reply":"2025-01-23T08:27:26.195514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"season_dtype = pd.CategoricalDtype(categories=['Spring', 'Summer', 'Fall', 'Winter'])\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n\n# chọn cột trong dataframe mà trên cột có chứa chuỗi con like=\nseason_columns_train  = train.filter(like='Season').columns\n\n#train[season_columns_train] = train[season_columns_train].astype(season_dtype)\n#train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:26.197840Z","iopub.execute_input":"2025-01-23T08:27:26.198108Z","iopub.status.idle":"2025-01-23T08:27:26.270297Z","shell.execute_reply.started":"2025-01-23T08:27:26.198088Z","shell.execute_reply":"2025-01-23T08:27:26.269485Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"season_dtype = pd.CategoricalDtype(categories=['Spring', 'Summer', 'Fall', 'Winter'])\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\n# chọn cột trong dataframe mà trên cột có chứa chuỗi con like=\nseason_columns_test  = test.filter(like='Season').columns\n\n#test[season_columns_test] = test[season_columns_test].astype(season_dtype)\n#test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:26.271299Z","iopub.execute_input":"2025-01-23T08:27:26.271520Z","iopub.status.idle":"2025-01-23T08:27:26.280816Z","shell.execute_reply.started":"2025-01-23T08:27:26.271500Z","shell.execute_reply":"2025-01-23T08:27:26.280150Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:26.281556Z","iopub.execute_input":"2025-01-23T08:27:26.281811Z","iopub.status.idle":"2025-01-23T08:27:26.292331Z","shell.execute_reply.started":"2025-01-23T08:27:26.281791Z","shell.execute_reply":"2025-01-23T08:27:26.291728Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let's identify the features that are related to the target variable and that are not present in the test set.","metadata":{}},{"cell_type":"code","source":"train.columns.difference(test.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:26.293154Z","iopub.execute_input":"2025-01-23T08:27:26.293365Z","iopub.status.idle":"2025-01-23T08:27:26.305991Z","shell.execute_reply.started":"2025-01-23T08:27:26.293347Z","shell.execute_reply":"2025-01-23T08:27:26.305289Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Parent-Child Internet Addiction Test (PCIAT): contains 20 items (PCIAT-PCIAT_01 to PCIAT-PCIAT_20), each assessing a different aspect of a child's behavior related to internet use. The items are answered on a scale (from 0 to 5), and the total score provides an indication of the severity of internet addiction.\n\nWe also have season of participation in PCIAT-Season and total Score in PCIAT-PCIAT_Total; so there are 22 PCIAT test-related columns in total.","metadata":{}},{"cell_type":"markdown","source":"Let's verify that the PCIAT-PCIAT_Total align with the corresponding sii categories by calculating its minimum and maximum scores for each sii category:","metadata":{}},{"cell_type":"code","source":"pciat_min_max = train.groupby('sii')['PCIAT-PCIAT_Total'].agg(['min', 'max'])\npciat_min_max = pciat_min_max.rename(\n    columns={'min': 'Minimum PCIAT total Score', 'max': 'Maximum total PCIAT Score'}\n)\npciat_min_max","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:26.306702Z","iopub.execute_input":"2025-01-23T08:27:26.306891Z","iopub.status.idle":"2025-01-23T08:27:26.333338Z","shell.execute_reply.started":"2025-01-23T08:27:26.306874Z","shell.execute_reply":"2025-01-23T08:27:26.332697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dict[data_dict['Field'] == 'PCIAT-PCIAT_Total']['Value Labels'].iloc[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:26.334127Z","iopub.execute_input":"2025-01-23T08:27:26.334310Z","iopub.status.idle":"2025-01-23T08:27:26.340823Z","shell.execute_reply.started":"2025-01-23T08:27:26.334293Z","shell.execute_reply":"2025-01-23T08:27:26.340091Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Missing Values**","metadata":{}},{"cell_type":"markdown","source":"All columns have a substantial proportion of missing values, except id and the three basic demographic columns for sex, age and season of enrollment. Even the target sii has missing values:","metadata":{}},{"cell_type":"code","source":"temp = train.copy()\nmissing_count = temp.isnull().sum() # Series\nmissing_count_df = (missing_count.reset_index()) # dataframe\nmissing_count_df.columns = ['feature', 'null_count']\nmissing_count_df = missing_count_df.sort_values(by='null_count', ascending=False)\nmissing_count_df['null_ratio'] = missing_count_df['null_count']/len(train)\n\nplt.figure(figsize=(6, 15))\nplt.title(f'Missing values over the {len(train)} samples')\nplt.barh(np.arange(len(missing_count_df)), missing_count_df['null_ratio'], color='coral', label='missing')\nplt.barh(np.arange(len(missing_count_df)), \n         1 - missing_count_df['null_ratio'],\n         left=missing_count_df['null_ratio'],\n         color='darkseagreen', label='available')\nplt.yticks(np.arange(len(missing_count_df)), missing_count_df['feature'])\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0,1)\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:26.341555Z","iopub.execute_input":"2025-01-23T08:27:26.341740Z","iopub.status.idle":"2025-01-23T08:27:27.279497Z","shell.execute_reply.started":"2025-01-23T08:27:26.341724Z","shell.execute_reply":"2025-01-23T08:27:27.278608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Copy the data\ntrain_df = train.copy()\n\n# Define the mapping for `sii` levels\nsii_map = {\n    0: \"None\",\n    1: \"Mild\",\n    2: \"Moderate\",\n    3: \"Severe\",\n    np.nan: \"NaN\"  # Explicitly map NaN for clarity\n}\n\n# Map the sii levels, leaving NaNs as they are\ntrain_df['sii_label'] = train_df['sii'].map(sii_map)\n\n# Count occurrences for each `sii` level, including NaN values\nsii_counts = train['sii'].value_counts(dropna=False)\nsii_percentages = (sii_counts / sii_counts.sum()) * 100\n# Create DataFrame for plotting\nsii_data = pd.DataFrame({\n    'SII Level': sii_counts.index,\n    'Count': sii_counts.values,\n    'Percentage': sii_percentages.values\n})\n\n# Reorder categories to place NaN last\nsii_data['SII Level'] = pd.Categorical(\n    sii_data['SII Level'],\n    categories=[\"None\", \"Mild\", \"Moderate\", \"Severe\", \"NaN\"],\n    ordered=True\n)\nsii_data = sii_data.sort_values('SII Level')\n\n# Calculate participants with non-null sii\ntotal_participants = len(train)\nnon_null_sii_count = train['sii'].notnull().sum()\nnull_sii_count = total_participants - non_null_sii_count\n\n# Data for the pie chart\nlabels = ['sii Not Null', 'sii Null']\nsizes = [non_null_sii_count, null_sii_count]\ncolors_pie = ['#4CAF50', '#FF6347']  # Green for not null, red for null\nexplode = (0.1, 0)  # Slightly explode the 'sii Not Null' slice\n\n# Create the pie chart\nplt.figure(figsize=(8, 8))\nplt.pie(\n    sizes,\n    labels=labels,\n    autopct=lambda p: f'{p:.1f}%\\n({non_null_sii_count if p > 50 else null_sii_count:,})', \n    colors=colors_pie,\n    explode=explode,\n    startangle=140,\n    textprops={'fontsize': 12}\n)\n\n# Title for the pie chart\nplt.title('Participants with sii Availability', fontsize=16)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:27.280219Z","iopub.execute_input":"2025-01-23T08:27:27.280438Z","iopub.status.idle":"2025-01-23T08:27:27.395135Z","shell.execute_reply.started":"2025-01-23T08:27:27.280419Z","shell.execute_reply":"2025-01-23T08:27:27.394396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"supervised_usable = train.dropna(subset=['sii'])\nmissing_count = supervised_usable.isnull().sum() # Series\nmissing_count_df = (missing_count.reset_index()) # dataframe\nmissing_count_df.columns = ['feature', 'null_count']\nmissing_count_df = missing_count_df.sort_values(by='null_count', ascending=False)\nmissing_count_df['null_ratio'] = missing_count_df['null_count']/len(supervised_usable)\n\nplt.figure(figsize=(6, 15))\nplt.title(f'Missing values over the {len(supervised_usable)} samples which have a target')\nplt.barh(np.arange(len(missing_count_df)), missing_count_df['null_ratio'], color='coral', label='missing')\nplt.barh(np.arange(len(missing_count_df)), \n         1 - missing_count_df['null_ratio'],\n         left=missing_count_df['null_ratio'],\n         color='darkseagreen', label='available')\nplt.yticks(np.arange(len(missing_count_df)), missing_count_df['feature'])\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0,1)\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:27.398221Z","iopub.execute_input":"2025-01-23T08:27:27.398443Z","iopub.status.idle":"2025-01-23T08:27:28.244540Z","shell.execute_reply.started":"2025-01-23T08:27:27.398425Z","shell.execute_reply":"2025-01-23T08:27:28.243652Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Data for the first plot\npciat_columns = [f\"PCIAT-PCIAT_{i:02}\" for i in range(1, 21)]\npciat_complete = temp[pciat_columns].notnull().all(axis=1)\ncomplete_count = pciat_complete.sum()\nincomplete_count = len(pciat_complete) - complete_count\n\n# Data for the first pie chart\nlabels1 = ['Complete PCIAT Data', 'Incomplete PCIAT Data']\ncounts1 = [complete_count, incomplete_count]\ncolors1 = ['#4CAF50', '#FF5722']\n\n# Data for the second plot\nnon_null_sii = temp[~temp['sii'].isnull()]\ncomplete_pciat_data = non_null_sii[pciat_columns].notnull().all(axis=1).sum()\ntotal_non_null_sii = len(non_null_sii)\nincomplete_pciat_data = total_non_null_sii - complete_pciat_data\n\n# Data for the second pie chart\nlabels2 = ['Complete PCIAT Data', 'Incomplete PCIAT Data']\ncounts2 = [complete_pciat_data, incomplete_pciat_data]\ncolors2 = ['#4CAF50', '#FF5722']\n\n# Create the merged plot\nfig, axes = plt.subplots(1, 2, figsize=(12, 6))\n\n# First pie chart\naxes[0].pie(\n    counts1,\n    labels=labels1,\n    autopct=lambda p: f'{p:.1f}%\\n({int(p*sum(counts1)/100):,})',\n    startangle=90,\n    colors=colors1,\n    textprops={'fontsize': 12}\n)\naxes[0].set_title(\"Completeness of PCIAT Data\", fontsize=16)\n\n# Second pie chart\naxes[1].pie(\n    counts2,\n    labels=labels2,\n    autopct=lambda p: f'{p:.1f}%\\n({int(p*sum(counts2)/100):,})',\n    startangle=90,\n    colors=colors2,\n    textprops={'fontsize': 12}\n)\naxes[1].set_title(\"Completeness of PCIAT Data\\nfor Participants with Non-Null 'sii'\", fontsize=14)\n\n# Adjust layout\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:28.246220Z","iopub.execute_input":"2025-01-23T08:27:28.246441Z","iopub.status.idle":"2025-01-23T08:27:28.436897Z","shell.execute_reply.started":"2025-01-23T08:27:28.246423Z","shell.execute_reply":"2025-01-23T08:27:28.436062Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print((train['PCIAT-PCIAT_Total'].isnull() == train['sii'].isnull()).mean())\naggregated_results = (train.groupby('sii')['PCIAT-PCIAT_Total'].agg(PCIAT_Total_min='min',\n                          PCIAT_Total_max='max',\n                          count='count')\n                     .reset_index()\n                     .sort_values('sii'))\naggregated_results","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:28.437655Z","iopub.execute_input":"2025-01-23T08:27:28.437864Z","iopub.status.idle":"2025-01-23T08:27:28.450772Z","shell.execute_reply.started":"2025-01-23T08:27:28.437846Z","shell.execute_reply":"2025-01-23T08:27:28.450001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate data for outer ring (SII completeness)\ntotal_participants = len(train)\nnull_sii = train['sii'].isnull().sum()\nnon_null_sii = total_participants - null_sii\n\n# Calculate data for inner ring (PCIAT completeness for non-null SII)\nnon_null_sii_df = train[~train['sii'].isnull()]\npciat_columns = [f\"PCIAT-PCIAT_{i:02}\" for i in range(1, 21)]\ncomplete_pciat_data = non_null_sii_df[pciat_columns].notnull().all(axis=1).sum()\nincomplete_pciat_data = len(non_null_sii_df) - complete_pciat_data\n\n# Print summary statistics\nprint(f\"Total participants: {total_participants}\")\nprint(f\"Participants with non-null 'sii': {non_null_sii}\")\nprint(f\"Participants with null 'sii': {null_sii} percentage of total participants : {null_sii/total_participants*100:.2f}%\")\nprint(f\"Participants with complete PCIAT (among non-null sii): {complete_pciat_data}\")\nprint(f\"Participants with incomplete PCIAT (among non-null sii): {incomplete_pciat_data}\")\nprint(f\"Percentage of non-null sii participants with complete PCIAT: {complete_pciat_data/non_null_sii*100:.2f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:28.451607Z","iopub.execute_input":"2025-01-23T08:27:28.451845Z","iopub.status.idle":"2025-01-23T08:27:28.461339Z","shell.execute_reply.started":"2025-01-23T08:27:28.451827Z","shell.execute_reply":"2025-01-23T08:27:28.460581Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Total participants: 3960\nParticipants with non-null 'sii': 2736\nParticipants with null 'sii': 1224 percentage of total participants : 30.91%\nParticipants with complete PCIAT (among non-null sii): 2671\nParticipants with incomplete PCIAT (among non-null sii): 65\nPercentage of non-null sii participants with complete PCIAT: 97.62%","metadata":{}},{"cell_type":"markdown","source":"For now, we can conclude that the SII score is sometimes incorrect, exactly for 65 participants.","metadata":{}},{"cell_type":"code","source":"temp = train.copy()\n\nsii_map = {0: '0 (None)', 1: '1 (Mild)', 2: '2 (Moderate)', 3: '3 (Severe)'}\ntemp['sii'] = temp['sii'].map(sii_map).fillna('Missing')\n\nsii_order = ['Missing', '0 (None)', '1 (Mild)', '2 (Moderate)', '3 (Severe)']\ntrain['sii'] = pd.Categorical(train['sii'], categories=sii_order, ordered=True)\nsii_order = ['Missing', '0 (None)', '1 (Mild)', '2 (Moderate)', '3 (Severe)']\ntemp['sii'] = pd.Categorical(temp['sii'], categories=sii_order, ordered=True)\n\nPCIAT_cols = [f'PCIAT-PCIAT_{i+1:02d}' for i in range(20)]\nrecalc_total_score = temp[PCIAT_cols].sum(\n    axis=1, skipna=True\n)\ntemp['complete_resp_total'] = temp['PCIAT-PCIAT_Total'].where(\n    temp[PCIAT_cols].notna().all(axis=1), np.nan\n)\nsii_counts = temp['sii'].value_counts().reset_index()\ntotal = sii_counts['count'].sum()\nsii_counts['percentage'] = (sii_counts['count'] / total) * 100\n\nfig, axes = plt.subplots(1, 2, figsize=(14, 5))\n\n# SII\nsns.barplot(x='sii', y='count', data=sii_counts, palette='Blues_d', ax=axes[0])\naxes[0].set_title('Distribution of Severity Impairment Index (sii)', fontsize=14)\nfor p in axes[0].patches:\n    height = p.get_height()\n    percentage = sii_counts.loc[sii_counts['count'] == height, 'percentage'].values[0]\n    axes[0].text(\n        p.get_x() + p.get_width() / 2,\n        height + 5, f'{int(height)} ({percentage:.1f}%)',\n        ha=\"center\", fontsize=12\n    )\n\n# PCIAT_Total for complete responses\nsns.histplot(temp['complete_resp_total'].dropna(), bins=20, ax=axes[1])\naxes[1].set_title('Distribution of PCIAT_Total', fontsize=14)\naxes[1].set_xlabel('PCIAT_Total for Complete PCIAT Responses')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:28.462195Z","iopub.execute_input":"2025-01-23T08:27:28.462480Z","iopub.status.idle":"2025-01-23T08:27:28.859992Z","shell.execute_reply.started":"2025-01-23T08:27:28.462453Z","shell.execute_reply":"2025-01-23T08:27:28.859180Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(temp[temp['complete_resp_total'] == 0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:28.860846Z","iopub.execute_input":"2025-01-23T08:27:28.861107Z","iopub.status.idle":"2025-01-23T08:27:28.867069Z","shell.execute_reply.started":"2025-01-23T08:27:28.861088Z","shell.execute_reply":"2025-01-23T08:27:28.866398Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Apparently, 40% of the participants were not affected by Internet use, 31% were not assessed, and only the minority (~10%) are moderately to severely impaired. There are 307 participants who scored 0 on all PCIAT questions.\n","metadata":{}},{"cell_type":"markdown","source":"**SII by age and sex**","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(2, 2, sharex='col', figsize=(12, 8))\n\nfor sex in range(2):\n    ax = axs[sex, 0]\n    vc = train_df[train_df['Basic_Demos-Sex'] == sex]['Basic_Demos-Age'].value_counts().sort_index()\n    ax.bar(vc.index, vc.values, color=['lightblue', 'coral'][sex], label=['boys', 'girls'][sex])\n    ax.set_ylabel('count')\n    ax.legend()\n\n# Title for Age Distribution column\naxs[0, 0].set_title('Age Distribution')\naxs[1, 0].set_xlabel('years')\naxs[1, 0].set_xticks(range(2, 23, 2))\n\nfor sex in range(2):\n    ax = axs[sex, 1]\n    vc = train_df[train_df['Basic_Demos-Sex'] == sex]['sii'].value_counts(normalize=True)\n    ax.bar(vc.index, vc.values, color=['lightblue', 'coral'][sex], label=['boys', 'girls'][sex])\n    ax.set_xticks(np.arange(4))\n    ax.set_xticklabels(target_labels)\n    ax.set_ylabel('proportion')\n    ax.legend()\n\n# Title for SII Distribution column\naxs[0, 1].set_title('Target Distribution')\naxs[1, 1].set_xlabel('Severity Impairment Index (sii)')\n\n# Overall title\nplt.suptitle('Combined Distribution Analysis')\n# Adjust layout\nplt.tight_layout(rect=[0, 0, 1, 0.96])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:28.867805Z","iopub.execute_input":"2025-01-23T08:27:28.868096Z","iopub.status.idle":"2025-01-23T08:27:29.630133Z","shell.execute_reply.started":"2025-01-23T08:27:28.868065Z","shell.execute_reply":"2025-01-23T08:27:29.629299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pciat_columns = [f\"PCIAT-PCIAT_{i:02}\" for i in range(1, 21)]\n# Filter the DataFrame where 'sii' is not NaN and all PCIAT columns are not NaN\nfiltered_train = train_df.dropna(subset=['sii'] + pciat_columns, how='any')\nsii_labels = {0: \"None\", 1: \"Mild\", 2: \"Moderate\", 3: \"Severe\"}\nfiltered_train['sii_label'] = filtered_train['sii'].map(sii_labels)\n\n# Set the order for the SII levels\nsii_order = [\"None\", \"Mild\", \"Moderate\", \"Severe\"]\nfiltered_train['sii_label'] = pd.Categorical(\n    filtered_train['sii_label'], \n    categories=sii_order, \n    ordered=True\n)\n# Ensure sii_label is categorized and ordered using .loc to avoid warnings\nfiltered_train.loc[:, 'sii_label'] = pd.Categorical(\n    filtered_train['sii_label'], \n    categories=[\"None\", \"Mild\", \"Moderate\", \"Severe\"], \n    ordered=True\n)\n\n# Plot\nplt.figure(figsize=(12, 6))\n\n# Create violin plot with box plot inside\nsns.violinplot(\n    data=filtered_train, \n    x='sii_label',\n    y='Basic_Demos-Age',\n    hue='Basic_Demos-Sex',\n    split=True,\n    inner='box',  # Shows box plot inside violin\n    palette=['#1E90FF', '#FF69B4'],  # Blue for boys, Pink for girls\n    order=[\"None\", \"Mild\", \"Moderate\", \"Severe\"]  # Ensure the correct order\n\n)\n\n# Customize the plot\nplt.title('Age Distribution by SII Severity Level and Sex', fontsize=16)\nplt.xlabel('SII Severity Level', fontsize=14)\nplt.ylabel('Age', fontsize=14)\n\n# Manually create a legend to fix the issue\nhandles = [\n    plt.Line2D([0], [0], color='#1E90FF', lw=4, label='Boys'),\n    plt.Line2D([0], [0], color='#FF69B4', lw=4, label='Girls')\n]\nplt.legend(handles=handles, title='Sex', fontsize=12, title_fontsize=12, loc='upper right')\n\n# Add grid for better readability\nplt.grid(axis='y', linestyle='--', alpha=0.7)\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:29.630912Z","iopub.execute_input":"2025-01-23T08:27:29.631219Z","iopub.status.idle":"2025-01-23T08:27:29.892397Z","shell.execute_reply.started":"2025-01-23T08:27:29.631190Z","shell.execute_reply":"2025-01-23T08:27:29.891646Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Older children tend to have more severe impairment, and the progression from \"None\" → \"Mild\" → \"Moderate\" → \"Severe\" shows a gradual increase in mean age, suggesting a potential time-dependent progression of impairment severity.","metadata":{}},{"cell_type":"code","source":"temp['Age Group'] = pd.cut(\n    temp['Basic_Demos-Age'],\n    bins=[4, 12, 18, 22],\n    labels=['Children (5-12)', 'Adolescents (13-18)', 'Adults (19-22)']\n)\n\nfig, axes = plt.subplots(1, 3, figsize=(18, 5))\n\n# SII by Age\nsns.boxplot(y=temp['Basic_Demos-Age'], x=temp['sii'], ax=axes[0], palette=\"Set3\")\naxes[0].set_title('SII by Age')\naxes[0].set_ylabel('Age')\naxes[0].set_xlabel('SII')\n\n# Complete PCIAT Responses by Age Group\nsns.boxplot(\n    x='Age Group', y='complete_resp_total',\n    data=temp, palette=\"Set3\", ax=axes[1]\n)\naxes[1].set_title('Complete PCIAT Responses by Age Group')\naxes[1].set_ylabel('PCIAT_Total for Complete Responses')\naxes[1].set_xlabel('Age Group')\n\n# PCIAT_Total by Sex\nsns.histplot(\n    data=temp, x='complete_resp_total',\n    hue='Basic_Demos-Sex', multiple='stack',\n    palette=\"Set3\", bins=20, ax=axes[2]\n)\naxes[2].set_title('PCIAT_Total Distribution by Sex')\naxes[2].set_xlabel('PCIAT_Total for Complete Responses')\naxes[2].set_ylabel('Frequency')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:29.893264Z","iopub.execute_input":"2025-01-23T08:27:29.893537Z","iopub.status.idle":"2025-01-23T08:27:30.505043Z","shell.execute_reply.started":"2025-01-23T08:27:29.893516Z","shell.execute_reply":"2025-01-23T08:27:30.504207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = temp.groupby(['Age Group', 'sii']).size().unstack(fill_value=0)\nfig, axes = plt.subplots(1, len(stats), figsize=(18, 5))\n\nfor i, age_group in enumerate(stats.index):\n    group_counts = stats.loc[age_group] / stats.loc[age_group].sum()\n    axes[i].pie(\n        group_counts, labels=group_counts.index, autopct='%1.1f%%',\n        startangle=90, colors=sns.color_palette(\"Set3\"),\n        labeldistance=1.05, pctdistance=0.80\n    )\n    axes[i].set_title(f'SII Distribution for {age_group}')\n    axes[i].axis('equal')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:30.506126Z","iopub.execute_input":"2025-01-23T08:27:30.506460Z","iopub.status.idle":"2025-01-23T08:27:31.104869Z","shell.execute_reply.started":"2025-01-23T08:27:30.506424Z","shell.execute_reply":"2025-01-23T08:27:31.104001Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The box plots are different representations of the target variable in it's categorised (SII) and numerical (PCIAT_Total) form. They show that higher SII scores are generally associated with older age groups, but there's considerable overlap in the age ranges within each category, and the median PCIAT_Total is higher in adolescents, suggesting a U-shaped relationship between age and PIU impairment (the peak of Internet-related problems may occur during adolescence).\nAccordingly, in the pie charts, the distribution of SII for children and adults is skewed towards lower values (none and mild), whereas, for adolescents, the distribution is more balanced across the categories of none, mild and moderate.\nThe overall distribution of SII is skewed towards lower values and severe cases are rare. So there may be relationships that we cannot see with such unequal sample sizes and under-representation of severe cases.\nThe differences between males and females are relatively subtle.","metadata":{}},{"cell_type":"markdown","source":"**Internet Use**","metadata":{}},{"cell_type":"markdown","source":"Internet usage data is crucial to this task because Problematic internet use (PIU), also known as internet addiction or compulsive internet use, refers to excessive and unhealthy use of the internet that interferes with a person’s daily life, responsibilities, and social relationships. The internet usage data provides a direct measure of how much time each participant spends online","metadata":{}},{"cell_type":"code","source":"data = temp[temp['PreInt_EduHx-computerinternet_hoursday'].notna()]\nage_range = data['Basic_Demos-Age']\nprint(\n    f\"Age range for participants with measured PreInt_EduHx-computerinternet_hoursday data:\"\n    f\" {age_range.min()} - {age_range.max()} years\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:31.105745Z","iopub.execute_input":"2025-01-23T08:27:31.106026Z","iopub.status.idle":"2025-01-23T08:27:31.112824Z","shell.execute_reply.started":"2025-01-23T08:27:31.105974Z","shell.execute_reply":"2025-01-23T08:27:31.112094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"param_map = {0: '< 1h/day', 1: '~ 1h/day', 2: '~ 2hs/day', 3: '> 3hs/day'}\ntemp['internet_use_encoded'] = temp[\n    'PreInt_EduHx-computerinternet_hoursday'\n].map(param_map).fillna('Missing')\n\nparam_ord = ['Missing', '< 1h/day', '~ 1h/day', '~ 2hs/day', '> 3hs/day']\ntemp['internet_use_encoded'] = pd.Categorical(\n    temp['internet_use_encoded'], categories=param_ord,\n    ordered=True\n)\n\nfig, axes = plt.subplots(1, 3, figsize=(18, 5))\n\n# Hours of Internet Use\nax1 = sns.countplot(x='internet_use_encoded', data=temp, palette=\"Set3\", ax=axes[0])\naxes[0].set_title('Distribution of Hours of Internet Use')\naxes[0].set_xlabel('Hours per Day Group')\naxes[0].set_ylabel('Count')\n\ntotal = len(temp['internet_use_encoded'])\nfor p in ax1.patches:\n    count = int(p.get_height())\n    percentage = '{:.1f}%'.format(100 * count / total)\n    ax1.annotate(f'{count} ({percentage})', (p.get_x() + p.get_width() / 2., p.get_height()), \n                 ha='center', va='baseline', fontsize=10, color='black', xytext=(0, 5), \n                 textcoords='offset points')\n\n# Hours of Internet Use by Age\nsns.boxplot(y=temp['Basic_Demos-Age'], x=temp['internet_use_encoded'], ax=axes[1], palette=\"Set3\")\naxes[1].set_title('Hours of Internet Use by Age')\naxes[1].set_ylabel('Age')\naxes[1].set_xlabel('Hours per Day Group')\n\n# Hours of Internet Use (numeric) by Age Group\nsns.boxplot(y='PreInt_EduHx-computerinternet_hoursday', x='Age Group', data=temp, ax=axes[2], palette=\"Set3\")\naxes[2].set_title('Internet Hours by Age Group')\naxes[2].set_ylabel('Hours per Day (Numeric)')\naxes[2].set_xlabel('Age Group')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:31.113770Z","iopub.execute_input":"2025-01-23T08:27:31.114141Z","iopub.status.idle":"2025-01-23T08:27:31.667752Z","shell.execute_reply.started":"2025-01-23T08:27:31.114110Z","shell.execute_reply":"2025-01-23T08:27:31.666867Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = temp.groupby(\n    ['Age Group', 'internet_use_encoded']\n).size().unstack(fill_value=0)\nfig, axes = plt.subplots(1, len(stats), figsize=(18, 5))\n\nfor i, age_group in enumerate(stats.index):\n    group_counts = stats.loc[age_group] / stats.loc[age_group].sum()\n    axes[i].pie(group_counts, labels=group_counts.index, autopct='%1.1f%%',\n                startangle=90, colors=sns.color_palette(\"Set3\"), labeldistance=1.1)\n    axes[i].set_title(f'Distribution of Hours of Internet Use\\n{age_group}')\n    axes[i].axis('equal')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:31.668682Z","iopub.execute_input":"2025-01-23T08:27:31.669114Z","iopub.status.idle":"2025-01-23T08:27:32.007799Z","shell.execute_reply.started":"2025-01-23T08:27:31.669078Z","shell.execute_reply":"2025-01-23T08:27:32.006646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"temp_non_na = temp.dropna(subset=['PreInt_EduHx-computerinternet_hoursday'])\nrows = (temp_non_na['PreInt_EduHx-computerinternet_hoursday'] == 3).sum()\nprint(f\"Non-NA Rows - Internet use 3h or more: {(rows / len(temp_non_na)) * 100:.2f}%\")\n\nrows = (temp_non_na['PreInt_EduHx-computerinternet_hoursday'] == 0).sum()\nprint(f\"Non-NA Rows - Internet use 1h or less: {(rows / len(temp_non_na)) * 100:.2f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:32.009294Z","iopub.execute_input":"2025-01-23T08:27:32.009652Z","iopub.status.idle":"2025-01-23T08:27:32.019521Z","shell.execute_reply.started":"2025-01-23T08:27:32.009608Z","shell.execute_reply":"2025-01-23T08:27:32.018663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = temp.groupby(['Basic_Demos-Sex', 'internet_use_encoded']\n).size().unstack(fill_value=0)\nstats_prop = stats.div(stats.sum(axis=1), axis=0) * 100\n\nstats = stats.astype(str) +' (' + stats_prop.round(1).astype(str) + '%)'\nstats","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:32.020265Z","iopub.execute_input":"2025-01-23T08:27:32.020464Z","iopub.status.idle":"2025-01-23T08:27:32.045879Z","shell.execute_reply.started":"2025-01-23T08:27:32.020447Z","shell.execute_reply":"2025-01-23T08:27:32.045067Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Internet usage data is missing for 16.6% of participants, while 38.5% reported using the Internet less than hour a day.\nSimilar to the SII data, the box plots reveals that higher daily internet usage is associated with older age, with considerable overlap in age ranges within each internet usage category.\nThe pie charts for age groups are well aligned and shows the same.\nCreating an interaction feature between internet use and age could potentially be useful for modeling.\nInternet use is fairly similar for both sexes.","metadata":{}},{"cell_type":"markdown","source":"**Internet usage vs SII (target)**","metadata":{}},{"cell_type":"code","source":"sii_reported = temp[temp['sii'] != \"Missing\"]\nsii_reported.loc[:, 'sii'] = sii_reported['sii'].cat.remove_unused_categories()\n\nfig = plt.figure(figsize=(12, 10))\ngs = fig.add_gridspec(2, 2, height_ratios=[1, 1.5])\n\n# SII vs Hours of Internet Use\nax1 = fig.add_subplot(gs[0, 0])\nsns.boxplot(\n    x='sii', y='PreInt_EduHx-computerinternet_hoursday',\n    data=sii_reported,\n    ax=ax1, palette=\"Set3\"\n)\nax1.set_title('SII vs Hours of Internet Use')\nax1.set_ylabel('Hours per Day')\nax1.set_xlabel('SII')\n\n# PCIAT_Total for Complete PCIAT Responses by Hours of Internet Use\nax2 = fig.add_subplot(gs[0, 1])\nsns.boxplot(\n    x='internet_use_encoded', y='complete_resp_total',\n    data=sii_reported,\n    palette=\"Set3\", ax=ax2\n)\nax2.set_title('PCIAT_Total by Hours of Internet Use')\nax2.set_ylabel('PCIAT_Total for Complete PCIAT Responses')\nax2.set_xlabel('Hours per Day Group')\n\n# SII vs Hours of Internet Use by Age Group (Full width)\nax3 = fig.add_subplot(gs[1, :])\nsns.boxplot(\n    x='internet_use_encoded', y='complete_resp_total',\n    data=sii_reported,\n    hue='Age Group', ax=ax3, palette=\"Set3\"\n)\nax3.set_title('PCIAT_Total vs Hours of Internet Use by Age Group')\nax3.set_ylabel('PCIAT_Total for Complete PCIAT Responses')\nax3.set_xlabel('Hours per Day Group')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:32.046790Z","iopub.execute_input":"2025-01-23T08:27:32.047152Z","iopub.status.idle":"2025-01-23T08:27:32.736812Z","shell.execute_reply.started":"2025-01-23T08:27:32.047132Z","shell.execute_reply":"2025-01-23T08:27:32.735834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = sii_reported.groupby(\n    ['sii', 'internet_use_encoded']\n).size().unstack(fill_value=0)\nfig, axes = plt.subplots(1, len(stats), figsize=(18, 5))\n\nfor i, sii_group in enumerate(stats.index):\n    group_counts = stats.loc[sii_group] / stats.loc[sii_group].sum()\n    axes[i].pie(\n        group_counts, labels=group_counts.index, autopct='%1.1f%%',\n        startangle=90, colors=sns.color_palette(\"Set3\"), labeldistance=1.1\n    )\n    axes[i].set_title(f'Hours of using computer/internet\\n for SII = {sii_group}')\n    axes[i].axis('equal')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:32.737689Z","iopub.execute_input":"2025-01-23T08:27:32.737975Z","iopub.status.idle":"2025-01-23T08:27:33.139063Z","shell.execute_reply.started":"2025-01-23T08:27:32.737954Z","shell.execute_reply":"2025-01-23T08:27:33.138258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = sii_reported.groupby(\n    ['sii', 'internet_use_encoded']\n).size().unstack(fill_value=0)\nstats_prop = stats.div(stats.sum(axis=1), axis=0) * 100\n\nstats = stats.astype(str) +' (' + stats_prop.round(1).astype(str) + '%)'\nstats","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:33.140137Z","iopub.execute_input":"2025-01-23T08:27:33.140466Z","iopub.status.idle":"2025-01-23T08:27:33.157908Z","shell.execute_reply.started":"2025-01-23T08:27:33.140435Z","shell.execute_reply":"2025-01-23T08:27:33.157150Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"In the box plots, despite the considerable overlap between the different SII and internet use categories, we see a positive trend between PIU impairment and internet use, with people with higher SII scores spending more time online (it would be strange if this wasn't the case, as excessive internet use is assumed by the PIU definition).\nHowever, when the relationship between PCIAT_Total and hours of Internet use is further broken down by age group (bottom boxplot), the non-linear relationship between age, Internet use and PIU emerges, with adolescents standing out as the most affected age group across all categories of Internet use.\nThe pie charts also show that there is a significant proportion of participants (83 in total), of all ages, who spend very little time online (less than 1 hour per day) but have high SII scores (20.7% with SII 2 - moderately impaired and 14.7% with SII = 3 - severely impaired).","metadata":{}},{"cell_type":"markdown","source":"Summary of Findings\nThe SII scores tend to increase with age but show a U-shaped relationship, with adolescents having the highest median PCIAT scores.\nThe higher the age, the more hours participants spent online (clear linear trend).\nPeople with higher SII scores generally spend more time online, but adolescents stand out as the most affected age group across all categories of internet use.\nThere are participants of almost all ages (5 to 21) who spend less than an hour a day online and have high SII scores.\nNote: These results should be interpreted with caution, as there is considerable overlap between the different SII and internet use categories, and severe cases and adults are under-represented in the data.","metadata":{}},{"cell_type":"markdown","source":"Interpretation\nThe relationship between SII and internet usage time is not as straightforward as one might expect, given the definition of PIU (aka internet addiction). This is because factors beyond just the hours spent online contribute to the SII, and the above analysis suggest that the age is particularly crucial: adolescents appear to have the highest SII across all levels of internet use... But how to interpret this: are they more susceptible to PIU, or is this questionnaire just more sensitive to PIU in this age group? Let's try to understand what exacly our target variable reflect.\n\nThe questions in the PCIAT questionnaire (used to derive the SII, see data_dict.csv) appear to be designed to measure emotional and social impacts associated with internet use (emotional dependence on the internet, social isolation, neglect of responsibilities, and the impact of internet use on relationships and mood). In other words, the intend was to measure show how problematic the behaviour associated with internet use is. However, parents' perceptions are naturally biased and influenced by various factors - such as their own internet habits, cultural attitudes, or their wishes/ideas about how their children should behave.\n\nFurthermore, can you imagine that spending less than an hour a day online could in itself lead to problems such as emotional distress, neglect of duties or withdrawal from family? I don't think an hour a day of any content on the internet can lead to any of those things... The presence of this in the data only proves that respondents are not being honest in answering the PCIAT questions and internet use, or that SII scores are being influenced by other factors that have nothing to do with the PIU.\n\nThe former point out that the single feature that links all the other data we have (physical activity, accelerometer data, sleep, etc.) to internet use (and we need this connection to predict the impact of PIU) may be unreliable and biased as well as the target variable.\n\nThe latter imply that participants may have various pre-existing social behaviours or moods that are unrelated to Internet use (and PIU) per se. The questionnaire is a subjective tool, even when completed by parents, while adolescence is an outstanding period in life - a time of identity formation, evolving peer relationships, and a search for independence - all of which can amplify behaviors like mood swings, disobedience, and impulsivity. Consequently, the SII may be capturing how problematic are these broader developmental behaviors rather than internet use specifically.\n\nAdditionally, the applicability of the questionnaire across the age range is questionable. I think all the questions in the PCIAT are much more suitable for adolescents. For example:\n\nA 5-7-year-old may not have household chores, as this depends on cultural norms.\nThe question about academic impact may not apply to younger children who are not yet in school or graduated adults.\nEmail use and receiving phone calls from \"online friends\" seem out of context for young children too.\nQuestions about reaction to the time allowed to spend on the internet (there are at least 3 of them) are not applicable to adults... usually.\nQuestions not applicable to a participant’s age could lead to skewed or irrelevant responses. All of these challenges the construct validity of SII and considers whether it accurately measures PIU or is affected by other behavioral factors.","metadata":{}},{"cell_type":"markdown","source":"**Features EDA by Groups**","metadata":{}},{"cell_type":"markdown","source":"Here’s how we can classify types of the features in this dataset:\n\nCategorical: Variables with discrete categories but no inherent order (represented as strings, e.g., season of enrollment)\nEncoded categorical features (already encoded as integers, e.g. sex)\nContinuous: Variables that can take any value within a range (e.g., age, enmo, heart_rate).\nOrdinal: Variables with a defined order but not necessarily equidistant categories (e.g., questionnaire responses).","metadata":{}},{"cell_type":"markdown","source":"Behavioral (subjective reported):\n\nA person can't have PIU if they don't use the internet, so I would expect PreInt_EduHx-computerinternet_hoursday to be the most important feature, but as we saw above, its relationship with the target can be non-linear.\nBehavioural tendencies associated with PIU may be reflected in the physical activity score derived from the questionnaires (PAQ_A-PAQ_A_Total and PAQ_C-PAQ_C_Total).\nHowever, both features are self-reports and are likely to be biased and inaccurate, so I would expect noise here.\n\nPhysical Health and Fitness (objective measurements):\n\nThe Children's Global Assessment Scale (CGAS-CGAS_Score) is a clinician-rated score reflecting general functioning. For individuals with PIU, this score can indicate how PIU impacts overall functioning.\nPhysical health measures include body composition and vital signs (feature columns starting with Physical-), and may reflect how problematic internet use is in terms of its impact on general health (note that height alone may not be as relevant, but combined with weight it gives BMI - a measure of body fat).\nBio-electric Impedance Analysis assess body composition and metabolic health (body fat, muscle mass, water content, metabolic rate, etc.), PIU, if assosiated with sedentary behavior could be reflected through changes in these variables (lower bone density, lower lean muscle mass, reduced daily energy expenditure, poor hydration, decrease in fat-free mass, higher body fat percentages, and so on).\nObjective measures of physical activity include FitnessGram results (endurance, curl, grip, push-up, sit & reach, trunk lift - feature columns starting with Fitness_ and FGC-FGC_). These can indicate how problematic internet use is in terms of its impact on muscle strength and tonus.\nAn assessment of sleep-related issues (feature columns SDS-SDS_Total_Raw, SDS-SDS_Total_T) could reflect the extent to which PIU disrupts sleep patterns.\nDemographic features:\nAge and gender can be extremely important, as there may be gender and especially age-specific patterns (as we have already seen above) associated with Internet use and PIU)","metadata":{}},{"cell_type":"code","source":"groups = data_dict.groupby('Instrument')['Field'].apply(list).to_dict()\n\ndata_dict = data_dict[data_dict['Instrument'] != 'Parent-Child Internet Addiction Test']\ncontinuous_cols = data_dict[data_dict['Type'].str.contains(\n    'float|int', case=False\n)]['Field'].tolist()\n\ngroups.get('Demographics', [])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:33.158632Z","iopub.execute_input":"2025-01-23T08:27:33.158965Z","iopub.status.idle":"2025-01-23T08:27:33.168662Z","shell.execute_reply.started":"2025-01-23T08:27:33.158935Z","shell.execute_reply":"2025-01-23T08:27:33.167868Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Season-related columns**","metadata":{}},{"cell_type":"code","source":"season_columns = [col for col in temp.columns if 'Season' in col]\nseason_df = temp[season_columns]\n\ntemp[season_columns] = temp[season_columns].fillna(\"Missing\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:33.169506Z","iopub.execute_input":"2025-01-23T08:27:33.169789Z","iopub.status.idle":"2025-01-23T08:27:33.187644Z","shell.execute_reply.started":"2025-01-23T08:27:33.169763Z","shell.execute_reply":"2025-01-23T08:27:33.186879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize=(18, 6))\n\n# Season of Enrollment\nseason_counts = train['Basic_Demos-Enroll_Season'].value_counts(dropna=False)\naxes[0].pie(\n    season_counts, labels=season_counts.index,\n    autopct='%1.1f%%', startangle=90,\n    colors=sns.color_palette(\"Set3\")\n)\naxes[0].set_title('Season of Enrollment')\naxes[0].axis('equal')\n\n# Age Distribution by Sex\nsns.histplot(\n    data=train, x='Basic_Demos-Age',\n    hue='Basic_Demos-Sex', multiple='dodge',\n    palette=\"Set2\", bins=20, ax=axes[1]\n)\naxes[1].set_title('Age Distribution by Sex')\naxes[1].set_xlabel('Age')\naxes[1].set_ylabel('Count')\n\n# Sex of Participant (third plot)\nvc = train['Basic_Demos-Sex'].value_counts() \naxes[2].pie(\n    vc, labels=vc.index.map({0: 'Male', 1: 'Female'}), \n    autopct='%1.1f%%', startangle=90,\n    colors=sns.color_palette(\"Pastel1\")\n)\naxes[2].set_title('Sex of Participant')\naxes[2].axis('equal')\n\n# Adjust layout to prevent overlap\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:33.188456Z","iopub.execute_input":"2025-01-23T08:27:33.188725Z","iopub.status.idle":"2025-01-23T08:27:33.663628Z","shell.execute_reply.started":"2025-01-23T08:27:33.188698Z","shell.execute_reply":"2025-01-23T08:27:33.662886Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The distribution of enrollment by season is relatively balanced, with the highest enrollment in Spring (28.5%) and the lowest in Fall (21.9%).\nThere is a higher number of males (0) across most age groups, with fewer females (1) particularly visible in younger age groups.\nThe relationships with the target variable are shown in the section 'SII by age and sex' (no difference in SII between men and women, U-shaped relationship between age and PIU impairment).","metadata":{}},{"cell_type":"markdown","source":"**- Children's Global Assessment Scale**","metadata":{}},{"cell_type":"code","source":"groups.get(\"Children's Global Assessment Scale\", [])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:33.664715Z","iopub.execute_input":"2025-01-23T08:27:33.665035Z","iopub.status.idle":"2025-01-23T08:27:33.669693Z","shell.execute_reply.started":"2025-01-23T08:27:33.664990Z","shell.execute_reply":"2025-01-23T08:27:33.669080Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = temp[temp['CGAS-CGAS_Score'].notnull()]\nage_range = data['Basic_Demos-Age']\nprint(\n    f\"Age range for participants with CGAS-CGAS_Score data:\"\n    f\" {age_range.min()} - {age_range.max()} years\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:33.670392Z","iopub.execute_input":"2025-01-23T08:27:33.670681Z","iopub.status.idle":"2025-01-23T08:27:33.685377Z","shell.execute_reply.started":"2025-01-23T08:27:33.670653Z","shell.execute_reply":"2025-01-23T08:27:33.684583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"temp[temp['CGAS-CGAS_Score'] > 100]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:33.686206Z","iopub.execute_input":"2025-01-23T08:27:33.686510Z","iopub.status.idle":"2025-01-23T08:27:33.731702Z","shell.execute_reply.started":"2025-01-23T08:27:33.686483Z","shell.execute_reply":"2025-01-23T08:27:33.731074Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There is one extreme value outlier (CGAS-CGAS_Score = 999), which is obviously an error.","metadata":{}},{"cell_type":"code","source":"train.loc[train['CGAS-CGAS_Score'] == 999, 'CGAS-CGAS_Score'] = np.nan\ntemp.loc[temp['CGAS-CGAS_Score'] == 999, 'CGAS-CGAS_Score'] = np.nan\n\nplt.figure(figsize=(12, 5))\n\n# CGAS-Season\nplt.subplot(1, 2, 1)\ncgas_season_counts = temp['CGAS-Season'].value_counts(normalize=True)\nplt.pie(\n    cgas_season_counts, \n    labels=cgas_season_counts.index, \n    autopct='%1.1f%%', \n    startangle=90, \n    colors=sns.color_palette(\"Set3\")\n)\nplt.title('CGAS-Season')\nplt.axis('equal')\n\n# CGAS-CGAS_Score without outliers (score == 999)\nplt.subplot(1, 2, 2)\nsns.histplot(\n    temp['CGAS-CGAS_Score'].dropna(),\n    bins=20, kde=True\n)\nplt.title('CGAS-CGAS_Score (Without Outlier)')\nplt.xlabel('CGAS Score')\nplt.ylabel('Count')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:33.732257Z","iopub.execute_input":"2025-01-23T08:27:33.732456Z","iopub.status.idle":"2025-01-23T08:27:34.052717Z","shell.execute_reply.started":"2025-01-23T08:27:33.732438Z","shell.execute_reply":"2025-01-23T08:27:34.051864Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**CGAS Interpretation**","metadata":{}},{"cell_type":"markdown","source":"CGAS is a rating of general functioning for children and young people aged 4-16 years old. The CGAS asks the clinician to rate the child from 1 to 100 based on their lowest level of functioning, regardless of treatment or prognosis, over a specified time period.\n\nSince the CGAS is a measure of general functioning, and the SII reflects the severity of the impact of Internet use on that functioning, I expect this feature, along with Internet use, to be the most important in predicting the SII.\n\nLet's bin the CGAS-CGAS_Score column based on the established score categories and draw counts:","metadata":{}},{"cell_type":"code","source":"bins = np.arange(0, 101, 10)\nlabels = [\n    \"1-10: Needs constant supervision (24 hour care)\",\n    \"11-20: Needs considerable supervision\",\n    \"21-30: Unable to function in almost all areas\",\n    \"31-40: Major impairment in functioning in several areas\",\n    \"41-50: Moderate degree of interference in functioning\",\n    \"51-60: Variable functioning with sporadic difficulties\",\n    \"61-70: Some difficulty in a single area\",\n    \"71-80: No more than slight impairment in functioning\",\n    \"81-90: Good functioning in all areas\",\n    \"91-100: Superior functioning\"\n]\n\ntemp['CGAS_Score_Bin'] = pd.cut(\n    temp['CGAS-CGAS_Score'], bins=bins, labels=labels\n)\n\ncounts = temp['CGAS_Score_Bin'].value_counts().reindex(labels)\nprop = (counts / counts.sum() * 100).round(1)\ncount_prop_labels = counts.astype(str) + \" (\" + prop.astype(str) + \"%)\"\n\nplt.figure(figsize=(18, 6))\nbars = plt.barh(labels, counts)\nplt.xlabel('Count')\nplt.title('CGAS Score Distribution')\n\nfor bar, label in zip(bars, count_prop_labels):\n    plt.text(\n        bar.get_width(), bar.get_y() + bar.get_height() / 2, label, va='center'\n    )\n\nplt.gca().invert_yaxis()\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:34.053568Z","iopub.execute_input":"2025-01-23T08:27:34.053814Z","iopub.status.idle":"2025-01-23T08:27:34.320838Z","shell.execute_reply.started":"2025-01-23T08:27:34.053785Z","shell.execute_reply":"2025-01-23T08:27:34.320022Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The majority of individuals have CGAS scores between 51-80 (79.7%), i.e. sporadic difficulties to only slight impairments\nTwo participants have extreme difficulty in functioning","metadata":{}},{"cell_type":"code","source":"temp_filt = temp.dropna(subset=['CGAS_Score_Bin', 'complete_resp_total'])\ntemp_filt.loc[:, 'CGAS_Score_Bin'] = temp_filt['CGAS_Score_Bin'].cat.remove_unused_categories()\ntemp_filt.loc[:, 'sii'] = temp_filt['sii'].cat.remove_unused_categories()\n\nfig, axes = plt.subplots(1, 2, figsize=(16, 5))\n\n# CGAS-CGAS_Score vs sii\nsns.boxplot(\n    data=temp_filt,\n    x='sii', y='CGAS-CGAS_Score',\n    palette='Set3', ax=axes[0]\n)\naxes[0].set_xlabel('SII Score')\naxes[0].set_ylabel('CGAS Score')\naxes[0].set_title('Distribution of CGAS Scores by SII')\n\n# complete_resp_total vs CGAS_Score_Bin\nsns.boxplot(\n    data=temp_filt,\n    x='CGAS_Score_Bin', y='complete_resp_total',\n    ax=axes[1], palette='Set3'\n)\n\n# Get the tick positions and match the labels\nrange_labels = [label.split(\":\")[0] for label in temp_filt['CGAS_Score_Bin'].cat.categories]\naxes[1].set_xticklabels(range_labels)\n\naxes[1].set_xlabel('CGAS Score category')\naxes[1].set_ylabel('PCIAT_Total for Complete PCIAT Responses')\naxes[1].set_title('Distribution of PCIAT_Total by CGAS Score categories')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:34.321587Z","iopub.execute_input":"2025-01-23T08:27:34.321804Z","iopub.status.idle":"2025-01-23T08:27:34.722287Z","shell.execute_reply.started":"2025-01-23T08:27:34.321786Z","shell.execute_reply":"2025-01-23T08:27:34.721576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"score_min_max = temp.groupby('sii')['CGAS-CGAS_Score'].agg(['min', 'max'])\nscore_min_max = score_min_max.rename(\n    columns={'min': 'Minimum CGAS Score', 'max': 'Maximum CGAS Score'}\n)\nscore_min_max","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:34.728915Z","iopub.execute_input":"2025-01-23T08:27:34.729153Z","iopub.status.idle":"2025-01-23T08:27:34.739742Z","shell.execute_reply.started":"2025-01-23T08:27:34.729133Z","shell.execute_reply":"2025-01-23T08:27:34.739085Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let's check the SII and Internet usage data for the participants with the worst and best global functioning:","metadata":{}},{"cell_type":"code","source":"temp_filt[temp_filt['CGAS-CGAS_Score'] < 35][\n    ['Basic_Demos-Age', 'Basic_Demos-Sex', 'sii',\n     'CGAS-CGAS_Score',\n     'PreInt_EduHx-computerinternet_hoursday']\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:34.742825Z","iopub.execute_input":"2025-01-23T08:27:34.743124Z","iopub.status.idle":"2025-01-23T08:27:34.753212Z","shell.execute_reply.started":"2025-01-23T08:27:34.743103Z","shell.execute_reply":"2025-01-23T08:27:34.752404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"temp[temp['CGAS-CGAS_Score'] > 90][\n    ['Basic_Demos-Age', 'Basic_Demos-Sex', 'sii',\n     'CGAS-CGAS_Score',\n     'PreInt_EduHx-computerinternet_hoursday']\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:34.754101Z","iopub.execute_input":"2025-01-23T08:27:34.754312Z","iopub.status.idle":"2025-01-23T08:27:34.774098Z","shell.execute_reply.started":"2025-01-23T08:27:34.754282Z","shell.execute_reply":"2025-01-23T08:27:34.773461Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"I would expect the higher the SII, the lower the median CGAS score, but the decrease is very small here.\nHowever, there are no participants with the highest SII scores (3 or severely problematic internet use) who have good CGAS scores (81-100: good/superior functioning in all domains). This suggests that parental responses to the PCIAT questionnaire (our target variable) may reflect some effects of PIU on global health and functioning.\nThe participants with the worst and best CGAS scores all have SII 0 or 1 (no or mild PIU severity) and report varying internet use (less than 1 hours/day to 3 or more hours/day). This means in the train data there are participants with significant health issues not related to PIU.\nThe high variability makes it hard to draw a clear, consistent conclusion about the relationship between CGAS and SII scores.\nSmall sample sizes in certain CGAS categories make it difficult to generalize findings and may lead to biased interpretations.","metadata":{}},{"cell_type":"markdown","source":"**Physical Measures**","metadata":{}},{"cell_type":"code","source":"groups.get('Physical Measures', [])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:34.774876Z","iopub.execute_input":"2025-01-23T08:27:34.775148Z","iopub.status.idle":"2025-01-23T08:27:34.789477Z","shell.execute_reply.started":"2025-01-23T08:27:34.775106Z","shell.execute_reply":"2025-01-23T08:27:34.788842Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_physical = groups.get('Physical Measures', [])\ncols = [col for col in features_physical if col in continuous_cols]\n\nplt.figure(figsize=(24, 10))\nn_cols = 4\nn_rows = len(cols) // n_cols + 1\n\nfor i, col in enumerate(cols):\n    plt.subplot(n_rows, n_cols, i + 1)\n    temp[col].hist(bins=20)\n    plt.title(col)\n\nplt.subplot(n_rows, n_cols, len(cols) + 1)\nseason_counts = temp['Physical-Season'].value_counts(dropna=False)\nplt.pie(\n    season_counts,\n    labels=season_counts.index,\n    autopct='%1.1f%%',\n    startangle=90,\n    colors=sns.color_palette(\"Set3\")\n)\nplt.title('Physical-Season')\n\nplt.suptitle('Histograms for Physical Measures and Physical-Season Pie Chart', y=1.05)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:34.790131Z","iopub.execute_input":"2025-01-23T08:27:34.790328Z","iopub.status.idle":"2025-01-23T08:27:36.257357Z","shell.execute_reply.started":"2025-01-23T08:27:34.790301Z","shell.execute_reply":"2025-01-23T08:27:36.256523Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Weight and Height**","metadata":{}},{"cell_type":"code","source":"wh_cols = [\n    'Physical-BMI', 'Physical-Height',\n    'Physical-Weight', 'Physical-Waist_Circumference'\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:36.258238Z","iopub.execute_input":"2025-01-23T08:27:36.258523Z","iopub.status.idle":"2025-01-23T08:27:36.261779Z","shell.execute_reply.started":"2025-01-23T08:27:36.258502Z","shell.execute_reply":"2025-01-23T08:27:36.261086Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The minimum values of 0 for measures like BMI, weight, and blood pressure are biologically unrealistic, and likely indicate missing or erroneous data. Let's check number of zeros in these columns:","metadata":{}},{"cell_type":"code","source":"(train[wh_cols] == 0).sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:36.262437Z","iopub.execute_input":"2025-01-23T08:27:36.262616Z","iopub.status.idle":"2025-01-23T08:27:36.278957Z","shell.execute_reply.started":"2025-01-23T08:27:36.262600Z","shell.execute_reply":"2025-01-23T08:27:36.278201Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Replace the 0 values by NaN","metadata":{}},{"cell_type":"code","source":"train[wh_cols] = train[wh_cols].replace(0, np.nan)\ntemp[wh_cols] = temp[wh_cols].replace(0, np.nan)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:36.279642Z","iopub.execute_input":"2025-01-23T08:27:36.279830Z","iopub.status.idle":"2025-01-23T08:27:36.292957Z","shell.execute_reply.started":"2025-01-23T08:27:36.279814Z","shell.execute_reply":"2025-01-23T08:27:36.292337Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Convert weight to kilograms, and height to centimeters and recalculate BMI:","metadata":{}},{"cell_type":"code","source":"lbs_to_kg = 0.453592\ninches_to_cm = 2.54\n\ntemp['Physical-Weight'] = temp['Physical-Weight'] * lbs_to_kg\ntemp['Physical-Height'] = temp['Physical-Height'] * inches_to_cm\ntemp['Physical-Waist_Circumference'] = temp['Physical-Waist_Circumference'] * inches_to_cm\n\n# Recalculate BMI: BMI = weight (kg) / (height (m)^2)\ntemp['Physical-BMI'] = np.where(\n    temp['Physical-Weight'].notna() & temp['Physical-Height'].notna(),\n    temp['Physical-Weight'] / ((temp['Physical-Height'] / 100) ** 2),\n    np.nan  # If either is NaN, set BMI to NaN\n)\n\nprint(\"Physical-Weight max: \", temp['Physical-Weight'].max())\nprint(\"Physical-Weight min: \", temp['Physical-Weight'].min())\n\nprint(\"Physical-Height max: \", temp['Physical-Height'].max())\nprint(\"Physical-Height min: \", temp['Physical-Height'].min())\n\nprint(\"Physical-Waist_Circumference max: \", temp['Physical-Waist_Circumference'].max())\nprint(\"Physical-Waist_Circumference min: \", temp['Physical-Waist_Circumference'].min())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:36.293715Z","iopub.execute_input":"2025-01-23T08:27:36.293921Z","iopub.status.idle":"2025-01-23T08:27:36.310032Z","shell.execute_reply.started":"2025-01-23T08:27:36.293893Z","shell.execute_reply":"2025-01-23T08:27:36.309357Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"A lot of values seem to be out of normal ranges... especially max values of weight (142kg) and waist circumference (127cm).","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(18, 5))\n\n# Physical-Weight by Age\nplt.subplot(1, 3, 1)\nsns.scatterplot(x='Basic_Demos-Age', y='Physical-Weight', data=temp)\nplt.title('Physical-Weight by Age')\nplt.xlabel('Age')\nplt.ylabel('Weight (kg)')\n\n# Physical-Height by Age\nplt.subplot(1, 3, 2)\nsns.scatterplot(x='Basic_Demos-Age', y='Physical-Height', data=temp)\nplt.title('Physical-Height by Age')\nplt.xlabel('Age')\nplt.ylabel('Height (cm)')\n\n# Physical-Waist_Circumference vs Physical-Weight\nplt.subplot(1, 3, 3)\nsns.scatterplot(x='Physical-Weight', y='Physical-Waist_Circumference', data=temp)\nplt.title('Waist Circumference vs Weight')\nplt.xlabel('Weight (kg)')\nplt.ylabel('Waist Circumference (cm)')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:36.310713Z","iopub.execute_input":"2025-01-23T08:27:36.310896Z","iopub.status.idle":"2025-01-23T08:27:36.912706Z","shell.execute_reply.started":"2025-01-23T08:27:36.310881Z","shell.execute_reply":"2025-01-23T08:27:36.911871Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Weight and height both increase with age, and waist circumference and weight are highly correlated, as expected.\nHowever, there are individuals who are unusually tall for their age group or who are extremely overweight.\nThere are also a few outliers in the waist circumference measurements, which are possible artifacts (e.g. 100 cm for a weight of 40 kg).\nThe problem with data cleaning here is that we cannot guess which of the data is correct. For example, we may see an unrealistic combination of a waist circumference of 100cm and a weight of 40kg for a participant, but where is the error in the waist circumference or the weight? Or a height of around 175cm for a child of 7... has the height or age been entered incorrectly? Or this is true data and the child has gigantism or another disorder related to the growth hormone?","metadata":{}},{"cell_type":"markdown","source":"**Blood Pressure & Heart Rate**","metadata":{}},{"cell_type":"markdown","source":"There is 1000% incorrect data in the BP/HR columns as the minimum values are lethal to humans. We can clean up these kinds of mistakes.","metadata":{}},{"cell_type":"code","source":"bp_hr_cols = [\n    'Physical-Diastolic_BP', 'Physical-Systolic_BP',\n    'Physical-HeartRate'\n]\n\n(temp[bp_hr_cols] < 50).sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:36.913734Z","iopub.execute_input":"2025-01-23T08:27:36.914107Z","iopub.status.idle":"2025-01-23T08:27:36.921159Z","shell.execute_reply.started":"2025-01-23T08:27:36.914079Z","shell.execute_reply":"2025-01-23T08:27:36.920420Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We also know that systolic BP cannot be lower than diastolic BP:","metadata":{}},{"cell_type":"code","source":"temp[temp['Physical-Systolic_BP'] <= temp['Physical-Diastolic_BP']][bp_hr_cols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:36.921879Z","iopub.execute_input":"2025-01-23T08:27:36.922171Z","iopub.status.idle":"2025-01-23T08:27:36.940299Z","shell.execute_reply.started":"2025-01-23T08:27:36.922138Z","shell.execute_reply":"2025-01-23T08:27:36.939500Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"These are certainly incorrect measurements. But again, we can't be sure which information is correct, so we can either flag these rows for further manual inspection one by one, or replace all suspicious values with NaN. For this analysis I only remove 0 values and both BP if systolic is lower or equal to diastolic.","metadata":{}},{"cell_type":"code","source":"train[cols] = train[cols].replace(0, np.nan)\ntrain.loc[train['Physical-Systolic_BP'] <= train['Physical-Diastolic_BP'], bp_hr_cols] = np.nan\n\ntemp[cols] = temp[cols].replace(0, np.nan)\ntemp.loc[temp['Physical-Systolic_BP'] <= temp['Physical-Diastolic_BP'], bp_hr_cols] = np.nan","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:36.941027Z","iopub.execute_input":"2025-01-23T08:27:36.941239Z","iopub.status.idle":"2025-01-23T08:27:36.959677Z","shell.execute_reply.started":"2025-01-23T08:27:36.941221Z","shell.execute_reply":"2025-01-23T08:27:36.958929Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Blood Pressure vs Heart Rate**","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12, 5))\n\n# Diastolic BP vs Heart Rate\nplt.subplot(1, 2, 1)\nsns.scatterplot(x='Physical-Diastolic_BP', y='Physical-HeartRate', data=temp)\nplt.title('Diastolic BP vs Heart Rate')\nplt.xlabel('Diastolic Blood Pressure (mmHg)')\nplt.ylabel('Heart rate (beats/min)')\n\n# Systolic BP vs Heart Rate\nplt.subplot(1, 2, 2)\nsns.scatterplot(x='Physical-Systolic_BP', y='Physical-HeartRate', data=temp)\nplt.title('Systolic BP vs Heart Rate')\nplt.xlabel('Systolic Blood Pressure (mmHg)')\nplt.ylabel('Heart rate (beats/min)')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:36.960342Z","iopub.execute_input":"2025-01-23T08:27:36.960588Z","iopub.status.idle":"2025-01-23T08:27:37.356625Z","shell.execute_reply.started":"2025-01-23T08:27:36.960559Z","shell.execute_reply":"2025-01-23T08:27:37.355900Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The absence of a clear direct correlation between heart rate and blood pressure in the plots suggests that the measurements were likely taken in a resting state or under non-stressful conditions.","metadata":{}},{"cell_type":"markdown","source":"**Blood pressure vs Body Mass Index (BMI)**","metadata":{}},{"cell_type":"markdown","source":"Typically, systolic (SBP) and diastolic (DBP) blood pressure are positively correlated, as they both reflect the functioning of the cardiovascular system. However, there can be deviations:\n\nIsolated Systolic Hypertension: High SBP with normal DBP\nIsolated Diastolic Hypertension: Normal SBP with high DBP\nGeneral Hypertension: Both SBP and DBP are elevated\nBMI is often used as an indicator of overall body fat and can correlate with blood pressure (e.g. higher BMI values indicating overweight or obesity are commonly associated with elevated blood pressure). Let's see if this is true for the study participants.","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(16, 6))\n\n# BMI vs Systolic Blood Pressure\nsns.scatterplot(x='Physical-BMI', y='Physical-Systolic_BP', data=temp, ax=axes[0], color='b')\naxes[0].set_title('BMI vs Systolic Blood Pressure')\naxes[0].set_xlabel('Body Mass Index (BMI) (kg/m^2)')\naxes[0].set_ylabel('Systolic Blood Pressure (mmHg)')\n\n# Systolic Blood Pressure vs Diastolic Blood Pressure\nsns.scatterplot(\n    x='Physical-Systolic_BP', y='Physical-Diastolic_BP',\n    data=temp, ax=axes[1], color='g'\n)\naxes[1].set_title('Systolic Blood Pressure vs Diastolic Blood Pressure')\naxes[1].set_xlabel('Systolic Blood Pressure (mmHg)')\naxes[1].set_ylabel('Diastolic Blood Pressure (mmHg)')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:37.357577Z","iopub.execute_input":"2025-01-23T08:27:37.357796Z","iopub.status.idle":"2025-01-23T08:27:37.788700Z","shell.execute_reply.started":"2025-01-23T08:27:37.357778Z","shell.execute_reply":"2025-01-23T08:27:37.787811Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There does not appear to be a strong, clear correlation between body mass index (BMI) and systolic blood pressure (BP).\nAs expected, there is a strong positive correlation between systolic and diastolic BP, but there are notable cases of isolated systolic or diastolic hypertension (or errors in the data, who knows?)","metadata":{}},{"cell_type":"markdown","source":"**Relationships with the target variable (PCIAT_Total for complete PCIAT responses)**","metadata":{}},{"cell_type":"code","source":"data_subset = temp[cols + ['complete_resp_total']]\n\ncorr_matrix = data_subset.corr()\n\nplt.figure(figsize=(10, 8))\nsns.heatmap(corr_matrix, annot=True, cmap='coolwarm', fmt='.2f', vmin=-1, vmax=1)\nplt.title('Correlation Heatmap')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:37.789588Z","iopub.execute_input":"2025-01-23T08:27:37.789825Z","iopub.status.idle":"2025-01-23T08:27:38.499819Z","shell.execute_reply.started":"2025-01-23T08:27:37.789805Z","shell.execute_reply":"2025-01-23T08:27:38.498992Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The positive correlation with the target is for height, weight, and waist circumference, which means that taller and fatter people tend to have a higher SII. But as these physical parameters increase with age, and we already know that SII tends to be highest in adolescents, this could indicate that they acts as a proxy for age (likely reflect age-related trends).\nCardiovascular measures (systolic blood pressure, diastolic blood pressure and heart rate) also change with age, but do not vary as drastically between childhood and adolescence as physical measures, and may not be as sensitive to behaviours such as internet use. They also have a higher degree of variability, as we saw in the graphs above, so the weak correlation may indicate that cardiovascular health is not strongly linked to PIU, or that these data are just more scattered and noisy and the relationship with PIU is diluted.","metadata":{}},{"cell_type":"markdown","source":"**Bio-electric Impedance Analysis**","metadata":{}},{"cell_type":"markdown","source":"There is no information in the competition description about what equipment was used, is this raw data or did they use some BIA equation models to estimate the parameters. But it's likely that the BIA data has already been processed using a BIA equation model. It is very important to note that BIA is not a precise method, for example it tends to overestimate muscle mass, so equations have been developed to estimate muscle mass based on factors such as age, sex, height, weight and resistance and/or reactance estimated by BIA... a large number of prediction equation models have been generated through various validation studies (link30478-4/fulltext)). It is essential that all recordings are processed with the same equation, but we cannot be sure.","metadata":{}},{"cell_type":"code","source":"bia_data_dict = data_dict[data_dict['Instrument'] == 'Bio-electric Impedance Analysis']\ncategorical_columns = bia_data_dict[bia_data_dict['Type'] == 'categorical int']['Field'].tolist()\ncontinuous_columns = bia_data_dict[bia_data_dict['Type'] == 'float']['Field'].tolist()\n\nfig, axes = plt.subplots(1, 3, figsize=(18, 5))\n\n# Season\nseason_counts = temp['BIA-Season'].value_counts(normalize=True)\naxes[0].pie(\n    season_counts, \n    labels=season_counts.index, \n    autopct='%1.1f%%', \n    startangle=90, \n    colors=sns.color_palette(\"Set3\")\n)\naxes[0].set_title(\n    f\"{bia_data_dict[bia_data_dict['Field'] == 'BIA-Season']['Description'].values[0]}\"\n)\naxes[0].axis('equal')\n\n# Other categorical columns\nfor idx, col in enumerate(categorical_columns):\n    sns.countplot(x=col, data=temp, palette=\"Set3\", ax=axes[idx+1])\n    axes[idx+1].set_title(data_dict[data_dict['Field'] == col]['Description'].values[0])\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:38.500849Z","iopub.execute_input":"2025-01-23T08:27:38.501179Z","iopub.status.idle":"2025-01-23T08:27:38.898368Z","shell.execute_reply.started":"2025-01-23T08:27:38.501150Z","shell.execute_reply":"2025-01-23T08:27:38.897546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(24, 20))\n\nfor idx, col in enumerate(continuous_columns):\n    plt.subplot(4, 4, idx + 1)\n    sns.histplot(temp[col].dropna(), bins=20, kde=True)\n    plt.title(data_dict[data_dict['Field'] == col]['Description'].values[0])\n    plt.xlabel('Value')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:38.899258Z","iopub.execute_input":"2025-01-23T08:27:38.899574Z","iopub.status.idle":"2025-01-23T08:27:41.896831Z","shell.execute_reply.started":"2025-01-23T08:27:38.899542Z","shell.execute_reply":"2025-01-23T08:27:41.895876Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The distribution of the various bioelectrical impedance analysis measurements in the data set indicates that most of them are not useful: highly skewed, with the majority of participants having marginal values and a few outliers (potential measurement errors).\nSome variables, such as Fat Mass Index and Body Fat Percentage, show implausible negative values, and almost all - extreme high values, indicating potential data quality issues","metadata":{}},{"cell_type":"markdown","source":"**Compare the two measured BMI**","metadata":{}},{"cell_type":"code","source":"bmi_data = temp[['BIA-BIA_BMI', 'Physical-BMI']].dropna()\n\nplt.figure(figsize=(8, 6))\nsns.scatterplot(\n    x='BIA-BIA_BMI', y='Physical-BMI',\n    data=bmi_data,\n    color='b'\n)\nplt.title('Comparison of BIA-BMI vs Physical-BMI')\nplt.xlabel('BIA-BMI')\nplt.ylabel('Physical-BMI')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:41.897676Z","iopub.execute_input":"2025-01-23T08:27:41.897922Z","iopub.status.idle":"2025-01-23T08:27:42.112405Z","shell.execute_reply.started":"2025-01-23T08:27:41.897900Z","shell.execute_reply":"2025-01-23T08:27:42.111546Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**FitnessGram Vitals and Treadmill**","metadata":{}},{"cell_type":"code","source":"groups.get('FitnessGram Vitals and Treadmill', [])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:42.113273Z","iopub.execute_input":"2025-01-23T08:27:42.113550Z","iopub.status.idle":"2025-01-23T08:27:42.118370Z","shell.execute_reply.started":"2025-01-23T08:27:42.113518Z","shell.execute_reply":"2025-01-23T08:27:42.117746Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = temp[temp['Fitness_Endurance-Max_Stage'].notnull()]\nage_range = data['Basic_Demos-Age']\nprint(\n    f\"Age range for participants with Fitness_Endurance-Max_Stage data:\"\n    f\" {age_range.min()} - {age_range.max()} years\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:42.119185Z","iopub.execute_input":"2025-01-23T08:27:42.119376Z","iopub.status.idle":"2025-01-23T08:27:42.135209Z","shell.execute_reply.started":"2025-01-23T08:27:42.119360Z","shell.execute_reply":"2025-01-23T08:27:42.134511Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 4, figsize=(24, 5))\n\n# Fitness Endurance Season\ntemp['Fitness_Endurance-Season'].value_counts(normalize=True).plot.pie(\n    autopct='%1.1f%%', colors=plt.cm.Set3.colors, ax=axes[0]\n)\naxes[0].set_title('Fitness Endurance Season')\naxes[0].axis('equal')  # Equal aspect ratio ensures the pie is drawn as a circle.\n\n# Box plot for Max Stage by Season\nsns.violinplot(\n    x='Fitness_Endurance-Season',\n    y='Fitness_Endurance-Max_Stage',\n    data=temp, palette=\"Set3\",\n    ax=axes[1]\n)\naxes[1].set_title('Max Stage by Season')\naxes[1].set_xlabel('Season')\naxes[1].set_ylabel('Max Stage')\n\n# Fitness Endurance Time (Minutes)\nsns.histplot(temp['Fitness_Endurance-Time_Mins'], bins=20, kde=True, ax=axes[2])\naxes[2].set_title('Fitness Endurance Time (Minutes)')\naxes[2].set_xlabel('Time (Minutes)')\n\n# Fitness Endurance Time (Seconds)\nsns.histplot(temp['Fitness_Endurance-Time_Sec'], bins=20, kde=True, ax=axes[3])\naxes[3].set_title('Fitness Endurance Time (Seconds)')\naxes[3].set_xlabel('Time (Seconds)')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:42.135915Z","iopub.execute_input":"2025-01-23T08:27:42.136183Z","iopub.status.idle":"2025-01-23T08:27:42.887528Z","shell.execute_reply.started":"2025-01-23T08:27:42.136164Z","shell.execute_reply":"2025-01-23T08:27:42.886682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 5))\n\nsns.violinplot(x='Basic_Demos-Age', y='Fitness_Endurance-Max_Stage', data=temp, palette=\"Set3\")\nplt.title('Fitness Endurance Max Stage by Age')\nplt.xlabel('Age')\nplt.ylabel('Max Stage')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:42.888543Z","iopub.execute_input":"2025-01-23T08:27:42.888858Z","iopub.status.idle":"2025-01-23T08:27:43.199107Z","shell.execute_reply.started":"2025-01-23T08:27:42.888827Z","shell.execute_reply":"2025-01-23T08:27:43.198238Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Fitness_Endurance-Max_Stage: likely represents the maximum stage reached during an endurance test. In fitness endurance tests like a treadmill test or a multi-stage fitness test (beep test), participants progress through increasing levels of difficulty (speed or incline), and this column records the highest level or stage completed by the participant before stopping.\nFitness_Endurance-Time_Mins: could be the duration a participant was able to sustain the test before reaching exhaustion, measured in minutes\nFitness_Endurance-Time_Sec: I guess combining both columns (minutes and seconds) would give the exact total time of the endurance test completed by the participants.","metadata":{}},{"cell_type":"markdown","source":"On average, participants reached stage 5 in the endurance test.\nSome participants failed to complete the first stage (min = 0), or these are errors in data again.\nThere is a small number of participants with exceptionally high endurance of age 7-8 years.\nThere is a substantial amount of missing data (over 80% of the dataset lacks this information).","metadata":{}},{"cell_type":"markdown","source":"**FitnessGram Child**","metadata":{}},{"cell_type":"code","source":"fgc_data_dict = data_dict[data_dict['Instrument'] == 'FitnessGram Child']\n\nfgc_columns = []\n\nfor index, row in fgc_data_dict.iterrows():\n    if '_Zone' not in row['Field']:\n        measure_field = row['Field']\n        measure_desc = row['Description']\n        \n        zone_field = measure_field + '_Zone'\n        zone_row = fgc_data_dict[fgc_data_dict['Field'] == zone_field]\n        \n        if not zone_row.empty:\n            zone_desc = zone_row['Description'].values[0]\n            fgc_columns.append((measure_field, zone_field, measure_desc, zone_desc))\n            \nfig, axes = plt.subplots(2, 4, figsize=(24, 10))\n\nfor idx, (measure, zone, measure_desc, zone_desc) in enumerate(fgc_columns):\n    row = idx // 4\n    col = idx % 4\n    \n    sns.histplot(\n        data=temp, x=measure,\n        hue=zone, bins=20, palette='Set2',\n        ax=axes[row, col], kde=True\n    )\n    axes[row, col].set_title(f'{measure_desc}')\n\nseason_counts = temp['FGC-Season'].value_counts(normalize=True)\naxes[1, 3].pie(\n    season_counts, labels=season_counts.index,\n    autopct='%1.1f%%', startangle=90,\n    colors=sns.color_palette(\"Set3\")\n)\naxes[1, 3].set_title('Season of participation')\naxes[1, 3].axis('equal') \n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:43.199855Z","iopub.execute_input":"2025-01-23T08:27:43.200102Z","iopub.status.idle":"2025-01-23T08:27:45.285652Z","shell.execute_reply.started":"2025-01-23T08:27:43.200083Z","shell.execute_reply":"2025-01-23T08:27:45.284729Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Most of the distributions are skewed towards lower performance totals.\nStrangely, a greater proportion of participants achieved a healthy fitness zone for the trunk lift.\nI would expect different ranges for each zone, but the values for different zones overlap significantly. This may be because the zone ranges are different for different ages.","metadata":{}},{"cell_type":"markdown","source":"**Relationships with the target variable (PCIAT_Total for complete PCIAT responses)**","metadata":{}},{"cell_type":"code","source":"temp['Fitness_Endurance-Total_Time_Sec'] = temp[\n    'Fitness_Endurance-Time_Mins'\n] * 60 + temp['Fitness_Endurance-Time_Sec']\n\ncols = [col for col in temp.columns if col.startswith('FGC-') \n        and 'Zone' not in col and 'Season' not in col]\ncols.extend(['Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Total_Time_Sec'])\n\ndata_subset = temp[cols + ['complete_resp_total']]\n\ncorr_matrix = data_subset.corr()\n\nplt.figure(figsize=(10, 8))\nsns.heatmap(corr_matrix, annot=True, cmap='coolwarm', fmt='.2f', vmin=-1, vmax=1)\nplt.title('Correlation Heatmap')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:45.286495Z","iopub.execute_input":"2025-01-23T08:27:45.286750Z","iopub.status.idle":"2025-01-23T08:27:45.739174Z","shell.execute_reply.started":"2025-01-23T08:27:45.286729Z","shell.execute_reply":"2025-01-23T08:27:45.737986Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There are noticeable intercorrelations between the fitness measures (FGC-FGC_GSD (grip strength dominant) and FGC-FGC_GSND (grip strength non-dominant), FGC-FGC_SRL (sit & reach left) and FGC-FGC_SRR (sit & reach right)) and they are expected to be similar.\nThe relationships with the target variable appear to be a counterintuitive: curl-ups and push-ups show moderate positive relationships with PIU severity, and trunk lift and grip strength show a weak positive correlation, suggesting that physical performance improves as PIU severity increases...\nBetter performance in fitness tests does not necessarily indicate a higher level of daily physical activity. Besides, fitness measures might reflect past - we do not know the timing of the measurements.\nBut the main thing to remember here is that physical performance also improves with age, so the positive correlation between physical performance and PIU severity is likely just driven by age.\nAnd here is another unknown: were the fitness tests conducted in a standardized way across all participants?","metadata":{}},{"cell_type":"markdown","source":"Let's see how the picture changes when we plot the same thing by age group, and add age to see if the measures still correlate with age.","metadata":{}},{"cell_type":"code","source":"age_groups = temp['Age Group'].unique()\n\nfig, axes = plt.subplots(1, 3, figsize=(18, 6), sharey=True)\n\nfor i, age_group in enumerate(age_groups):\n    group_data = temp[temp['Age Group'] == age_group]\n    corr_matrix = group_data[cols + ['complete_resp_total', 'Basic_Demos-Age']].corr()\n    sns.heatmap(corr_matrix, annot=True, cmap='coolwarm', fmt='.1f',\n                vmin=-1, vmax=1, ax=axes[i], cbar=i == 0)\n    axes[i].set_title(f'{age_group}')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:45.739972Z","iopub.execute_input":"2025-01-23T08:27:45.740238Z","iopub.status.idle":"2025-01-23T08:27:47.422238Z","shell.execute_reply.started":"2025-01-23T08:27:45.740216Z","shell.execute_reply":"2025-01-23T08:27:47.421242Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"In each age group we see that age correlates well with most measures of physical performance (especially for kids and adults).\nThe correlation between age and PIU severity persists in children aged 5-12 years, confounding the relationship between fitness and PIU.\nFor adolescents, the correlations of the target variable with all measures of fitness are weak or null, and for adults who pass the fitness test, only 1 has data on PIU severity.\nIn overall, fitness measures do not show noticable correlations with PIU severity, and it appears that age may be driving both increased fitness performance and higher PIU severity","metadata":{}},{"cell_type":"markdown","source":"**Sleep Disturbance Scale**","metadata":{}},{"cell_type":"code","source":"groups.get('Sleep Disturbance Scale', [])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:47.423141Z","iopub.execute_input":"2025-01-23T08:27:47.423385Z","iopub.status.idle":"2025-01-23T08:27:47.427940Z","shell.execute_reply.started":"2025-01-23T08:27:47.423363Z","shell.execute_reply":"2025-01-23T08:27:47.427287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = temp[temp['SDS-SDS_Total_Raw'].notnull()]\nage_range = data['Basic_Demos-Age']\nprint(\n    f\"Age range for participants with SDS-SDS_Total_Raw data:\"\n    f\" {age_range.min()} - {age_range.max()} years\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:47.429071Z","iopub.execute_input":"2025-01-23T08:27:47.429363Z","iopub.status.idle":"2025-01-23T08:27:47.468488Z","shell.execute_reply.started":"2025-01-23T08:27:47.429333Z","shell.execute_reply":"2025-01-23T08:27:47.467631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(18, 5))\n\n# SDS-Season (Pie Chart)\nplt.subplot(1, 3, 1)\nsds_season_counts = temp['SDS-Season'].value_counts(normalize=True)\nplt.pie(\n    sds_season_counts, \n    labels=sds_season_counts.index, \n    autopct='%1.1f%%', \n    startangle=90, \n    colors=sns.color_palette(\"Set3\")\n)\nplt.title('SDS-Season')\n\n# SDS-SDS_Total_Raw\nplt.subplot(1, 3, 2)\nsns.histplot(temp['SDS-SDS_Total_Raw'].dropna(), bins=20, kde=True)\nplt.title('SDS-SDS_Total_Raw')\nplt.xlabel('Value')\n\n# SDS-SDS_Total_T\nplt.subplot(1, 3, 3)\nsns.histplot(temp['SDS-SDS_Total_T'].dropna(), bins=20, kde=True)\nplt.title('SDS-SDS_Total_T')\nplt.xlabel('Value')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:47.469359Z","iopub.execute_input":"2025-01-23T08:27:47.469635Z","iopub.status.idle":"2025-01-23T08:27:47.990718Z","shell.execute_reply.started":"2025-01-23T08:27:47.469615Z","shell.execute_reply":"2025-01-23T08:27:47.989754Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Columns of interest\nsleep_columns = ['SDS-SDS_Total_Raw', 'SDS-SDS_Total_T']\n\n# Create subplots for boxplots against 'sii_severity_level'\nplt.figure(figsize=(12, 5))\nfor i, col in enumerate(sleep_columns, 1):\n    plt.subplot(1, 2, i)\n    sns.boxplot(\n        data=filtered_train, \n        x='sii', \n        y=col, \n        palette='viridis'\n    )\n    plt.title(f'{col} by SII Severity Level', fontsize=14)\n    plt.xlabel('SII Severity Level', fontsize=12)\n    plt.ylabel(col, fontsize=12)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:47.991613Z","iopub.execute_input":"2025-01-23T08:27:47.991830Z","iopub.status.idle":"2025-01-23T08:27:48.357585Z","shell.execute_reply.started":"2025-01-23T08:27:47.991812Z","shell.execute_reply":"2025-01-23T08:27:48.356569Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Both the raw and T-scores for sleep disturbance are moderately variable, with some extreme values indicating severe sleep disturbances in a subset of participants.\nFurther Analysis (coming soon): to explore whether specific demographic factors (e.g., age, gender, season) are associated with higher sleep disturbance scores.","metadata":{}},{"cell_type":"markdown","source":"**Physical Activity Questionnaire**","metadata":{}},{"cell_type":"code","source":"groups.get('Physical Activity Questionnaire (Adolescents)', [])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:48.358566Z","iopub.execute_input":"2025-01-23T08:27:48.358933Z","iopub.status.idle":"2025-01-23T08:27:48.364372Z","shell.execute_reply.started":"2025-01-23T08:27:48.358907Z","shell.execute_reply":"2025-01-23T08:27:48.363536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = temp[temp['PAQ_A-PAQ_A_Total'].notnull()]\nage_range = data['Basic_Demos-Age']\nprint(\n    f\"Age range for Adolescents (with PAQ_A_Total data):\"\n    f\" {age_range.min()} - {age_range.max()} years\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:48.365242Z","iopub.execute_input":"2025-01-23T08:27:48.365508Z","iopub.status.idle":"2025-01-23T08:27:48.382441Z","shell.execute_reply.started":"2025-01-23T08:27:48.365486Z","shell.execute_reply":"2025-01-23T08:27:48.381636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(18, 5))\n\n# PAQ_A-Season\nplt.subplot(1, 3, 1)\ntemp['PAQ_A-Season'].value_counts(normalize=True).plot.pie(\n    autopct='%1.1f%%', colors=plt.cm.Set3.colors\n)\nplt.title('PAQ_A-Season (Adolescents)')\n\n# PAQ_A-PAQ_A_Total\nplt.subplot(1, 3, 2)\nsns.histplot(temp['PAQ_A-PAQ_A_Total'], bins=20, kde=True)\nplt.title('PAQ_A-PAQ_A_Total (Adolescents)')\n\n# PAQ_A_Total by Season\nplt.subplot(1, 3, 3)\nsns.violinplot(x='PAQ_A-Season', y='PAQ_A-PAQ_A_Total', data=temp, palette=\"Set3\")\nplt.title('PAQ_A_Total by Season (Adolescents)')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:48.383178Z","iopub.execute_input":"2025-01-23T08:27:48.383390Z","iopub.status.idle":"2025-01-23T08:27:49.025148Z","shell.execute_reply.started":"2025-01-23T08:27:48.383372Z","shell.execute_reply":"2025-01-23T08:27:49.024261Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Children**","metadata":{}},{"cell_type":"code","source":"groups.get('Physical Activity Questionnaire (Children)', [])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:49.026046Z","iopub.execute_input":"2025-01-23T08:27:49.026349Z","iopub.status.idle":"2025-01-23T08:27:49.031254Z","shell.execute_reply.started":"2025-01-23T08:27:49.026320Z","shell.execute_reply":"2025-01-23T08:27:49.030455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = temp[temp['PAQ_C-PAQ_C_Total'].notnull()]\nage_range = data['Basic_Demos-Age']\nprint(\n    f\"Age range for Children (with PAQ_C_Total data):\"\n    f\" {age_range.min()} - {age_range.max()} years\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:49.031976Z","iopub.execute_input":"2025-01-23T08:27:49.032211Z","iopub.status.idle":"2025-01-23T08:27:49.048553Z","shell.execute_reply.started":"2025-01-23T08:27:49.032181Z","shell.execute_reply":"2025-01-23T08:27:49.047845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(18, 5))\n\n# PAQ_C-Season\nplt.subplot(1, 3, 1)\ntemp['PAQ_C-Season'].value_counts(normalize=True).plot.pie(\n    autopct='%1.1f%%', colors=plt.cm.Set3.colors\n)\nplt.title('PAQ_C-Season (Children)')\n\n# PAQ_C-PAQ_C_Total\nplt.subplot(1, 3, 2)\nsns.histplot(temp['PAQ_C-PAQ_C_Total'], bins=20, kde=True)\nplt.title('PAQ_C-PAQ_C_Total (Children)')\n\n# PAQ_C_Total by Season\nplt.subplot(1, 3, 3)\nsns.violinplot(x='PAQ_C-Season', y='PAQ_C-PAQ_C_Total', data=temp, palette=\"Set3\")\nplt.title('PAQ_C_Total by Season (Children)')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:49.049212Z","iopub.execute_input":"2025-01-23T08:27:49.049391Z","iopub.status.idle":"2025-01-23T08:27:49.596211Z","shell.execute_reply.started":"2025-01-23T08:27:49.049376Z","shell.execute_reply":"2025-01-23T08:27:49.595338Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The division into adolescents and children seems to be incorrect (participants with data in the children columns (PAQ_C_Total) are 7 - 17 years old - overlapping with those with non-missing data in the adolescents columns - 13 - 18 years old).\nPhysical activity levels are fairly stable over the seasons, with only minor variations, although are slightly lower in the fall and winter for adolescents and children, respectively.\nThere are many missing values for these features","metadata":{}},{"cell_type":"code","source":"corr_matrix = supervised_usable[\n    [\n        'PCIAT-PCIAT_Total', 'Basic_Demos-Age', 'Basic_Demos-Sex', 'Physical-BMI', \n        'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n        'Physical-Diastolic_BP', 'Physical-Systolic_BP', 'Physical-HeartRate',\n        'PreInt_EduHx-computerinternet_hoursday', 'SDS-SDS_Total_T', 'PAQ_A-PAQ_A_Total',\n        'PAQ_C-PAQ_C_Total', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', \n        'Fitness_Endurance-Time_Sec', 'FGC-FGC_CU', 'FGC-FGC_GSND', 'FGC-FGC_GSD', \n        'FGC-FGC_PU', 'FGC-FGC_SRL', 'FGC-FGC_SRR', 'FGC-FGC_TL', 'BIA-BIA_Activity_Level_num', \n        'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', \n        'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num', \n        'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW'\n    ]\n].corr()\nsii_corr = corr_matrix['PCIAT-PCIAT_Total'].drop('PCIAT-PCIAT_Total')\nfiltered_corr  = sii_corr[(sii_corr > 0.1) | (sii_corr < -0.1)]\n\nplt.figure(figsize=(8, 6))\nfiltered_corr.sort_values().plot(kind='barh', color='coral')\nplt.title('Features with Correlation > 0.1 or < -0.1 with PCIAT-PCIAT_Total')\nplt.xlabel('Correlation coefficient')\nplt.ylabel('Features')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:49.597315Z","iopub.execute_input":"2025-01-23T08:27:49.597643Z","iopub.status.idle":"2025-01-23T08:27:49.852841Z","shell.execute_reply.started":"2025-01-23T08:27:49.597611Z","shell.execute_reply":"2025-01-23T08:27:49.851926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"actigraphy = pd.read_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id=0417c91e/part-0.parquet')\nactigraphy.sample(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:49.853735Z","iopub.execute_input":"2025-01-23T08:27:49.854038Z","iopub.status.idle":"2025-01-23T08:27:50.046419Z","shell.execute_reply.started":"2025-01-23T08:27:49.853988Z","shell.execute_reply":"2025-01-23T08:27:50.045509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample = train[train['id'] == '0417c91e']\nsample","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:50.047349Z","iopub.execute_input":"2025-01-23T08:27:50.047584Z","iopub.status.idle":"2025-01-23T08:27:50.092297Z","shell.execute_reply.started":"2025-01-23T08:27:50.047565Z","shell.execute_reply":"2025-01-23T08:27:50.091687Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"series_train = actigraphy\n\ndef plot_series_data(df, col='non-wear_flag',\n                     label='Worn (0 = Worn, 1 = Not Worn)',\n                     title='Non-Wear Flag',\n                     x_col='day_time', x_label='Day Relative to PCIAT + Time'):\n    plt.figure(figsize=(18, 12))\n    \n    # ENMO\n    plt.subplot(4, 1, 1)\n    plt.scatter(df[x_col], df['enmo'], label='ENMO', color='green', s=1)\n    plt.title('ENMO (Euclidean Norm Minus One)')\n    plt.ylabel('Movement Intensity')\n\n    # Angle Z\n    plt.subplot(4, 1, 2)\n    plt.scatter(df[x_col], df['anglez'], label='Angle Z', color='blue', s=1)\n    plt.title('Angle Z')\n    plt.ylabel('Angle (degrees)')\n\n    # Light\n    plt.subplot(4, 1, 3)\n    plt.scatter(df[x_col], df['light'], label='Light', color='orange', s=1)\n    plt.title('Ambient Light')\n    plt.ylabel('Light (lux)')\n\n    # Any other column\n    plt.subplot(4, 1, 4)\n    plt.scatter(df[x_col], df[col], label=col, color='red', s=1)\n    plt.title(f'{title}')\n    plt.ylabel(f'{label}')\n    plt.xlabel(f'{x_label}')\n\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:50.093036Z","iopub.execute_input":"2025-01-23T08:27:50.093277Z","iopub.status.idle":"2025-01-23T08:27:50.099519Z","shell.execute_reply.started":"2025-01-23T08:27:50.093248Z","shell.execute_reply":"2025-01-23T08:27:50.098735Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"So we'll look at the data collected from a 6-year-old girl, with no data on sleep or activity based on Physical Activity Questionnaire, who spends no time on the Internet and has an SII of 0.\n\nHer physical data includes a BMI of 15.48, height of 45.5 inches, and weight of 45.6 pounds, normal blood pressure readings (Systolic: 115, Diastolic: 73), and a heart rate of 86 bpm\n\nThe fitness endurance test shows a max stage of 6, with a duration of 9 minutes and 5 seconds.\n\nThe girl completed a small number of curl-ups, no push-ups, and had moderate sit-and-reach flexibility on both sides. There is missing data for grip strength tests, and the participant's trunk lift performance is in zone 1.0\n\nA CGAS score of 40 suggests that the child has significant impairment in daily functioning","metadata":{}},{"cell_type":"code","source":"plot_series_data(series_train, x_col='step', x_label='Step')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:50.100281Z","iopub.execute_input":"2025-01-23T08:27:50.100472Z","iopub.status.idle":"2025-01-23T08:27:51.357757Z","shell.execute_reply.started":"2025-01-23T08:27:50.100456Z","shell.execute_reply":"2025-01-23T08:27:51.356854Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"the wear flag looks correct (always indicating that the device was worn). But when using the step column as the x-axis, the time points are assumed to be equidistant, meaning that each step represents a uniform interval (e.g., every 5 seconds). But in reality, there may be irregular time gaps between data points.","metadata":{}},{"cell_type":"markdown","source":"Create a continuous time scale in days by transforming the time_of_day column (which is in nanoseconds) to hours and then combining it with relative_date_PCIAT:","metadata":{}},{"cell_type":"code","source":"series_train['time_of_day_hours'] = (\n    series_train['time_of_day'] / 1e9 / 3600 #nanoseconds to hours\n)\nseries_train['day_time'] = series_train['relative_date_PCIAT'] + (\n    series_train['time_of_day_hours'] / 24\n)\nplot_series_data(series_train, x_col='day_time', x_label='Day Relative to PCIAT + Time')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:51.358587Z","iopub.execute_input":"2025-01-23T08:27:51.358807Z","iopub.status.idle":"2025-01-23T08:27:52.598526Z","shell.execute_reply.started":"2025-01-23T08:27:51.358788Z","shell.execute_reply":"2025-01-23T08:27:52.597627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(series_train[series_train['relative_date_PCIAT'] > 36])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:52.599350Z","iopub.execute_input":"2025-01-23T08:27:52.599632Z","iopub.status.idle":"2025-01-23T08:27:52.606161Z","shell.execute_reply.started":"2025-01-23T08:27:52.599607Z","shell.execute_reply":"2025-01-23T08:27:52.605464Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Note: Here, the data is plotted with real time on the x-axis, revealing gaps (especially on the right end) - the data for a few last weeks (after day 36) is only 1,830 measurements out of 287,179. These gaps suggest that some data was either not collected at all or was intentionally cut. Let's zoom in.","metadata":{}},{"cell_type":"markdown","source":"Plot data for specific day(s):","metadata":{}},{"cell_type":"code","source":"start_day = 2 # second day of wearing the device\nshow_days = 1\n\nfirst_day = min(series_train['relative_date_PCIAT']) + start_day - 1\nfiltered_data = series_train[\n    (series_train['relative_date_PCIAT'] >= first_day) &\n    (series_train['relative_date_PCIAT'] <= first_day + show_days - 1)\n].copy()\n\nplot_series_data(filtered_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:52.607113Z","iopub.execute_input":"2025-01-23T08:27:52.607413Z","iopub.status.idle":"2025-01-23T08:27:53.370865Z","shell.execute_reply.started":"2025-01-23T08:27:52.607386Z","shell.execute_reply":"2025-01-23T08:27:53.369980Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The left part of the plot likely represents sleep time.\nThe activity level peaks in the middle of the day.\nThe spikes in the ENMO, Angle Z, and Ambient Light suggest short bursts of activity that may correspond to specific tasks or movements, but there are many periods of low or no movement, and the overall pattern of activity appears to be consistent with significant impairment in daily functioning according to the CGAS score.\nFor a healty child of this age, you would generally expect to see more consistent periods of active movement throughout the day, especially during the daytime hours when children typically play, run, etc.","metadata":{}},{"cell_type":"markdown","source":"Plot data for specific hour(s):","metadata":{}},{"cell_type":"code","source":"show_day = 2 # show second day\nhour_from = 13 # starting from 1 pm\nhour_to = 14 # to 2 pm\n\nshow_day = min(series_train['relative_date_PCIAT']) + show_day - 1\nfiltered_data = series_train[\n    (series_train['relative_date_PCIAT'] == show_day) &\n    (series_train['time_of_day_hours'] >= hour_from) & \n    (series_train['time_of_day_hours'] < hour_to)\n].copy()\n\nplot_series_data(filtered_data, x_col='time_of_day_hours', x_label='Time, hours')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:53.371661Z","iopub.execute_input":"2025-01-23T08:27:53.371901Z","iopub.status.idle":"2025-01-23T08:27:54.135436Z","shell.execute_reply.started":"2025-01-23T08:27:53.371881Z","shell.execute_reply":"2025-01-23T08:27:54.134608Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The ENMO (Euclidean Norm Minus One) shows small but continuous movement, and there are changes in orientation, indicating that even minor movements are detected, which can be interpreted as the device being worn.\nThe same is true if we look at any day within the last 2 weeks, where there are few measurements left, yet the worn flag seems to align well with movement patterns.\nThe light data in this part of the plot does not look completely consistent or reliable, especially because of the flatline, linear trend in the middle and sudden changes (data processing artifacts?).","metadata":{}},{"cell_type":"markdown","source":"Maybe the data has been truncated for periods of low battery? Let's check the battery voltage over time.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(18, 5))\n\nplt.scatter(series_train['day_time'], series_train['battery_voltage'], \n            color='purple', label='Battery Voltage (mV)', s=1)\n\nplt.xlabel('Day Relative to PCIAT + Time')\nplt.ylabel('Battery Voltage (mV)')\nplt.title('Battery Voltage and Irregular Time Intervals')\nplt.legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:54.136563Z","iopub.execute_input":"2025-01-23T08:27:54.136905Z","iopub.status.idle":"2025-01-23T08:27:55.965388Z","shell.execute_reply.started":"2025-01-23T08:27:54.136873Z","shell.execute_reply":"2025-01-23T08:27:55.964515Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The battery did not seem to run out completely at the end. I guess, the smooth decline suggests that the device was likely functioning properly during this time.","metadata":{}},{"cell_type":"markdown","source":"Note: The purple areas (representing the day period) toward the end of the timeline appear stretched and irregular, but we already know that a number of records have been cut from the data.","metadata":{}},{"cell_type":"markdown","source":"**Main Statistics for all Participants**","metadata":{}},{"cell_type":"markdown","source":"Here I calculate the percentage of rows with wear flag = 1 (device not worn) and the main statistics for key parameters for periods when the device was worn, ignoring time gaps, to get an idea of the values distributions.","metadata":{}},{"cell_type":"code","source":"DIR = '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet'\n\ndef process_file(file_path, participant_id):\n    data = pd.read_parquet(file_path)\n    non_wear_percentage = (data['non-wear_flag'].sum() / len(data)) * 100\n    worn_data = data[data['non-wear_flag'] == 0]\n\n    return {\n        'id': participant_id,\n        'non_wear_percentage': non_wear_percentage,\n        'enmo_stats': worn_data['enmo'].describe(),\n        'anglez_stats': worn_data['anglez'].describe(),\n        'light_stats': worn_data['light'].describe(),\n        'battery_voltage_stats': worn_data['battery_voltage'].describe(),\n        'unique_days': worn_data['relative_date_PCIAT'].nunique()\n    }\n\nresults = []\nfor participant_id in os.listdir(DIR):\n    file_path = os.path.join(DIR, participant_id, 'part-0.parquet')\n    result = process_file(file_path, participant_id)\n    results.append(result)\n\nfinal_results = []\nfor result in results:\n    flat_row = {\n        'id': result['id'].replace('id=', ''),\n        'non_wear_percentage': result['non_wear_percentage'],\n        'unique_days': result['unique_days']\n    }\n    \n    for key, stats in result.items():\n        if isinstance(stats, pd.Series):\n            for stat_name, stat_value in stats.items():\n                flat_row[f'{key.replace(\"_stats\", \"\")}_{stat_name}'] = stat_value\n    \n    final_results.append(flat_row)\n    \nstats_df = pd.DataFrame(final_results)\n\n# non_wear_percentage\nplt.figure(figsize=(12, 5))\nplt.subplot(1, 2, 1)\nplt.hist(stats_df['non_wear_percentage'], bins=20, edgecolor='black')\nplt.title('Non-Wear Percentage')\nplt.xlabel('Non-Wear Percentage')\nplt.ylabel('Frequency')\n\n# unique_days\nplt.subplot(1, 2, 2)\nplt.hist(stats_df['unique_days'], bins=20, edgecolor='black')\nplt.title('Number of Days')\nplt.xlabel('Unique Days')\nplt.ylabel('Frequency')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:27:55.966347Z","iopub.execute_input":"2025-01-23T08:27:55.966698Z","iopub.status.idle":"2025-01-23T08:29:43.972835Z","shell.execute_reply.started":"2025-01-23T08:27:55.966665Z","shell.execute_reply":"2025-01-23T08:29:43.971947Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"In total we have actigraphy data for 996 participants\nThe description says that \"During their participation in the HBN study, some participants were given an accelerometer to wear for up to 30 days continually while at home and going about their regular daily lives\", but we can see that participants actually wore the device from 1 to 81 days. Half of them wore it for 24 days or less.","metadata":{}},{"cell_type":"markdown","source":"Let's make simple plots to get a general overview of the variability in min, max, median, and std across the parameters (enmo, anglez, light, and battery_voltage)","metadata":{}},{"cell_type":"code","source":"def plot_parameter_statistics(stats_df, parameter):\n    stats_to_plot = ['min', 'max', '50%', 'std']\n    stat_labels = ['Min', 'Max', 'Median', 'Std']\n\n    plt.figure(figsize=(14, 5))\n\n    for j, stat in enumerate(stats_to_plot):\n        plt.subplot(1, 4, j + 1)\n        \n        data = stats_df[f'{parameter}_{stat}']\n        plt.hist(data, bins=20, alpha=0.7, edgecolor='black')\n        \n        plt.title(f'{parameter.capitalize()} - {stat_labels[j]}')\n        plt.xlabel('Values')\n        plt.ylabel('Frequency')\n        plt.grid(True)\n\n    plt.tight_layout()\n    plt.show()\nplot_parameter_statistics(stats_df, 'enmo')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:29:43.973739Z","iopub.execute_input":"2025-01-23T08:29:43.974030Z","iopub.status.idle":"2025-01-23T08:29:44.686679Z","shell.execute_reply.started":"2025-01-23T08:29:43.973985Z","shell.execute_reply":"2025-01-23T08:29:44.685848Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There are participants without no movement periods because the minimum value can be non-zero.\nMost max values are between 2 and 6, but some go up to 10 or more, and apparently there are participants not moving at all? (as max can be around 0)","metadata":{}},{"cell_type":"code","source":"plot_parameter_statistics(stats_df, 'anglez')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:29:44.687569Z","iopub.execute_input":"2025-01-23T08:29:44.687791Z","iopub.status.idle":"2025-01-23T08:29:45.374703Z","shell.execute_reply.started":"2025-01-23T08:29:44.687765Z","shell.execute_reply":"2025-01-23T08:29:45.373796Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Angle Z in actigraphy generally measures the vertical orientation or tilt of the wrist along the z-axis, which is typically aligned with the forearm’s longitudinal axis. Given that the device is fixed to the wrist, this angle provides information on how the wrist is rotated vertically (up or down).\nI think, these results make sense considering normal wrist movements throughout the day.","metadata":{}},{"cell_type":"code","source":"plot_parameter_statistics(stats_df, 'light')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:29:45.375588Z","iopub.execute_input":"2025-01-23T08:29:45.375884Z","iopub.status.idle":"2025-01-23T08:29:46.104658Z","shell.execute_reply.started":"2025-01-23T08:29:45.375863Z","shell.execute_reply":"2025-01-23T08:29:46.103644Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"All minimum values are at or very close to 0, which indicates that during each period of observation, the device experienced moments with no or minimal light exposure\nMost of the max values are concentrated between 0-2500 lux, and as with ENMO, it seems that there are participants who live in total darkness? also a few have very high values (up to around 20,000), I guess because they have a device with a higher lux range.\nDuring most recording periods, the median light exposure is low (e.g., indoors, shaded areas), but also is highly variable within samples.","metadata":{}},{"cell_type":"markdown","source":"**Activity and Light Exposure**","metadata":{}},{"cell_type":"markdown","source":"Investigate the correlation between ambient light levels and activity levels by looking at how ENMO varies with light exposure.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 5))\nplt.scatter(worn_data['light'], worn_data['enmo'], alpha=0.5)\nplt.xlabel('Light (Lux)')\nplt.ylabel('ENMO (Activity)')\nplt.title('ENMO vs Light Exposure')\nplt.show()\n\ncorrelation_light_enmo = worn_data[['light', 'enmo']].corr().iloc[0, 1]\nprint(f\"Correlation between Light and ENMO: {correlation_light_enmo}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:29:51.000712Z","iopub.execute_input":"2025-01-23T08:29:51.001049Z","iopub.status.idle":"2025-01-23T08:29:51.831468Z","shell.execute_reply.started":"2025-01-23T08:29:51.000994Z","shell.execute_reply":"2025-01-23T08:29:51.830783Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There seems to be a slight positive correlation od 0.2 between ENMO and Light Exposure, meaning that higher light exposure tends to be associated with higher activity, but no clear trend\nMost of the recorded activity happened in relatively low-light environments.","metadata":{}},{"cell_type":"code","source":"!pip -q install /kaggle/input/pytorchtabnet/pytorch_tabnet-4.1.0-py3-none-any.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:29:51.832254Z","iopub.execute_input":"2025-01-23T08:29:51.832575Z","iopub.status.idle":"2025-01-23T08:29:56.050523Z","shell.execute_reply.started":"2025-01-23T08:29:51.832546Z","shell.execute_reply":"2025-01-23T08:29:56.049412Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pytorch_tabnet.tab_model import TabNetRegressor\nimport torch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:29:56.051466Z","iopub.execute_input":"2025-01-23T08:29:56.051749Z","iopub.status.idle":"2025-01-23T08:29:59.064986Z","shell.execute_reply.started":"2025-01-23T08:29:56.051727Z","shell.execute_reply":"2025-01-23T08:29:59.064294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[season_columns_train] = train[season_columns_train].astype(season_dtype)\ntest[season_columns_test] = test[season_columns_test].astype(season_dtype)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:29:59.065800Z","iopub.execute_input":"2025-01-23T08:29:59.066064Z","iopub.status.idle":"2025-01-23T08:29:59.089142Z","shell.execute_reply.started":"2025-01-23T08:29:59.066042Z","shell.execute_reply":"2025-01-23T08:29:59.088190Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    stats, indexes = zip(*results)\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nclass AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.ReLU()\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.ReLU(),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.ReLU(),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n        \n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\n        \ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    \n    data_tensor = torch.FloatTensor(df_scaled)\n    \n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    \n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n    \n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n            \n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n                 \n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n        \n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    \n    return df_encoded\n\ndef feature_engineering(df):\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1) \n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:29:59.090003Z","iopub.execute_input":"2025-01-23T08:29:59.090279Z","iopub.status.idle":"2025-01-23T08:29:59.108680Z","shell.execute_reply.started":"2025-01-23T08:29:59.090259Z","shell.execute_reply":"2025-01-23T08:29:59.107933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train  = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest   = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts  = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:29:59.109529Z","iopub.execute_input":"2025-01-23T08:29:59.109772Z","iopub.status.idle":"2025-01-23T08:31:04.943962Z","shell.execute_reply.started":"2025-01-23T08:29:59.109740Z","shell.execute_reply":"2025-01-23T08:31:04.943048Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = train_ts.drop('id', axis=1)\ndf_test  = test_ts .drop('id', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:31:04.944999Z","iopub.execute_input":"2025-01-23T08:31:04.945357Z","iopub.status.idle":"2025-01-23T08:31:04.950585Z","shell.execute_reply.started":"2025-01-23T08:31:04.945324Z","shell.execute_reply":"2025-01-23T08:31:04.949763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts_encoded = perform_autoencoder(df_train, encoding_dim=60, epochs=100, batch_size=32)\ntest_ts_encoded  = perform_autoencoder(df_test,  encoding_dim=60, epochs=100, batch_size=32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:31:04.951452Z","iopub.execute_input":"2025-01-23T08:31:04.951716Z","iopub.status.idle":"2025-01-23T08:31:16.388278Z","shell.execute_reply.started":"2025-01-23T08:31:04.951694Z","shell.execute_reply":"2025-01-23T08:31:16.387545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"time_series_cols = train_ts_encoded.columns.tolist()\ntrain_ts_encoded['id'] = train_ts['id']\ntest_ts_encoded['id']=test_ts[\"id\"]\n\ntrain = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest  = pd.merge(test, train_ts_encoded, how=\"left\", on='id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:31:16.389084Z","iopub.execute_input":"2025-01-23T08:31:16.390083Z","iopub.status.idle":"2025-01-23T08:31:16.404187Z","shell.execute_reply.started":"2025-01-23T08:31:16.390058Z","shell.execute_reply":"2025-01-23T08:31:16.403507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f'{train.shape=}')\nprint(f'{test.shape=}')\ntrain.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:31:16.405140Z","iopub.execute_input":"2025-01-23T08:31:16.405436Z","iopub.status.idle":"2025-01-23T08:31:16.496944Z","shell.execute_reply.started":"2025-01-23T08:31:16.405413Z","shell.execute_reply":"2025-01-23T08:31:16.496187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import KNNImputer\nimputer = KNNImputer(n_neighbors=5)\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n        \ntrain = train_imputed\ntrain = feature_engineering(train)\ntrain = train.dropna(thresh=10, axis=0)\ntest = feature_engineering(test)\n\ntrain = train.drop('id', axis=1)\ntest  = test .drop('id', axis=1)   \n\n\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW']\n\nfeaturesCols += time_series_cols\ntest = test[featuresCols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:31:16.497861Z","iopub.execute_input":"2025-01-23T08:31:16.498171Z","iopub.status.idle":"2025-01-23T08:31:23.547253Z","shell.execute_reply.started":"2025-01-23T08:31:16.498142Z","shell.execute_reply":"2025-01-23T08:31:23.546525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if np.any(np.isinf(train)):\n    train = train.replace([np.inf, -np.inf], np.nan)\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:31:23.547981Z","iopub.execute_input":"2025-01-23T08:31:23.548235Z","iopub.status.idle":"2025-01-23T08:31:23.559214Z","shell.execute_reply.started":"2025-01-23T08:31:23.548214Z","shell.execute_reply":"2025-01-23T08:31:23.558454Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 42\nn_splits = 5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:31:23.560241Z","iopub.execute_input":"2025-01-23T08:31:23.560562Z","iopub.status.idle":"2025-01-23T08:31:23.569496Z","shell.execute_reply.started":"2025-01-23T08:31:23.560530Z","shell.execute_reply":"2025-01-23T08:31:23.568754Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:31:23.570033Z","iopub.execute_input":"2025-01-23T08:31:23.570243Z","iopub.status.idle":"2025-01-23T08:31:23.580740Z","shell.execute_reply.started":"2025-01-23T08:31:23.570220Z","shell.execute_reply":"2025-01-23T08:31:23.579964Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Model parameters for LightGBM\nLightGBM_Params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01,  # Increased from 2.68e-06\n    'device': 'gpu'\n\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED,\n    'tree_method': 'gpu_hist',\n\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    # 'cat_features': cat_c,\n    'verbose': 0,\n    'l2_leaf_reg': 10,  # Increase this value\n    'task_type': 'GPU'\n\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:31:23.581468Z","iopub.execute_input":"2025-01-23T08:31:23.581755Z","iopub.status.idle":"2025-01-23T08:31:23.599520Z","shell.execute_reply.started":"2025-01-23T08:31:23.581724Z","shell.execute_reply":"2025-01-23T08:31:23.598793Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# New: TabNet\n\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split\nfrom pytorch_tabnet.callbacks import Callback\nimport os\nimport torch\nfrom pytorch_tabnet.callbacks import Callback\n\nclass TabNetWrapper(BaseEstimator, RegressorMixin):\n    def __init__(self, **kwargs):\n        self.model = TabNetRegressor(**kwargs)\n        self.kwargs = kwargs\n        self.imputer = SimpleImputer(strategy='median')\n        self.best_model_path = 'best_tabnet_model.pt'\n        \n    def fit(self, X, y):\n        # Handle missing values\n        X_imputed = self.imputer.fit_transform(X)\n        \n        if hasattr(y, 'values'):\n            y = y.values\n            \n        # Create internal validation set\n        X_train, X_valid, y_train, y_valid = train_test_split(\n            X_imputed, \n            y, \n            test_size=0.2,\n            random_state=42\n        )\n        \n        # Train TabNet model\n        history = self.model.fit(\n            X_train=X_train,\n            y_train=y_train.reshape(-1, 1),\n            eval_set=[(X_valid, y_valid.reshape(-1, 1))],\n            eval_name=['valid'],\n            eval_metric=['mse'],\n            max_epochs=500,\n            patience=50,\n            batch_size=1024,\n            virtual_batch_size=128,\n            num_workers=0,\n            drop_last=False,\n            callbacks=[\n                TabNetPretrainedModelCheckpoint(\n                    filepath=self.best_model_path,\n                    monitor='valid_mse',\n                    mode='min',\n                    save_best_only=True,\n                    verbose=True\n                )\n            ]\n        )\n        \n        # Load the best model\n        if os.path.exists(self.best_model_path):\n            self.model.load_model(self.best_model_path)\n            os.remove(self.best_model_path)  # Remove temporary file\n        \n        return self\n    \n    def predict(self, X):\n        X_imputed = self.imputer.transform(X)\n        return self.model.predict(X_imputed).flatten()\n    \n    def __deepcopy__(self, memo):\n        # Add deepcopy support for scikit-learn\n        cls = self.__class__\n        result = cls.__new__(cls)\n        memo[id(self)] = result\n        for k, v in self.__dict__.items():\n            setattr(result, k, deepcopy(v, memo))\n        return result\n\n# TabNet hyperparameters\nTabNet_Params = {\n    'n_d': 64,              # Width of the decision prediction layer\n    'n_a': 64,              # Width of the attention embedding for each step\n    'n_steps': 5,           # Number of steps in the architecture\n    'gamma': 1.5,           # Coefficient for feature selection regularization\n    'n_independent': 2,     # Number of independent GLU layer in each GLU block\n    'n_shared': 2,          # Number of shared GLU layer in each GLU block\n    'lambda_sparse': 1e-4,  # Sparsity regularization\n    'optimizer_fn': torch.optim.Adam,\n    'optimizer_params': dict(lr=2e-2, weight_decay=1e-5),\n    'mask_type': 'entmax',\n    'scheduler_params': dict(mode=\"min\", patience=10, min_lr=1e-5, factor=0.5),\n    'scheduler_fn': torch.optim.lr_scheduler.ReduceLROnPlateau,\n    'verbose': 1,\n    'device_name': 'cuda' if torch.cuda.is_available() else 'cpu'\n}\n\nclass TabNetPretrainedModelCheckpoint(Callback):\n    def __init__(self, filepath, monitor='val_loss', mode='min', \n                 save_best_only=True, verbose=1):\n        super().__init__()  # Initialize parent class\n        self.filepath = filepath\n        self.monitor = monitor\n        self.mode = mode\n        self.save_best_only = save_best_only\n        self.verbose = verbose\n        self.best = float('inf') if mode == 'min' else -float('inf')\n        \n    def on_train_begin(self, logs=None):\n        self.model = self.trainer  # Use trainer itself as model\n        \n    def on_epoch_end(self, epoch, logs=None):\n        logs = logs or {}\n        current = logs.get(self.monitor)\n        if current is None:\n            return\n        \n        # Check if current metric is better than best\n        if (self.mode == 'min' and current < self.best) or \\\n           (self.mode == 'max' and current > self.best):\n            if self.verbose:\n                print(f'\\nEpoch {epoch}: {self.monitor} improved from {self.best:.4f} to {current:.4f}')\n            self.best = current\n            if self.save_best_only:\n                self.model.save_model(self.filepath)  # Save the entire model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:31:23.600421Z","iopub.execute_input":"2025-01-23T08:31:23.600711Z","iopub.status.idle":"2025-01-23T08:31:23.614610Z","shell.execute_reply.started":"2025-01-23T08:31:23.600682Z","shell.execute_reply":"2025-01-23T08:31:23.613951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create model instances\nLight = LGBMRegressor(**LightGBM_Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\nTabNet_Model = TabNetWrapper(**TabNet_Params) # New","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:31:23.615404Z","iopub.execute_input":"2025-01-23T08:31:23.615701Z","iopub.status.idle":"2025-01-23T08:31:23.637727Z","shell.execute_reply.started":"2025-01-23T08:31:23.615674Z","shell.execute_reply":"2025-01-23T08:31:23.637081Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    ('tabnet', TabNet_Model)\n])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:31:23.646321Z","iopub.execute_input":"2025-01-23T08:31:23.646524Z","iopub.status.idle":"2025-01-23T08:31:23.650098Z","shell.execute_reply.started":"2025-01-23T08:31:23.646507Z","shell.execute_reply":"2025-01-23T08:31:23.649278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f'{train.shape=}')\nnum_nan_sii = train['sii'].isna().sum()\nnum_nan_sii","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:31:23.650993Z","iopub.execute_input":"2025-01-23T08:31:23.651254Z","iopub.status.idle":"2025-01-23T08:31:23.664954Z","shell.execute_reply.started":"2025-01-23T08:31:23.651235Z","shell.execute_reply":"2025-01-23T08:31:23.664270Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission1 = TrainML(voting_model, test)\n\nSubmission1.to_csv('submission.csv', index=False)\n\nSubmission1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:31:23.665715Z","iopub.execute_input":"2025-01-23T08:31:23.665903Z","iopub.status.idle":"2025-01-23T08:34:20.888826Z","shell.execute_reply.started":"2025-01-23T08:31:23.665887Z","shell.execute_reply":"2025-01-23T08:34:20.888124Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n        \ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\nprint(f'{train.shape=}')\nprint(f'{test.shape=}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:34:20.889563Z","iopub.execute_input":"2025-01-23T08:34:20.889784Z","iopub.status.idle":"2025-01-23T08:35:27.201558Z","shell.execute_reply.started":"2025-01-23T08:34:20.889765Z","shell.execute_reply":"2025-01-23T08:35:27.200501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission\n\n# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01  # Increased from 2.68e-06\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'cat_features': cat_c,\n    'verbose': 0,\n    'l2_leaf_reg': 10  # Increase this value\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\nTabNet_Model = TabNetWrapper(**TabNet_Params)  # New:TAbNet\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    ('tabnet', TabNet_Model)  # New:TabNet\n])\n\n# Train the ensemble model\nSubmission2 = TrainML(voting_model, test)\n\n# Save submission\n#Submission2.to_csv('submission.csv', index=False)\nSubmission2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:35:27.202618Z","iopub.execute_input":"2025-01-23T08:35:27.202964Z","iopub.status.idle":"2025-01-23T08:37:30.903442Z","shell.execute_reply.started":"2025-01-23T08:35:27.202930Z","shell.execute_reply":"2025-01-23T08:37:30.902608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    return tp_rounded\n\nimputer = SimpleImputer(strategy='median')\n\nensemble = VotingRegressor(estimators=[\n    ('lgb', Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(random_state=SEED))])),\n    ('xgb', Pipeline(steps=[('imputer', imputer), ('regressor', XGBRegressor(random_state=SEED))])),\n    ('cat', Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(random_state=SEED, silent=True))])),\n    ('rf', Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=SEED))])),\n    ('gb', Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=SEED))])),\n    ('tabnet', Pipeline(steps=[('imputer', imputer), ('regressor', TabNetWrapper(**TabNet_Params))]))  # New:TabNet\n])\n\nSubmission3 = TrainML(ensemble, test)\n\nSubmission3 = TrainML(ensemble, test)\nSubmission3 = pd.DataFrame({\n    'id': sample['id'],\n    'sii': Submission3\n})\n\nSubmission3","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:37:30.904292Z","iopub.execute_input":"2025-01-23T08:37:30.904550Z","iopub.status.idle":"2025-01-23T08:44:49.534091Z","shell.execute_reply.started":"2025-01-23T08:37:30.904528Z","shell.execute_reply":"2025-01-23T08:44:49.533361Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub1 = Submission1\nsub2 = Submission2\nsub3 = Submission3\n\nsub1 = sub1.sort_values(by='id').reset_index(drop=True)\nsub2 = sub2.sort_values(by='id').reset_index(drop=True)\nsub3 = sub3.sort_values(by='id').reset_index(drop=True)\n\ncombined = pd.DataFrame({\n    'id': sub1['id'],\n    'sii_1': sub1['sii'],\n    'sii_2': sub2['sii'],\n    'sii_3': sub3['sii']\n})\n\ndef majority_vote(row):\n    return row.mode()[0]\n\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\n\nfinal_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n\nfinal_submission.to_csv('submission.csv', index=False)\n\nprint(\"Majority voting completed and saved to 'Final_Submission.csv'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:44:49.534967Z","iopub.execute_input":"2025-01-23T08:44:49.535271Z","iopub.status.idle":"2025-01-23T08:44:49.549161Z","shell.execute_reply.started":"2025-01-23T08:44:49.535250Z","shell.execute_reply":"2025-01-23T08:44:49.548365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-23T08:44:49.550091Z","iopub.execute_input":"2025-01-23T08:44:49.550409Z","iopub.status.idle":"2025-01-23T08:44:49.566800Z","shell.execute_reply.started":"2025-01-23T08:44:49.550376Z","shell.execute_reply":"2025-01-23T08:44:49.565946Z"}},"outputs":[],"execution_count":null}]}