{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":7453542,"sourceType":"datasetVersion","datasetId":921302}],"dockerImageVersionId":30823,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib.gridspec as gridspec\nimport seaborn as sns\nimport warnings\nimport polars as pl\nimport os\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:44.704643Z","iopub.execute_input":"2024-12-22T15:09:44.704924Z","iopub.status.idle":"2024-12-22T15:09:45.909563Z","shell.execute_reply.started":"2024-12-22T15:09:44.704896Z","shell.execute_reply":"2024-12-22T15:09:45.908734Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Import Data","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ndata_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:45.910456Z","iopub.execute_input":"2024-12-22T15:09:45.910893Z","iopub.status.idle":"2024-12-22T15:09:45.988060Z","shell.execute_reply.started":"2024-12-22T15:09:45.910862Z","shell.execute_reply":"2024-12-22T15:09:45.987214Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train","metadata":{}},{"cell_type":"code","source":"display(train.head())\nprint(f\"Train shape: {train.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:45.988923Z","iopub.execute_input":"2024-12-22T15:09:45.989272Z","iopub.status.idle":"2024-12-22T15:09:46.022443Z","shell.execute_reply.started":"2024-12-22T15:09:45.989238Z","shell.execute_reply":"2024-12-22T15:09:46.021637Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Test","metadata":{}},{"cell_type":"code","source":"display(test.head())\nprint(f\"Test shape: {test.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:46.023316Z","iopub.execute_input":"2024-12-22T15:09:46.023613Z","iopub.status.idle":"2024-12-22T15:09:46.051096Z","shell.execute_reply.started":"2024-12-22T15:09:46.023584Z","shell.execute_reply":"2024-12-22T15:09:46.050114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:46.052245Z","iopub.execute_input":"2024-12-22T15:09:46.052530Z","iopub.status.idle":"2024-12-22T15:09:46.066868Z","shell.execute_reply.started":"2024-12-22T15:09:46.052502Z","shell.execute_reply":"2024-12-22T15:09:46.066084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:46.069265Z","iopub.execute_input":"2024-12-22T15:09:46.069512Z","iopub.status.idle":"2024-12-22T15:09:46.082947Z","shell.execute_reply.started":"2024-12-22T15:09:46.069489Z","shell.execute_reply":"2024-12-22T15:09:46.082085Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data dict","metadata":{}},{"cell_type":"code","source":"data_dict.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:46.084704Z","iopub.execute_input":"2024-12-22T15:09:46.084898Z","iopub.status.idle":"2024-12-22T15:09:46.104103Z","shell.execute_reply.started":"2024-12-22T15:09:46.084880Z","shell.execute_reply":"2024-12-22T15:09:46.103437Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Features EDA","metadata":{}},{"cell_type":"markdown","source":"## Hàm thống kê mô tả colums","metadata":{}},{"cell_type":"code","source":"def calculate_stats(data, columns):\n    if isinstance(columns, str):\n        columns = [columns]\n\n    stats = []\n    for col in columns:\n        if data[col].dtype in ['object', 'category']:\n            counts = data[col].value_counts(dropna=False, sort=False)\n            percents = data[col].value_counts(normalize=True, dropna=False, sort=False) * 100\n            formatted = counts.astype(str) + ' (' + percents.round(2).astype(str) + '%)'\n            stats_col = pd.DataFrame({'count (%)': formatted})\n            stats.append(stats_col)\n        else:\n            stats_col = data[col].describe().to_frame().transpose()\n            stats_col['missing'] = data[col].isnull().sum()\n            stats_col.index.name = col\n            stats.append(stats_col)\n\n    return pd.concat(stats, axis=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:46.104842Z","iopub.execute_input":"2024-12-22T15:09:46.105140Z","iopub.status.idle":"2024-12-22T15:09:46.117953Z","shell.execute_reply.started":"2024-12-22T15:09:46.105094Z","shell.execute_reply":"2024-12-22T15:09:46.117184Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_cols = set(train.columns)\ntest_cols = set(test.columns)\ncolumns_not_in_test = sorted(list(train_cols - test_cols))\ndata_dict[data_dict['Field'].isin(columns_not_in_test)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:46.118783Z","iopub.execute_input":"2024-12-22T15:09:46.119061Z","iopub.status.idle":"2024-12-22T15:09:46.141801Z","shell.execute_reply.started":"2024-12-22T15:09:46.119040Z","shell.execute_reply":"2024-12-22T15:09:46.141183Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Parent-Child Internet Addiction Test (PCIAT): contains 20 items (PCIAT-PCIAT_01 to PCIAT-PCIAT_20), each assessing a different aspect of a child's behavior related to internet use. The items are answered on a scale (from 0 to 5), and the total score provides an indication of the severity of internet addiction.\r\n\r\nWe also have season of participation iPCIAT-Seasonon and total Score iPCIAT-PCIAT_Totalal; so there are 22 PCIAT test-related columns in total.\r\n\r\nLet's verify that PCIAT-PCIAT_Total tal align with the corresponding sii categories by calculating its minimum and maximum scores for  esii sii categ","metadata":{}},{"cell_type":"code","source":"# Suppress specific FutureWarnings from seaborn\nwarnings.filterwarnings(\"ignore\", category=FutureWarning, module=\"seaborn._oldcore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:46.142499Z","iopub.execute_input":"2024-12-22T15:09:46.142923Z","iopub.status.idle":"2024-12-22T15:09:46.155454Z","shell.execute_reply.started":"2024-12-22T15:09:46.142851Z","shell.execute_reply":"2024-12-22T15:09:46.154780Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Target Variables and Internet use","metadata":{}},{"cell_type":"code","source":"pciat_min_max = train.groupby('sii')['PCIAT-PCIAT_Total'].agg(['min', 'max'])\npciat_min_max = pciat_min_max.rename(\n    columns={'min': 'Minimum PCIAT total Score', 'max': 'Maximum total PCIAT Score'}\n)\npciat_min_max","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:46.156074Z","iopub.execute_input":"2024-12-22T15:09:46.156293Z","iopub.status.idle":"2024-12-22T15:09:46.183152Z","shell.execute_reply.started":"2024-12-22T15:09:46.156276Z","shell.execute_reply":"2024-12-22T15:09:46.182391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dict[data_dict['Field'] == 'PCIAT-PCIAT_Total']['Value Labels'].iloc[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:46.183918Z","iopub.execute_input":"2024-12-22T15:09:46.184220Z","iopub.status.idle":"2024-12-22T15:09:46.190182Z","shell.execute_reply.started":"2024-12-22T15:09:46.184198Z","shell.execute_reply":"2024-12-22T15:09:46.189446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_with_sii = train[train['sii'].notna()][columns_not_in_test]\ntrain_with_sii[train_with_sii.isna().any(axis=1)].head().style.map(\n    lambda x: 'background-color: #FFC0CB' if pd.isna(x) else ''\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:46.191015Z","iopub.execute_input":"2024-12-22T15:09:46.191320Z","iopub.status.idle":"2024-12-22T15:09:46.267276Z","shell.execute_reply.started":"2024-12-22T15:09:46.191288Z","shell.execute_reply":"2024-12-22T15:09:46.266629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PCIAT_cols = [f'PCIAT-PCIAT_{i+1:02d}' for i in range(20)]\nrecalc_total_score = train_with_sii[PCIAT_cols].sum(\n    axis=1, skipna=True\n)\n(recalc_total_score == train_with_sii['PCIAT-PCIAT_Total']).all()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:46.267907Z","iopub.execute_input":"2024-12-22T15:09:46.268157Z","iopub.status.idle":"2024-12-22T15:09:46.275724Z","shell.execute_reply.started":"2024-12-22T15:09:46.268139Z","shell.execute_reply":"2024-12-22T15:09:46.275043Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def recalculate_sii(row):\n    if pd.isna(row['PCIAT-PCIAT_Total']):\n        return np.nan\n    max_possible = row['PCIAT-PCIAT_Total'] + row[PCIAT_cols].isna().sum() * 5\n    if row['PCIAT-PCIAT_Total'] <= 30 and max_possible <= 30:\n        return 0\n    elif 31 <= row['PCIAT-PCIAT_Total'] <= 49 and max_possible <= 49:\n        return 1\n    elif 50 <= row['PCIAT-PCIAT_Total'] <= 79 and max_possible <= 79:\n        return 2\n    elif row['PCIAT-PCIAT_Total'] >= 80 and max_possible >= 80:\n        return 3\n    return np.nan\n\ntrain['recalc_sii'] = train.apply(recalculate_sii, axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:46.276693Z","iopub.execute_input":"2024-12-22T15:09:46.276963Z","iopub.status.idle":"2024-12-22T15:09:47.422012Z","shell.execute_reply.started":"2024-12-22T15:09:46.276935Z","shell.execute_reply":"2024-12-22T15:09:47.421373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mismatch_rows = train[\n    (train['recalc_sii'] != train['sii']) & train['sii'].notna()\n]\n\nmismatch_rows[PCIAT_cols + [\n    'PCIAT-PCIAT_Total', 'sii', 'recalc_sii'\n]].style.map(\n    lambda x: 'background-color: #FFC0CB' if pd.isna(x) else ''\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:47.422801Z","iopub.execute_input":"2024-12-22T15:09:47.423027Z","iopub.status.idle":"2024-12-22T15:09:47.443142Z","shell.execute_reply.started":"2024-12-22T15:09:47.423007Z","shell.execute_reply":"2024-12-22T15:09:47.442469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['sii'] = train['recalc_sii']\ntrain['complete_resp_total'] = train['PCIAT-PCIAT_Total'].where(\n    train[PCIAT_cols].notna().all(axis=1), np.nan\n)\n\nsii_map = {0: '0 (None)', 1: '1 (Mild)', 2: '2 (Moderate)', 3: '3 (Severe)'}\ntrain['sii'] = train['sii'].map(sii_map).fillna('Missing')\n\nsii_order = ['Missing', '0 (None)', '1 (Mild)', '2 (Moderate)', '3 (Severe)']\ntrain['sii'] = pd.Categorical(train['sii'], categories=sii_order, ordered=True)\n\ntrain.drop(columns='recalc_sii', inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:47.444060Z","iopub.execute_input":"2024-12-22T15:09:47.444412Z","iopub.status.idle":"2024-12-22T15:09:47.464864Z","shell.execute_reply.started":"2024-12-22T15:09:47.444376Z","shell.execute_reply":"2024-12-22T15:09:47.464208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sii_counts = train['sii'].value_counts().reset_index()\ntotal = sii_counts['count'].sum()\nsii_counts['percentage'] = (sii_counts['count'] / total) * 100\n\nfig, axes = plt.subplots(1, 2, figsize=(14, 5))\n\n# SII\nsns.barplot(x='sii', y='count', data=sii_counts, palette='Blues_d', ax=axes[0])\naxes[0].set_title('Distribution of Severity Impairment Index (sii)', fontsize=14)\nfor p in axes[0].patches:\n    height = p.get_height()\n    percentage = sii_counts.loc[sii_counts['count'] == height, 'percentage'].values[0]\n    axes[0].text(\n        p.get_x() + p.get_width() / 2,\n        height + 5, f'{int(height)} ({percentage:.1f}%)',\n        ha=\"center\", fontsize=12\n    )\n\n# PCIAT_Total for complete responses\nsns.histplot(train['complete_resp_total'].dropna(), bins=20, ax=axes[1])\naxes[1].set_title('Distribution of PCIAT_Total', fontsize=14)\naxes[1].set_xlabel('PCIAT_Total for Complete PCIAT Responses')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:47.465633Z","iopub.execute_input":"2024-12-22T15:09:47.465929Z","iopub.status.idle":"2024-12-22T15:09:47.988224Z","shell.execute_reply.started":"2024-12-22T15:09:47.465902Z","shell.execute_reply":"2024-12-22T15:09:47.987375Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Note: Apparently, 40% of the participants were not affected by Internet use, 31% were not assessed, and only the minority (~10%) are moderately to severely impaired. There are 307 participants who scored 0 on all PCIAT questions.","metadata":{}},{"cell_type":"markdown","source":"SII by sex and age","metadata":{}},{"cell_type":"code","source":"assert train['Basic_Demos-Age'].isna().sum() == 0\nassert train['Basic_Demos-Sex'].isna().sum() == 0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:47.988987Z","iopub.execute_input":"2024-12-22T15:09:47.989253Z","iopub.status.idle":"2024-12-22T15:09:47.993403Z","shell.execute_reply.started":"2024-12-22T15:09:47.989233Z","shell.execute_reply":"2024-12-22T15:09:47.992503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['Age Group'] = pd.cut(\n    train['Basic_Demos-Age'],\n    bins=[4, 10, 18, 22],\n    labels=['Children (5-12)', 'Adolescents (13-18)', 'Adults (19-22)']\n)\ncalculate_stats(train, 'Age Group')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:47.998215Z","iopub.execute_input":"2024-12-22T15:09:47.998417Z","iopub.status.idle":"2024-12-22T15:09:48.019625Z","shell.execute_reply.started":"2024-12-22T15:09:47.998399Z","shell.execute_reply":"2024-12-22T15:09:48.018871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sex_map = {0: 'Male', 1: 'Female'}\ntrain['Basic_Demos-Sex'] = train['Basic_Demos-Sex'].map(sex_map)\ncalculate_stats(train, 'Basic_Demos-Sex')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:48.022023Z","iopub.execute_input":"2024-12-22T15:09:48.022246Z","iopub.status.idle":"2024-12-22T15:09:48.031792Z","shell.execute_reply.started":"2024-12-22T15:09:48.022227Z","shell.execute_reply":"2024-12-22T15:09:48.031166Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize=(18, 5))\n\n# SII by Age\nsns.boxplot(y=train['Basic_Demos-Age'], x=train['sii'], ax=axes[0], palette=\"Set3\")\naxes[0].set_title('SII by Age')\naxes[0].set_ylabel('Age')\naxes[0].set_xlabel('SII')\n\n# Complete PCIAT Responses by Age Group\nsns.boxplot(\n    x='Age Group', y='complete_resp_total',\n    data=train, palette=\"Set3\", ax=axes[1]\n)\naxes[1].set_title('Complete PCIAT Responses by Age Group')\naxes[1].set_ylabel('PCIAT_Total for Complete Responses')\naxes[1].set_xlabel('Age Group')\n\n# PCIAT_Total by Sex\nsns.histplot(\n    data=train, x='complete_resp_total',\n    hue='Basic_Demos-Sex', multiple='stack',\n    palette=\"Set3\", bins=20, ax=axes[2]\n)\naxes[2].set_title('PCIAT_Total Distribution by Sex')\naxes[2].set_xlabel('PCIAT_Total for Complete Responses')\naxes[2].set_ylabel('Frequency')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:48.032315Z","iopub.execute_input":"2024-12-22T15:09:48.032497Z","iopub.status.idle":"2024-12-22T15:09:48.816523Z","shell.execute_reply.started":"2024-12-22T15:09:48.032480Z","shell.execute_reply":"2024-12-22T15:09:48.815712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nstats = train.groupby(['Age Group', 'sii']).size().unstack(fill_value=0)\nfig, axes = plt.subplots(1, len(stats), figsize=(18, 5))\n\nfor i, age_group in enumerate(stats.index):\n    group_counts = stats.loc[age_group] / stats.loc[age_group].sum()\n    axes[i].pie(\n        group_counts, labels=group_counts.index, autopct='%1.1f%%',\n        startangle=90, colors=sns.color_palette(\"Set3\"),\n        labeldistance=1.05, pctdistance=0.80\n    )\n    axes[i].set_title(f'SII Distribution for {age_group}')\n    axes[i].axis('equal')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:48.817527Z","iopub.execute_input":"2024-12-22T15:09:48.817744Z","iopub.status.idle":"2024-12-22T15:09:49.311744Z","shell.execute_reply.started":"2024-12-22T15:09:48.817726Z","shell.execute_reply":"2024-12-22T15:09:49.310932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = train.groupby(['Age Group', 'sii']).size().unstack(fill_value=0)\nstats_prop = stats.div(stats.sum(axis=1), axis=0) * 100\n\nstats = stats.astype(str) +' (' + stats_prop.round(1).astype(str) + '%)'\nstats","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:49.312627Z","iopub.execute_input":"2024-12-22T15:09:49.312866Z","iopub.status.idle":"2024-12-22T15:09:49.330633Z","shell.execute_reply.started":"2024-12-22T15:09:49.312844Z","shell.execute_reply":"2024-12-22T15:09:49.329972Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = train[train['sii'] != 'Missing'].groupby(\n    ['Age Group', 'sii']\n).size().unstack(fill_value=0)\nstats_prop = stats.div(stats.sum(axis=1), axis=0) * 100\n\nstats = stats.astype(str) +' (' + stats_prop.round(1).astype(str) + '%)'\nstats","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:49.331326Z","iopub.execute_input":"2024-12-22T15:09:49.331551Z","iopub.status.idle":"2024-12-22T15:09:49.350026Z","shell.execute_reply.started":"2024-12-22T15:09:49.331531Z","shell.execute_reply":"2024-12-22T15:09:49.349202Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"line-height:24px; font-size:16px;border-left: 5px solid silver; padding-left: 26px;\"> \n    💡 Note: \n    <ul style=\"list-style:circle\">\n<li>The box plots are different representations of the target variable in it's categorised (SII) and numerical (PCIAT_Total) form. They show that higher SII scores are generally associated with older age groups, but there's considerable overlap in the age ranges within each category, and the median PCIAT_Total is higher in adolescents, suggesting a U-shaped relationship between age and PIU impairment (the peak of Internet-related problems may occur during adolescence).\n<li>Accordingly, in the pie charts, the distribution of SII for children and adults is skewed towards lower values (none and mild), whereas, for adolescents, the distribution is more balanced across the categories of none, mild and moderate.\n<li>But what about the numbers (see tables)? The number of adolescents is much lower than that of children, and the number of adult participants is extremely low (88 in total and only 36 with SII)!\n<li>As we have seen from the graphs in the previous section, the overall distribution of SII is skewed towards lower values and severe cases are rare. So there may be relationships that we cannot see with such unequal sample sizes and under-representation of severe cases.\n<li>The differences between males and females are relatively subtle.\n    </ul>\n</div>\n","metadata":{}},{"cell_type":"markdown","source":"##  Internet Use","metadata":{}},{"cell_type":"code","source":"data = train[train['PreInt_EduHx-computerinternet_hoursday'].notna()]\nage_range = data['Basic_Demos-Age']\nprint(\n    f\"Age range for participants with measured PreInt_EduHx-computerinternet_hoursday data:\"\n    f\" {age_range.min()} - {age_range.max()} years\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:49.350918Z","iopub.execute_input":"2024-12-22T15:09:49.351213Z","iopub.status.idle":"2024-12-22T15:09:49.357275Z","shell.execute_reply.started":"2024-12-22T15:09:49.351184Z","shell.execute_reply":"2024-12-22T15:09:49.356560Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"param_map = {0: '< 1h/day', 1: '~ 1h/day', 2: '~ 2hs/day', 3: '> 3hs/day'}\ntrain['internet_use_encoded'] = train[\n    'PreInt_EduHx-computerinternet_hoursday'\n].map(param_map).fillna('Missing')\n\nparam_ord = ['Missing', '< 1h/day', '~ 1h/day', '~ 2hs/day', '> 3hs/day']\ntrain['internet_use_encoded'] = pd.Categorical(\n    train['internet_use_encoded'], categories=param_ord,\n    ordered=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:49.358118Z","iopub.execute_input":"2024-12-22T15:09:49.358448Z","iopub.status.idle":"2024-12-22T15:09:49.378714Z","shell.execute_reply.started":"2024-12-22T15:09:49.358418Z","shell.execute_reply":"2024-12-22T15:09:49.378075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calculate_stats(train, 'PreInt_EduHx-Season')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:49.379465Z","iopub.execute_input":"2024-12-22T15:09:49.379699Z","iopub.status.idle":"2024-12-22T15:09:49.397854Z","shell.execute_reply.started":"2024-12-22T15:09:49.379667Z","shell.execute_reply":"2024-12-22T15:09:49.397193Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize=(18, 5))\n\n# Hours of Internet Use\nax1 = sns.countplot(x='internet_use_encoded', data=train, palette=\"Set3\", ax=axes[0])\naxes[0].set_title('Distribution of Hours of Internet Use')\naxes[0].set_xlabel('Hours per Day Group')\naxes[0].set_ylabel('Count')\n\ntotal = len(train['internet_use_encoded'])\nfor p in ax1.patches:\n    count = int(p.get_height())\n    percentage = '{:.1f}%'.format(100 * count / total)\n    ax1.annotate(f'{count} ({percentage})', (p.get_x() + p.get_width() / 2., p.get_height()), \n                 ha='center', va='baseline', fontsize=10, color='black', xytext=(0, 5), \n                 textcoords='offset points')\n\n# Hours of Internet Use by Age\nsns.boxplot(y=train['Basic_Demos-Age'], x=train['internet_use_encoded'], ax=axes[1], palette=\"Set3\")\naxes[1].set_title('Hours of Internet Use by Age')\naxes[1].set_ylabel('Age')\naxes[1].set_xlabel('Hours per Day Group')\n\n# Hours of Internet Use (numeric) by Age Group\nsns.boxplot(y='PreInt_EduHx-computerinternet_hoursday', x='Age Group', data=train, ax=axes[2], palette=\"Set3\")\naxes[2].set_title('Internet Hours by Age Group')\naxes[2].set_ylabel('Hours per Day (Numeric)')\naxes[2].set_xlabel('Age Group')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:49.398583Z","iopub.execute_input":"2024-12-22T15:09:49.398792Z","iopub.status.idle":"2024-12-22T15:09:50.154432Z","shell.execute_reply.started":"2024-12-22T15:09:49.398775Z","shell.execute_reply":"2024-12-22T15:09:50.153519Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = train.groupby(\n    ['Age Group', 'internet_use_encoded']\n).size().unstack(fill_value=0)\nfig, axes = plt.subplots(1, len(stats), figsize=(18, 5))\n\nfor i, age_group in enumerate(stats.index):\n    group_counts = stats.loc[age_group] / stats.loc[age_group].sum()\n    axes[i].pie(group_counts, labels=group_counts.index, autopct='%1.1f%%',\n                startangle=90, colors=sns.color_palette(\"Set3\"), labeldistance=1.1)\n    axes[i].set_title(f'Distribution of Hours of Internet Use\\n{age_group}')\n    axes[i].axis('equal')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:50.155331Z","iopub.execute_input":"2024-12-22T15:09:50.155626Z","iopub.status.idle":"2024-12-22T15:09:50.635273Z","shell.execute_reply.started":"2024-12-22T15:09:50.155595Z","shell.execute_reply":"2024-12-22T15:09:50.634472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_non_na = train.dropna(subset=['PreInt_EduHx-computerinternet_hoursday'])\nrows = (train_non_na['PreInt_EduHx-computerinternet_hoursday'] == 3).sum()\nprint(f\"Non-NA Rows - Internet use 3h or more: {(rows / len(train_non_na)) * 100:.2f}%\")\n\nrows = (train_non_na['PreInt_EduHx-computerinternet_hoursday'] == 0).sum()\nprint(f\"Non-NA Rows - Internet use 1h or less: {(rows / len(train_non_na)) * 100:.2f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:50.636331Z","iopub.execute_input":"2024-12-22T15:09:50.636579Z","iopub.status.idle":"2024-12-22T15:09:50.645718Z","shell.execute_reply.started":"2024-12-22T15:09:50.636556Z","shell.execute_reply":"2024-12-22T15:09:50.644810Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = train.groupby(['Basic_Demos-Sex', 'internet_use_encoded']\n).size().unstack(fill_value=0)\nstats_prop = stats.div(stats.sum(axis=1), axis=0) * 100\n\nstats = stats.astype(str) +' (' + stats_prop.round(1).astype(str) + '%)'\nstats","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:50.646638Z","iopub.execute_input":"2024-12-22T15:09:50.646945Z","iopub.status.idle":"2024-12-22T15:09:50.670451Z","shell.execute_reply.started":"2024-12-22T15:09:50.646906Z","shell.execute_reply":"2024-12-22T15:09:50.669668Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"line-height:24px; font-size:16px;border-left: 5px solid silver; padding-left: 26px;\"> \n    💡 Note: \n    <ul style=\"list-style:circle\">\n<li>Internet usage data is missing for 16.6% of participants, while 38.5% reported using the Internet less than hour a day.\n<li>Similar to the SII data, the box plots reveals that higher daily internet usage is associated with older age, with considerable overlap in age ranges within each internet usage category. But here both the categorical and numeric representations of hours spent online indicate a consistent linear relationship.\n<li>The pie charts for age groups are well aligned and shows the same.\n<li>Creating an interaction feature between internet use and age could potentially be useful for modeling.\n<li>Internet use is fairly similar for both sexes.\n    </ul>\n</div>","metadata":{}},{"cell_type":"markdown","source":"## Feature EDA by group","metadata":{}},{"cell_type":"code","source":"groups = data_dict.groupby('Instrument')['Field'].apply(list).to_dict()\n\nfor instrument, features in groups.items():\n    print(f\"{instrument}: {features}\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:50.671242Z","iopub.execute_input":"2024-12-22T15:09:50.671526Z","iopub.status.idle":"2024-12-22T15:09:50.679412Z","shell.execute_reply.started":"2024-12-22T15:09:50.671497Z","shell.execute_reply":"2024-12-22T15:09:50.678535Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dict = data_dict[data_dict['Instrument'] != 'Parent-Child Internet Addiction Test']\ncontinuous_cols = data_dict[data_dict['Type'].str.contains(\n    'float|int', case=False\n)]['Field'].tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:50.680327Z","iopub.execute_input":"2024-12-22T15:09:50.680575Z","iopub.status.idle":"2024-12-22T15:09:50.696152Z","shell.execute_reply.started":"2024-12-22T15:09:50.680556Z","shell.execute_reply":"2024-12-22T15:09:50.695451Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Physical Measures","metadata":{}},{"cell_type":"code","source":"groups.get('Physical Measures', [])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:50.697106Z","iopub.execute_input":"2024-12-22T15:09:50.697438Z","iopub.status.idle":"2024-12-22T15:09:50.710377Z","shell.execute_reply.started":"2024-12-22T15:09:50.697407Z","shell.execute_reply":"2024-12-22T15:09:50.709735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_physical = groups.get('Physical Measures', [])\ncols = [col for col in features_physical if col in continuous_cols]\n\nplt.figure(figsize=(24, 10))\nn_cols = 4\nn_rows = len(cols) // n_cols + 1\n\nfor i, col in enumerate(cols):\n    plt.subplot(n_rows, n_cols, i + 1)\n    train[col].hist(bins=20)\n    plt.title(col)\n\nplt.subplot(n_rows, n_cols, len(cols) + 1)\nseason_counts = train['Physical-Season'].value_counts(dropna=False)\nplt.pie(\n    season_counts,\n    labels=season_counts.index,\n    autopct='%1.1f%%',\n    startangle=90,\n    colors=sns.color_palette(\"Set3\")\n)\nplt.title('Physical-Season')\n\nplt.suptitle('Histograms for Physical Measures and Physical-Season Pie Chart', y=1.05)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:50.711030Z","iopub.execute_input":"2024-12-22T15:09:50.711253Z","iopub.status.idle":"2024-12-22T15:09:52.329079Z","shell.execute_reply.started":"2024-12-22T15:09:50.711235Z","shell.execute_reply":"2024-12-22T15:09:52.328240Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calculate_stats(train, cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:52.330204Z","iopub.execute_input":"2024-12-22T15:09:52.330523Z","iopub.status.idle":"2024-12-22T15:09:52.358091Z","shell.execute_reply.started":"2024-12-22T15:09:52.330491Z","shell.execute_reply":"2024-12-22T15:09:52.357485Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"line-height:24px; font-size:16px;border-left: 5px solid silver; padding-left: 26px;\"> \n    💡 Note: \n    <ul style=\"list-style:circle\">\n<li>Weight and height both increase with age, and waist circumference and weight are highly correlated, as expected.\n<li>However, there are individuals who are unusually tall for their age group or who are extremely overweight.\n<li>There are also a few outliers in the waist circumference measurements, which are possible artifacts (e.g. 100 cm for a weight of 40 kg).\n<li>The problem with data cleaning here is that we cannot guess which of the data is correct. For example, we may see an unrealistic combination of a waist circumference of 100cm and a weight of 40kg for a participant, but where is the error in the waist circumference or the weight? Or a height of around 175cm for a child of 7... has the height or age been entered incorrectly? Or this is true data and the child has gigantism or another disorder related to the growth hormone?\n    </ul>\n</div>","metadata":{}},{"cell_type":"markdown","source":"### FitnessGram","metadata":{}},{"cell_type":"code","source":"data_dict[data_dict['Instrument'] == 'FitnessGram Child']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:52.358904Z","iopub.execute_input":"2024-12-22T15:09:52.359257Z","iopub.status.idle":"2024-12-22T15:09:52.369297Z","shell.execute_reply.started":"2024-12-22T15:09:52.359223Z","shell.execute_reply":"2024-12-22T15:09:52.368632Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fgc_data_dict = data_dict[data_dict['Instrument'] == 'FitnessGram Child']\n\nfgc_columns = []\n\nfor index, row in fgc_data_dict.iterrows():\n    if '_Zone' not in row['Field']:\n        measure_field = row['Field']\n        measure_desc = row['Description']\n        \n        zone_field = measure_field + '_Zone'\n        zone_row = fgc_data_dict[fgc_data_dict['Field'] == zone_field]\n        \n        if not zone_row.empty:\n            zone_desc = zone_row['Description'].values[0]\n            fgc_columns.append((measure_field, zone_field, measure_desc, zone_desc))\n            \nfig, axes = plt.subplots(2, 4, figsize=(24, 10))\n\nfor idx, (measure, zone, measure_desc, zone_desc) in enumerate(fgc_columns):\n    row = idx // 4\n    col = idx % 4\n    \n    sns.histplot(\n        data=train, x=measure,\n        hue=zone, bins=20, palette='Set2',\n        ax=axes[row, col], kde=True\n    )\n    axes[row, col].set_title(f'{measure_desc}')\n\nseason_counts = train['FGC-Season'].value_counts(normalize=True)\naxes[1, 3].pie(\n    season_counts, labels=season_counts.index,\n    autopct='%1.1f%%', startangle=90,\n    colors=sns.color_palette(\"Set3\")\n)\naxes[1, 3].set_title('Season of participation')\naxes[1, 3].axis('equal') \n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:52.370063Z","iopub.execute_input":"2024-12-22T15:09:52.370347Z","iopub.status.idle":"2024-12-22T15:09:55.032752Z","shell.execute_reply.started":"2024-12-22T15:09:52.370316Z","shell.execute_reply":"2024-12-22T15:09:55.031808Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Sleep Disturbance Scale","metadata":{}},{"cell_type":"code","source":"groups.get('Sleep Disturbance Scale', [])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:55.033484Z","iopub.execute_input":"2024-12-22T15:09:55.033711Z","iopub.status.idle":"2024-12-22T15:09:55.038642Z","shell.execute_reply.started":"2024-12-22T15:09:55.033691Z","shell.execute_reply":"2024-12-22T15:09:55.037875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = train[train['SDS-SDS_Total_Raw'].notnull()]\nage_range = data['Basic_Demos-Age']\nprint(\n    f\"Age range for participants with SDS-SDS_Total_Raw data:\"\n    f\" {age_range.min()} - {age_range.max()} years\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:55.039255Z","iopub.execute_input":"2024-12-22T15:09:55.039441Z","iopub.status.idle":"2024-12-22T15:09:55.057094Z","shell.execute_reply.started":"2024-12-22T15:09:55.039424Z","shell.execute_reply":"2024-12-22T15:09:55.056359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(18, 5))\n\n# SDS-Season (Pie Chart)\nplt.subplot(1, 3, 1)\nsds_season_counts = train['SDS-Season'].value_counts(normalize=True)\nplt.pie(\n    sds_season_counts, \n    labels=sds_season_counts.index, \n    autopct='%1.1f%%', \n    startangle=90, \n    colors=sns.color_palette(\"Set3\")\n)\nplt.title('SDS-Season')\n\n# SDS-SDS_Total_Raw\nplt.subplot(1, 3, 2)\nsns.histplot(train['SDS-SDS_Total_Raw'].dropna(), bins=20, kde=True)\nplt.title('SDS-SDS_Total_Raw')\nplt.xlabel('Value')\n\n# SDS-SDS_Total_T\nplt.subplot(1, 3, 3)\nsns.histplot(train['SDS-SDS_Total_T'].dropna(), bins=20, kde=True)\nplt.title('SDS-SDS_Total_T')\nplt.xlabel('Value')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:55.057965Z","iopub.execute_input":"2024-12-22T15:09:55.058217Z","iopub.status.idle":"2024-12-22T15:09:55.746002Z","shell.execute_reply.started":"2024-12-22T15:09:55.058197Z","shell.execute_reply":"2024-12-22T15:09:55.745227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calculate_stats(train, ['SDS-SDS_Total_Raw', 'SDS-SDS_Total_T'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:09:55.747033Z","iopub.execute_input":"2024-12-22T15:09:55.747352Z","iopub.status.idle":"2024-12-22T15:09:55.765366Z","shell.execute_reply.started":"2024-12-22T15:09:55.747315Z","shell.execute_reply":"2024-12-22T15:09:55.764480Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"line-height:24px; font-size:16px;border-left: 5px solid silver; padding-left: 26px;\"> \n    💡 Note: \n    <ul style=\"list-style:circle\">\n<li>Both the raw and T-scores for sleep disturbance are moderately variable, with some extreme values indicating severe sleep disturbances in a subset of participants.\n<li>Further Analysis (coming soon): to explore whether specific demographic factors (e.g., age, gender, season) are associated with higher sleep disturbance scores.\n    </ul>\n</div>","metadata":{}},{"cell_type":"markdown","source":"## Time series","metadata":{}},{"cell_type":"code","source":"df = pd.read_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id=0745c390/part-0.parquet')\n\n# Display the DataFrame\nprint(df.info())\nprint(df.describe())\nprint(\"Missing values:\\n\", df.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:17:38.644313Z","iopub.execute_input":"2024-12-22T15:17:38.644631Z","iopub.status.idle":"2024-12-22T15:17:38.703424Z","shell.execute_reply.started":"2024-12-22T15:17:38.644606Z","shell.execute_reply":"2024-12-22T15:17:38.702678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert time_of_day to datetime if necessary\ndf['time_of_day'] = pd.to_datetime(df['time_of_day'], unit='ns')\n\n# Calculate magnitude of motion\ndf['magnitude'] = (df['X']**2 + df['Y']**2 + df['Z']**2)**0.5\n\n# Proportion of non-wear time\nnon_wear_ratio = df['non-wear_flag'].mean()\nprint(f\"Non-wear time proportion: {non_wear_ratio:.2%}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:26:37.201497Z","iopub.execute_input":"2024-12-22T15:26:37.201841Z","iopub.status.idle":"2024-12-22T15:26:37.348915Z","shell.execute_reply.started":"2024-12-22T15:26:37.201813Z","shell.execute_reply":"2024-12-22T15:26:37.348066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Correlation heatmap\nplt.figure(figsize=(10, 8))\nsns.heatmap(df.corr(), annot=True, fmt='.2f', cmap='coolwarm')\nplt.title(\"Correlation Heatmap\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:26:38.827967Z","iopub.execute_input":"2024-12-22T15:26:38.828304Z","iopub.status.idle":"2024-12-22T15:26:39.475333Z","shell.execute_reply.started":"2024-12-22T15:26:38.828278Z","shell.execute_reply":"2024-12-22T15:26:39.474574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ENMO over time\nplt.figure(figsize=(12, 6))\nplt.plot(df['time_of_day'], df['enmo'], label='ENMO', alpha=0.8)\nplt.title(\"ENMO Over Time\")\nplt.xlabel(\"Time\")\nplt.ylabel(\"ENMO\")\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:26:41.827328Z","iopub.execute_input":"2024-12-22T15:26:41.827606Z","iopub.status.idle":"2024-12-22T15:26:42.174163Z","shell.execute_reply.started":"2024-12-22T15:26:41.827585Z","shell.execute_reply":"2024-12-22T15:26:42.173305Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- The ENMO (Estimated Net Metabolic Equivalent) values show higher variability during midday and evening hours (12:00 PM - 9:00 PM), indicating increased activity levels.\n- There are noticeable peaks that represent moments of intense activity, which might correspond to specific events or physical activities.","metadata":{}},{"cell_type":"code","source":"# Light vs. ENMO scatter plot\nplt.figure(figsize=(8, 6))\nplt.scatter(df['light'], df['enmo'], alpha=0.5)\nplt.title(\"Light vs. ENMO\")\nplt.xlabel(\"Light\")\nplt.ylabel(\"ENMO\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:26:44.273542Z","iopub.execute_input":"2024-12-22T15:26:44.273832Z","iopub.status.idle":"2024-12-22T15:26:44.560398Z","shell.execute_reply.started":"2024-12-22T15:26:44.273799Z","shell.execute_reply":"2024-12-22T15:26:44.559587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Average ENMO by weekday\nplt.figure(figsize=(8, 6))\ndf.groupby('weekday')['enmo'].mean().plot(kind='bar', color='skyblue')\nplt.title(\"Average ENMO by Weekday\")\nplt.xlabel(\"Weekday\")\nplt.ylabel(\"Average ENMO\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:26:46.845050Z","iopub.execute_input":"2024-12-22T15:26:46.845346Z","iopub.status.idle":"2024-12-22T15:26:47.013994Z","shell.execute_reply.started":"2024-12-22T15:26:46.845322Z","shell.execute_reply":"2024-12-22T15:26:47.013175Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- The average ENMO is highest on day 1 (Sunday) and day 5 (Friday).\n- Other weekdays show lower average ENMO levels, suggesting reduced or less intense activity during workdays.\n- This indicates a trend where users are more active during weekends or end-of-week periods.","metadata":{}},{"cell_type":"code","source":"# Motion magnitude over time\nplt.figure(figsize=(12, 6))\nplt.plot(df['time_of_day'], df['magnitude'], label='Motion Magnitude', color='orange', alpha=0.8)\nplt.title(\"Motion Magnitude Over Time\")\nplt.xlabel(\"Time\")\nplt.ylabel(\"Magnitude\")\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:26:48.759434Z","iopub.execute_input":"2024-12-22T15:26:48.759709Z","iopub.status.idle":"2024-12-22T15:26:49.085428Z","shell.execute_reply.started":"2024-12-22T15:26:48.759687Z","shell.execute_reply":"2024-12-22T15:26:49.084585Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Correlation","metadata":{}},{"cell_type":"code","source":"season_dtype = pl.Enum(['Spring', 'Summer', 'Fall', 'Winter'])\n\ntrain = (\n    pl.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n    .with_columns(pl.col('^.*Season$').cast(season_dtype))\n)\n\ntest = (\n    pl.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n    .with_columns(pl.col('^.*Season$').cast(season_dtype))\n)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.238945Z","iopub.status.idle":"2024-12-22T15:10:42.239273Z","shell.execute_reply":"2024-12-22T15:10:42.239154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"supervised_usable = (\n    train\n    .filter(pl.col('sii').is_not_null())\n)\n\nmissing_count = (\n    supervised_usable\n    .null_count()\n    .transpose(include_header=True,\n               header_name='feature',\n               column_names=['null_count'])\n    .sort('null_count', descending=True)\n    .with_columns((pl.col('null_count') / len(supervised_usable)).alias('null_ratio'))\n)\nplt.figure(figsize=(6, 15))\nplt.title(f'Missing values over the {len(supervised_usable)} samples which have a target')\nplt.barh(np.arange(len(missing_count)), missing_count.get_column('null_ratio'), color='coral', label='missing')\nplt.barh(np.arange(len(missing_count)), \n         1 - missing_count.get_column('null_ratio'),\n         left=missing_count.get_column('null_ratio'),\n         color='darkseagreen', label='available')\nplt.yticks(np.arange(len(missing_count)), missing_count.get_column('feature'))\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0, 1)\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.240381Z","iopub.status.idle":"2024-12-22T15:10:42.240737Z","shell.execute_reply":"2024-12-22T15:10:42.240564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(14, 12))\ncorr_matrix = supervised_usable.select([\n    'PCIAT-PCIAT_Total', 'Basic_Demos-Age', 'Basic_Demos-Sex', 'Physical-BMI', \n    'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n    'Physical-Diastolic_BP', 'Physical-Systolic_BP', 'Physical-HeartRate',\n    'PreInt_EduHx-computerinternet_hoursday', 'SDS-SDS_Total_T', 'PAQ_A-PAQ_A_Total',\n    'PAQ_C-PAQ_C_Total', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins','Fitness_Endurance-Time_Sec',\n    'FGC-FGC_CU', 'FGC-FGC_GSND','FGC-FGC_GSD','FGC-FGC_PU','FGC-FGC_SRL','FGC-FGC_SRR','FGC-FGC_TL','BIA-BIA_Activity_Level_num', \n    'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n    'BIA-BIA_FFMI','BIA-BIA_FMI', 'BIA-BIA_Fat','BIA-BIA_Frame_num','BIA-BIA_ICW','BIA-BIA_LDM','BIA-BIA_LST',\n    'BIA-BIA_SMM','BIA-BIA_TBW', 'sii'\n    # Add other relevant columns\n]).to_pandas().corr()\n\nsii_corr = corr_matrix['PCIAT-PCIAT_Total'].drop('PCIAT-PCIAT_Total')\nfiltered_corr = sii_corr[(sii_corr > 0.07) | (sii_corr < -0.07)]\n\nprint(filtered_corr)\n\nplt.figure(figsize=(8, 6))\nfiltered_corr.sort_values().plot(kind='barh', color='coral')\nplt.title('Features with Correlation > 0.1 or < -0.1 with PCIAT-PCIAT_Total')\nplt.xlabel('Correlation coefficient')\nplt.ylabel('Features')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.241687Z","iopub.status.idle":"2024-12-22T15:10:42.242047Z","shell.execute_reply":"2024-12-22T15:10:42.241895Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Models","metadata":{}},{"cell_type":"code","source":"score_models =[]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.242888Z","iopub.status.idle":"2024-12-22T15:10:42.243183Z","shell.execute_reply":"2024-12-22T15:10:42.243045Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model 1","metadata":{}},{"cell_type":"markdown","source":"### Input data","metadata":{}},{"cell_type":"code","source":"path = '../input/child-mind-institute-problematic-internet-use/'\ntrain = pd.read_csv(path + 'train.csv', index_col = 'id')\nprint(\"The train data has the shape: \",train.shape)\ntest = pd.read_csv(path + 'test.csv', index_col = 'id')\nprint(\"The test data has the shape: \",test.shape)\nprint(\"\")\nprint(\"Total number of missing training values: \", train.isna().sum().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.244142Z","iopub.status.idle":"2024-12-22T15:10:42.244471Z","shell.execute_reply":"2024-12-22T15:10:42.244325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = train['PCIAT-PCIAT_Total']\ny_sii = train['sii']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.245873Z","iopub.status.idle":"2024-12-22T15:10:42.246240Z","shell.execute_reply":"2024-12-22T15:10:42.246085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train = train[common_cols]\n# test = test[common_cols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.247221Z","iopub.status.idle":"2024-12-22T15:10:42.247507Z","shell.execute_reply":"2024-12-22T15:10:42.247387Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Data processing","metadata":{}},{"cell_type":"code","source":"#Preprocessing columns \"season\" in train, fill nan with value 0\ntrain_cat_columns = train.select_dtypes(exclude = 'number').columns\n\nfor season in train_cat_columns:\n    train[season] = train[season].fillna('Unknown')\n    train[season] = train[season].replace({'Spring': 1, 'Summer': 2, 'Fall': 3, 'Winter': 4, 'Unknown': 0})\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.248613Z","iopub.status.idle":"2024-12-22T15:10:42.248867Z","shell.execute_reply":"2024-12-22T15:10:42.248762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Preprocessing columns \"season\" in test, fill nan with value 0\ntest_cat_columns = test.select_dtypes(exclude = 'number').columns\n\nfor season in test_cat_columns:\n    test[season] = test[season].fillna('Unknown')\n    test[season] = test[season].replace({'Spring': 1, 'Summer': 2, 'Fall': 3, 'Winter': 4, 'Unknown': 0})\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.249838Z","iopub.status.idle":"2024-12-22T15:10:42.250308Z","shell.execute_reply":"2024-12-22T15:10:42.250019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PCIAT_cols = [val for val in train.columns[train.columns.str.contains('PCIAT')]]\nprint('Number of PCIAT features = ' , len(PCIAT_cols))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.251377Z","iopub.status.idle":"2024-12-22T15:10:42.251787Z","shell.execute_reply":"2024-12-22T15:10:42.251613Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PCIAT_cols.remove('PCIAT-PCIAT_Total') #Column này sử dụng để làm giá trị Ground truth cho train\ntrain = train.drop(columns = PCIAT_cols) #Test không có các column PCIAT nên drop để về đúng format.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.252515Z","iopub.status.idle":"2024-12-22T15:10:42.252865Z","shell.execute_reply":"2024-12-22T15:10:42.252744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.dropna(subset='sii')\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.253575Z","iopub.status.idle":"2024-12-22T15:10:42.253812Z","shell.execute_reply":"2024-12-22T15:10:42.253717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Đánh gái mức độ tương quan giữa các column với ground truth\ncorr = pd.DataFrame(train.corr()['PCIAT-PCIAT_Total'].sort_values(ascending = False))\ncorr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.254577Z","iopub.status.idle":"2024-12-22T15:10:42.254862Z","shell.execute_reply":"2024-12-22T15:10:42.254717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Chọn ra các feat có corr cao với PCIAT_Total và loại bớt các feat có độ tương đồng thấp hoặc tương đương với các feat đã chọn\nselection = corr[(corr['PCIAT-PCIAT_Total']>.07) | (corr['PCIAT-PCIAT_Total']<-.07)]\nselection = [val for val in selection.index]\nselection.remove('PCIAT-PCIAT_Total')\nselection.remove('sii')\nselection.remove('Physical-BMI')\nselection.remove('SDS-SDS_Total_Raw')\nselection","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.255777Z","iopub.status.idle":"2024-12-22T15:10:42.256063Z","shell.execute_reply":"2024-12-22T15:10:42.255960Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isna().sum().sort_values(ascending = False).head(46)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.256900Z","iopub.status.idle":"2024-12-22T15:10:42.257240Z","shell.execute_reply":"2024-12-22T15:10:42.257088Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Lọc missing > 50%","metadata":{}},{"cell_type":"code","source":"half_missing = [val for val in train.columns[train.isnull().sum()>len(train)/2]]\nhalf_missing","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.258255Z","iopub.status.idle":"2024-12-22T15:10:42.258507Z","shell.execute_reply":"2024-12-22T15:10:42.258396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selection = [i for i in selection if i not in half_missing]\nselection","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.259094Z","iopub.status.idle":"2024-12-22T15:10:42.259383Z","shell.execute_reply":"2024-12-22T15:10:42.259262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"describe = train[selection].describe().T\ndescribe[['min','max']].sort_index()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.260018Z","iopub.status.idle":"2024-12-22T15:10:42.260341Z","shell.execute_reply":"2024-12-22T15:10:42.260171Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Train Model\n","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.metrics import make_scorer, cohen_kappa_score\nimport xgboost as xgb\nfrom lightgbm import LGBMRegressor  \nfrom catboost import CatBoostRegressor \nfrom sklearn.ensemble import VotingRegressor  \nimport lightgbm as lgb\nfrom sklearn.model_selection import cross_val_score, StratifiedKFold\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.261263Z","iopub.status.idle":"2024-12-22T15:10:42.261528Z","shell.execute_reply":"2024-12-22T15:10:42.261422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train[selection]\ntest = test[selection]\ny = train['PCIAT-PCIAT_Total']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.262184Z","iopub.status.idle":"2024-12-22T15:10:42.262510Z","shell.execute_reply":"2024-12-22T15:10:42.262341Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def convert(scores):\n#Convert tu PCAIT_Total sang sii\n    scores = np.array(scores)*1.25\n    bins = np.zeros_like(scores)\n    bins[scores <= 30] = 0\n    bins[(scores > 30) & (scores < 50)] = 1\n    bins[(scores >= 50) & (scores < 80)] = 2\n    bins[scores >= 80] = 3\n    return bins","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.263200Z","iopub.status.idle":"2024-12-22T15:10:42.263497Z","shell.execute_reply":"2024-12-22T15:10:42.263393Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def quadratic_kappa(y_true, y_pred):\n    y_true_cat = convert(y_true)\n    y_pred_cat = convert(y_pred)\n    return cohen_kappa_score(y_true_cat, y_pred_cat, weights='quadratic')\n\nkappa_scorer = make_scorer(quadratic_kappa, greater_is_better=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.264141Z","iopub.status.idle":"2024-12-22T15:10:42.264475Z","shell.execute_reply":"2024-12-22T15:10:42.264319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# X_train, X_valid, y_train, y_valid = train_test_split(train, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.265417Z","iopub.status.idle":"2024-12-22T15:10:42.265716Z","shell.execute_reply":"2024-12-22T15:10:42.265586Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"XGB_Params ={\n    'max_depth': 3,\n    'n_estimators': 59,\n    'learning_rate': 0.075,\n    'subsample': 0.6,\n    'colsample_bytree': 0.91\n}\nlgb_Params = {\n    'learning_rate': 0.046,\n    'max_depth': -1,#9,\n    'num_leaves': 478,\n    'verbose':-1\n}\n\nCAT_Params = {\n    'iterations': 500,\n    'learning_rate': 0.1,\n    'depth': 6,\n}\n\nlgb_model = LGBMRegressor(**lgb_Params)\nxgb_model = xgb.XGBRegressor(**XGB_Params)\ncat_model = CatBoostRegressor(**CAT_Params, silent=True)\n\n# Define a voting regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lgb', lgb_model),\n    ('xgb', xgb_model),\n    ('cat', cat_model)\n],  weights=[4.5, 4.5, 5.2])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.266576Z","iopub.status.idle":"2024-12-22T15:10:42.266876Z","shell.execute_reply":"2024-12-22T15:10:42.266748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 42\nn_splits = 5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.267823Z","iopub.status.idle":"2024-12-22T15:10:42.268167Z","shell.execute_reply":"2024-12-22T15:10:42.268039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"skf =  StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\nscores = cross_val_score(voting_model, X, y, cv=skf, scoring=kappa_scorer)\nprint(\"QWK Scores:\", scores)\nprint(\"Mean QWK Score:\", np.mean(scores))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.269296Z","iopub.status.idle":"2024-12-22T15:10:42.269600Z","shell.execute_reply":"2024-12-22T15:10:42.269487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"score_models.append(np.mean(scores))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.270240Z","iopub.status.idle":"2024-12-22T15:10:42.270553Z","shell.execute_reply":"2024-12-22T15:10:42.270402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"voting_model.fit(X,y)\npreds = voting_model.predict(test)\npreds = convert(preds) # convert raw scores to sii categories if using regressor\npreds = pd.Series(preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.271410Z","iopub.status.idle":"2024-12-22T15:10:42.271713Z","shell.execute_reply":"2024-12-22T15:10:42.271601Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.DataFrame({\n        'id': test.index,  \n        'sii': preds\n})\n# submission.to_csv('submission.csv', index=False)\nsubmission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.272668Z","iopub.status.idle":"2024-12-22T15:10:42.272978Z","shell.execute_reply":"2024-12-22T15:10:42.272871Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model 2","metadata":{}},{"cell_type":"code","source":"import polars as pl\nfrom pathlib import Path\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport numpy as np\nimport pandas as pd\nfrom collections import defaultdict\nfrom catboost import CatBoostClassifier, Pool\nimport lightgbm as lgb\nfrom pathlib import Path\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport sklearn.neighbors, sklearn.metrics, sklearn.preprocessing\nfrom xgboost import XGBClassifier\nimport polars.selectors as cs\nfrom sklearn.metrics import cohen_kappa_score, ConfusionMatrixDisplay\nimport numpy as np\nfrom colorama import Fore, Style\nimport random\n\nimport polars as pl\nimport numpy as np\nfrom sklearn.preprocessing import StandardScaler\nfrom tqdm import tqdm\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport gc\nimport itertools\nimport pickle\nimport re\nimport time\nimport os\nimport logging \n\nN_THREADS = 3\nN_BAGS = 5\nN_FOLDS = 5\nN_SEEDS = 1\nSEED = 0\nGPU = 0\n\nos.environ['PYTHONHASHSEED'] = str(SEED)\nrandom.seed(SEED)\nnp.random.seed(SEED)\nos.environ['POLARS_MAX_THREADS'] = str(N_THREADS)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.273650Z","iopub.status.idle":"2024-12-22T15:10:42.273952Z","shell.execute_reply":"2024-12-22T15:10:42.273799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class MyLogger:\n    def init(self, logging_lbl: str):\n        self.logger = logging.getLogger(logging_lbl)\n        self.logger.setLevel(logging.ERROR)\n\n    def info(self, message):\n        pass\n\n    def warning(self, message):\n        pass\n\n    def error(self, message):\n        self.logger.error(message)\n\nl = MyLogger()\nl.init(logging_lbl = \"lightgbm_custom\")\nlgb.register_logger(l)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.274569Z","iopub.status.idle":"2024-12-22T15:10:42.274880Z","shell.execute_reply":"2024-12-22T15:10:42.274773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"season_dtype = pl.Enum(['Spring', 'Summer', 'Fall', 'Winter','Missing'])\ntarget_labels = ['None', 'Mild', 'Moderate', 'Severe']\n\ntrain = (\n    pl.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n    .with_columns(pl.col('^.*Season$').cast(season_dtype))\n    .drop('^PCIAT.*$','^PAQ_*$')\n)\n\ntest = (\n    pl.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n    .with_columns(pl.col('^.*Season$').cast(season_dtype))\n    .drop('^PCIAT.*$','^PAQ_*$')\n)\n\nmissing = train.select([(pl.col(c).is_null().mean()).alias(c) for c in train.columns])\n\nmissing = (\n    missing.transpose(include_header=True)  \n    .filter(pl.col(\"column_0\") < 0.5)       \n    .with_columns((pl.col(\"column_0\") * 100).round(4).alias('%')) \n    .sort('%', descending=True)         \n)\n\nprint(f\"{missing.height} columns have less than 50% missing.\")\n\ncols = [i for i in train.columns if i in missing['column'].to_numpy()]\ndftr = train[cols]\ndfte = test[cols[:-1]]\n\n# DON'T WANT TO USE SII.. yet anyways.\ndftr = dftr.filter(~pl.col('sii').is_null())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.275380Z","iopub.status.idle":"2024-12-22T15:10:42.275639Z","shell.execute_reply":"2024-12-22T15:10:42.275530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_eng(dftr):\n    cat_columns = dftr.select(~cs.numeric()).columns \n    dftr = dftr.with_columns([\n        pl.col(col).cast(pl.Utf8).fill_null(\"Missing\")  \n        for col in cat_columns\n    ]).with_columns(pl.col('^.*Season$').cast(season_dtype))\n\n    median_df = (\n        dftr.group_by(\"Basic_Demos-Age\")\n        .agg(\n            pl.col(\"Physical-Height\").median().alias(\"Feat_Height\"),\n            pl.col(\"Physical-Weight\").median().alias(\"Feat_Weight\"),\n        )\n    )\n\n    dftr = dftr.join(median_df, on=\"Basic_Demos-Age\", how=\"left\").with_columns(\n        pl.when(pl.col(\"Physical-Height\").is_null())\n        .then(pl.col(\"Feat_Height\"))\n        .otherwise(pl.col(\"Physical-Height\"))\n        .alias(\"Physical-Height\"),\n        \n        pl.when(pl.col(\"Physical-Weight\").is_null())\n        .then(pl.col(\"Feat_Weight\"))\n        .otherwise(pl.col(\"Physical-Weight\"))\n        .alias(\"Physical-Weight\"),\n    ).drop([\"Feat_Height\", \"Feat_Weight\"])\n\n    dftr = dftr.with_columns([\n        pl.when(pl.col('Basic_Demos-Age') < 10)\n        .then(0)\n        .when((pl.col('Basic_Demos-Age') >= 10) & (pl.col('Basic_Demos-Age') < 16))\n        .then(1)\n        .otherwise(2)\n        .alias('age_group'),\n    ])\n\n    return dftr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.276538Z","iopub.status.idle":"2024-12-22T15:10:42.276884Z","shell.execute_reply":"2024-12-22T15:10:42.276750Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\n\nunique_ids = dftr.to_pandas().drop_duplicates('id')[['id', 'sii']]\nIDS = sorted(list(unique_ids['id']))\n\nFOLDS = []\nstrat_kf = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=SEED)\n\nfor _ in range(N_BAGS):\n    ids_with_folds = []\n    for fold_idx, (train_idx, val_idx) in enumerate(strat_kf.split(unique_ids['id'], unique_ids['sii'])):\n        val_ids = unique_ids['id'].iloc[val_idx].tolist()\n        ids_with_folds.append(val_ids)\n    FOLDS.append(ids_with_folds)\n\ncat_columns = dftr.drop('id').select(~cs.numeric()).columns ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.277886Z","iopub.status.idle":"2024-12-22T15:10:42.278240Z","shell.execute_reply":"2024-12-22T15:10:42.278097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def quadratic_weighted_kappa(preds, data):\n    y_true = data.get_label()\n    y_pred = preds.clip(y_min, y_max).round()\n    qwk = cohen_kappa_score(y_true, y_pred, weights=\"quadratic\")\n    return 'QWK', qwk, True\n\n\ndef qwk_obj(preds, dtrain):\n    labels = dtrain.get_label()\n    preds = preds.clip(y_min, y_max)\n    f = 1/2 * np.sum((preds - labels)**2)\n    g = 1/2 * np.sum((preds - a)**2 + b)\n    df = preds - labels\n    dg = preds - a\n    grad = (df/g - f*dg/g**2)*len(labels)\n    hess = np.ones(len(labels))\n    return grad, hess","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.278995Z","iopub.status.idle":"2024-12-22T15:10:42.279335Z","shell.execute_reply":"2024-12-22T15:10:42.279181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_params = {\n    \"objective\": qwk_obj,\n    \"metric\": \"None\",\n    'lambda_l2': 1.5,\n    'feature_fraction': 0.5,\n    'max_bin': 120,\n    'num_leaves': 8,\n    \"verbosity\": -1,\n    'random_state': 1,\n    'extra_trees': True\n}\n\ninit_score = 2.0\n\nbst = ['PreInt_EduHx-computerinternet_hoursday',\n       'SDS-SDS_Total_T',\n       'SDS-SDS_Total_Raw',\n       'Basic_Demos-Sex',\n       'Basic_Demos-Age',\n       'Physical-Weight',\n       'FGC-FGC_CU',\n       'Physical-Height',\n       'FGC-FGC_PU',\n       'FGC-FGC_SRL_Zone']\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.280002Z","iopub.status.idle":"2024-12-22T15:10:42.280330Z","shell.execute_reply":"2024-12-22T15:10:42.280209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = dftr.get_column('sii')\nX = dftr.drop('id', 'sii').to_pandas()[bst]\n\ndfte = feature_eng(dfte)\nX_te = dfte.drop('id').to_pandas()[bst + ['age_group']]\n\na = y.mean()\nb = y.var(ddof=0)\n\ny_min = y.min()\ny_max = y.max()\n\nfeature_importance_lgb = np.zeros(X.shape[1]+1)\noof_raw_lgb = np.zeros(len(y), dtype=float)\noof = np.zeros(len(y), dtype=int)\ntr_preds_lgb = np.zeros(len(X), dtype=float)\nte_preds_lgb = np.zeros(len(X_te), dtype=float)\nn_splits = N_BAGS * N_FOLDS\n\nfor bag_idx, bag in enumerate(FOLDS):\n    for fold_idx in range(N_FOLDS):\n        fold_num = bag_idx * N_FOLDS + fold_idx\n    \n        valid_ids = bag[fold_idx]\n        train_ids = []\n        for i in range(N_FOLDS):\n            if i != fold_idx:\n                train_ids.extend(bag[i])\n\n        idx_tr = dftr.to_pandas().index[dftr.to_pandas()['id'].isin(train_ids)].to_numpy()\n        idx_va = dftr.to_pandas().index[dftr.to_pandas()['id'].isin(valid_ids)].to_numpy()\n\n        X_tr, X_va = X.iloc[idx_tr], X.iloc[idx_va]\n        y_tr, y_va = y[idx_tr], y[idx_va]\n\n        X_tr = feature_eng(pl.from_pandas(X_tr)).to_pandas()\n        X_va = feature_eng(pl.from_pandas(X_va)).to_pandas()\n\n        init_score_tr = np.full(len(y_tr), init_score)\n        init_score_va = np.full(len(y_va), init_score)\n        \n        lgb_train = lgb.Dataset(\n            X_tr,\n            label=y_tr.to_numpy(),\n            init_score=[init_score]*len(X_tr)\n        )\n        lgb_val = lgb.Dataset(\n            X_va,\n            label=y_va.to_numpy(),\n            init_score=[init_score]*len(X_va)\n        )\n\n        model_lgb = lgb.train(\n            lgb_params,\n            lgb_train,\n            valid_sets=[lgb_val],\n            num_boost_round=10000,\n            feval=quadratic_weighted_kappa,\n            callbacks=[\n                lgb.early_stopping(\n                    stopping_rounds=20,\n                    verbose=-1,\n                ),\n                lgb.log_evaluation(0)\n            ]\n        )\n\n        # -----------  val preds\n        y_pred_lgb = np.clip(model_lgb.predict(X_va) + init_score, 0, 3)\n        oof_raw_lgb[idx_va] = y_pred_lgb\n\n        # ----------- train preds\n        tr_pred = np.clip(model_lgb.predict(X_tr) + init_score, 0, 3)\n        tr_preds_lgb[idx_tr] += tr_pred\n        \n        # ----------- test preds\n        te_pred = np.clip(model_lgb.predict(X_te) + init_score, 0 ,3)\n        te_preds_lgb += te_pred\n        \n        feature_importance_lgb += model_lgb.feature_importance()\n\n        # ----------- per fold eval\n        trn_eval = np.round(cohen_kappa_score(y_tr, np.round(tr_pred).astype(int),\n                                      weights='quadratic'), 4)\n\n        val_eval = np.round(cohen_kappa_score(y_va, np.round(y_pred_lgb).astype(int),\n                                      weights='quadratic'), 4)\n\n        print(f\"\\n# Processing Fold {fold_num + 1}/{n_splits}...\")\n        print(f\"TRAIN: {trn_eval:.3}\")\n        print(f\"EVAL:  {val_eval:.3}\")\n\n\ntr_preds = tr_preds_lgb / n_splits\nte_preds = te_preds_lgb / n_splits\n\n#  ----- Calculate scores\noof_score = cohen_kappa_score(y, np.round(oof_raw_lgb).astype(int), weights='quadratic')\ntr_score = cohen_kappa_score(y, np.round(tr_preds).astype(int), weights='quadratic')\nprint()\nprint(f\"{Fore.YELLOW}{Style.BRIGHT}# TRAIN: tr_score={tr_score:.3f} {Style.RESET_ALL}\")\nprint(f\"{Fore.GREEN}{Style.BRIGHT}# OOF: oof_score={oof_score:.3f} {Style.RESET_ALL}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.281296Z","iopub.status.idle":"2024-12-22T15:10:42.281653Z","shell.execute_reply":"2024-12-22T15:10:42.281497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"score_models.append(oof_score)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.282389Z","iopub.status.idle":"2024-12-22T15:10:42.282761Z","shell.execute_reply":"2024-12-22T15:10:42.282561Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_test = test.to_pandas()\nsub_test['sii'] = np.round(te_preds).astype(int)\n\nSubmission2 = pd.DataFrame({\n        'id': sub_test['id'],\n        'sii': sub_test['sii']\n    })\nSubmission2 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.283593Z","iopub.status.idle":"2024-12-22T15:10:42.284041Z","shell.execute_reply":"2024-12-22T15:10:42.283766Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Evaluating Feature Importance","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nfeature_importance_lgb /= n_splits\n\nfeature_names = X.columns  \nfeature_importance_df = pd.DataFrame({\n    'Feature': feature_names,\n    'Importance': feature_importance_lgb[:-1]  \n})\nfeature_importance_df = feature_importance_df.sort_values(by='Importance', ascending=False)\nprint(feature_importance_df)\n\nplt.figure(figsize=(10, 8))\nplt.barh(feature_importance_df['Feature'], feature_importance_df['Importance'])\nplt.gca().invert_yaxis()  # Reverse order for better readability\nplt.xlabel('Importance')\nplt.title('Feature Importance')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.284732Z","iopub.status.idle":"2024-12-22T15:10:42.285086Z","shell.execute_reply":"2024-12-22T15:10:42.284971Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model 3","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nimport copy\nimport pickle\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\n\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\n\nimport plotly.express as px\n\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\nSEED = 42\nn_splits = 5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.285776Z","iopub.status.idle":"2024-12-22T15:10:42.286073Z","shell.execute_reply":"2024-12-22T15:10:42.285941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.287089Z","iopub.status.idle":"2024-12-22T15:10:42.287446Z","shell.execute_reply":"2024-12-22T15:10:42.287314Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.288256Z","iopub.status.idle":"2024-12-22T15:10:42.288561Z","shell.execute_reply":"2024-12-22T15:10:42.288453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef feature_engineering(df):\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1) \n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    df['BMI_PHR'] = df['Physical-BMI'] * df['Physical-HeartRate']\n    df['SDS_InternetHours'] = df['SDS-SDS_Total_T'] * df['PreInt_EduHx-computerinternet_hoursday']\n  \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.289545Z","iopub.status.idle":"2024-12-22T15:10:42.289865Z","shell.execute_reply":"2024-12-22T15:10:42.289758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.290632Z","iopub.status.idle":"2024-12-22T15:10:42.290946Z","shell.execute_reply":"2024-12-22T15:10:42.290802Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"        \n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.291644Z","iopub.status.idle":"2024-12-22T15:10:42.291930Z","shell.execute_reply":"2024-12-22T15:10:42.291808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.292784Z","iopub.status.idle":"2024-12-22T15:10:42.293067Z","shell.execute_reply":"2024-12-22T15:10:42.292962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.49, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    thresholds = KappaOPtimizer.x\n    print(\"threshold: \",thresholds)\n    print(\"off_von_rounded\",oof_non_rounded)\n    oof_tuned = threshold_Rounder(oof_non_rounded, thresholds)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n   \n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission, tKappa, model\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.293845Z","iopub.status.idle":"2024-12-22T15:10:42.294208Z","shell.execute_reply":"2024-12-22T15:10:42.294041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01  # Increased from 2.68e-06\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'cat_features': cat_c,\n    'verbose': 0,\n    'l2_leaf_reg': 10  # Increase this value\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n],weights=[4.9,5.2,4.9])\n#, weights=[4.0, 4.0, 5.0]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.295046Z","iopub.status.idle":"2024-12-22T15:10:42.295333Z","shell.execute_reply":"2024-12-22T15:10:42.295214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the ensemble model\nSubmission3, score, model = TrainML(voting_model, test)\n\nSubmission3","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.296285Z","iopub.status.idle":"2024-12-22T15:10:42.296558Z","shell.execute_reply":"2024-12-22T15:10:42.296452Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"score_models.append(score)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.297529Z","iopub.status.idle":"2024-12-22T15:10:42.297846Z","shell.execute_reply":"2024-12-22T15:10:42.297733Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model 4\n","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.298514Z","iopub.status.idle":"2024-12-22T15:10:42.298840Z","shell.execute_reply":"2024-12-22T15:10:42.298686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.299633Z","iopub.status.idle":"2024-12-22T15:10:42.299918Z","shell.execute_reply":"2024-12-22T15:10:42.299787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"time_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.300763Z","iopub.status.idle":"2024-12-22T15:10:42.301035Z","shell.execute_reply":"2024-12-22T15:10:42.300929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.49, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    thresholds = KappaOPtimizer.x\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, thresholds)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    # tpm = test_preds.mean(axis=1)\n    # tp_rounded = threshold_Rounder(tpm, thresholds)\n    fold_weights = [1.25, 1.0, 1.0, 1.0, 1.0]\n    tpm = test_preds.dot(fold_weights) / np.sum(fold_weights)\n    tpTuned = threshold_Rounder(tpm, thresholds)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n    return submission, tKappa, model\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.301695Z","iopub.status.idle":"2024-12-22T15:10:42.302010Z","shell.execute_reply":"2024-12-22T15:10:42.301869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"imputer = SimpleImputer(strategy='median')\n# imputer = KNNImputer(n_neighbors=5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.302809Z","iopub.status.idle":"2024-12-22T15:10:42.303067Z","shell.execute_reply":"2024-12-22T15:10:42.302963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\nensemble = VotingRegressor(estimators=[\n    ('lgb', Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(random_state=SEED))])),\n    ('xgb', Pipeline(steps=[('imputer', imputer), ('regressor', XGBRegressor(random_state=SEED))])),\n    ('cat', Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(random_state=SEED, silent=True))])),\n    ('rf', Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=SEED))])),\n    ('gb', Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=SEED))]))\n    \n])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.303890Z","iopub.status.idle":"2024-12-22T15:10:42.304391Z","shell.execute_reply":"2024-12-22T15:10:42.304252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission4, score_4, model_4 = TrainML(ensemble, test)\nSubmission4","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.305354Z","iopub.status.idle":"2024-12-22T15:10:42.305638Z","shell.execute_reply":"2024-12-22T15:10:42.305526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"score_models.append(score_4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.306253Z","iopub.status.idle":"2024-12-22T15:10:42.306586Z","shell.execute_reply":"2024-12-22T15:10:42.306413Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Final submission","metadata":{}},{"cell_type":"code","source":"sub1 = submission\nsub2 = Submission2\nsub3 = Submission3\nsub4 = Submission4\n\n\nsub1 = sub1.sort_values(by='id').reset_index(drop=True)\nsub2 = sub2.sort_values(by='id').reset_index(drop=True)\nsub3 = sub3.sort_values(by='id').reset_index(drop=True)\nsub4 = sub4.sort_values(by='id').reset_index(drop=True)\n\n\ncombined = pd.DataFrame({\n    'id': sub1['id'],\n    'sii_1': sub1['sii'],\n    'sii_2': sub2['sii'],\n    'sii_3': sub3['sii'],\n    'sii_4': sub4['sii'],\n    \n})\nprint(combined)\n\ndef majority_vote(row):\n    return row.mode()[0]\n\ncombined['final_sii'] = combined[['sii_1','sii_2','sii_3', 'sii_4' ]].apply(majority_vote, axis=1)\n\nfinal_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n\n\n\nprint(\"Majority voting completed and saved to 'Final_Submission.csv'\")\nfinal_submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.307451Z","iopub.status.idle":"2024-12-22T15:10:42.307750Z","shell.execute_reply":"2024-12-22T15:10:42.307633Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Evaluating results","metadata":{}},{"cell_type":"code","source":"score_models","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.308642Z","iopub.status.idle":"2024-12-22T15:10:42.308924Z","shell.execute_reply":"2024-12-22T15:10:42.308818Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Data for QWK, Public, Private scores of the models\nmodels = ['Model 1', 'Model 2', 'Model 3', 'Model 4']\n# Get data from history submission.\npublic_scores = [0.441, 0.455, 0.479, 0.462]\nprivate_scores = [0.430, 0.475, 0.409, 0.446]\n\n# Calculate voting (average of all models)\nmean_qwk = np.mean(score_models)\nmean_public_score = np.mean(public_scores)\nmean_private_score = np.mean(private_scores)\n\n# Create the chart\nx = np.arange(len(models))\n\nplt.figure(figsize=(12, 6))\n\n# Bar chart for Mean QWK\nplt.bar(x - 0.2, mean_qwk, width=0.2, label='Mean QWK', color='blue', alpha=0.7)\n# Bar chart for Public Score\nplt.bar(x, public_scores, width=0.2, label='Public Score', color='green', alpha=0.7)\n# Bar chart for Private Score\nplt.bar(x + 0.2, private_scores, width=0.2, label='Private Score', color='orange', alpha=0.7)\n\n# Voting average lines\nplt.axhline(y=mean_qwk, color='blue', linestyle='--', label=f'Mean QWK: {mean_qwk:.3f}')\nplt.axhline(y=mean_public_score, color='green', linestyle='--', label=f'Mean Public: {mean_public_score:.3f}')\nplt.axhline(y=mean_private_score, color='orange', linestyle='--', label=f'Mean Private: {mean_private_score:.3f}')\n\n# Add labels and styling\nplt.xticks(x, models)\nplt.xlabel('Models')\nplt.ylabel('Scores')\nplt.title('Comparison of Models')\nplt.legend()\nplt.grid(axis='y', linestyle='--', alpha=0.6)\nplt.tight_layout()\n\n# Display the chart\nplt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.309668Z","iopub.status.idle":"2024-12-22T15:10:42.309961Z","shell.execute_reply":"2024-12-22T15:10:42.309833Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n- **Model 2** has the highest **Private Score** (0.475), indicating it performs well and is stable with private data. This model stands out and may be the best choice if optimizing private scores is the goal.\n- **Model 3** has the lowest **Private Score** (0.409), showing it performs poorly on private data. While it has a high **public score** (0.479), the large discrepancy between public and private scores (0.070) suggests the model may not be stable.\n- **Model01** and **Model 4** have relatively consistent public and private scores, but neither outperforms **Model 2** in terms of private score.","metadata":{}},{"cell_type":"code","source":"\nmodels = [\"Model 1\", \"Model 2\", \"Model 3\", \"Model 4\", \"Voting Model\"]\n\npublic_scores = [0.441, 0.455, 0.479, 0.462, 0.474]\nprivate_scores = [0.430, 0.475, 0.409, 0.446, 0.442]\n\n# Calculate the difference between Public and Private scores\nscore_diff = [abs(public - private) for public, private in zip(public_scores, private_scores)]\n\n# Plotting the bar chart\nx = np.arange(len(models))\nwidth = 0.35\n\nfig, ax = plt.subplots(figsize=(10, 6))\n\n# Plot Public and Private scores\nbars1 = ax.bar(x - width/2, public_scores, width, label='Public Score', color='blue', alpha=0.7)\nbars2 = ax.bar(x + width/2, private_scores, width, label='Private Score', color='orange', alpha=0.7)\n\n# Annotate the score differences\nfor i, diff in enumerate(score_diff):\n    ax.text(i, max(public_scores[i], private_scores[i]) + 0.01, f\"Diff: {diff:.3f}\", ha='center', fontsize=9, color='red')\n\n# Add labels and title\nax.set_xlabel('Models')\nax.set_ylabel('Scores')\nax.set_title('Public and Private Scores of Models')\nax.set_xticks(x)\nax.set_xticklabels(models)\nax.legend()\n\nplt.ylim(0.4, 0.5)\nplt.grid(axis='y', linestyle='--', alpha=0.6)\nplt.tight_layout()\n\nplt.show()\n\n# Print evaluation results\nbest_model_idx = np.argmax(private_scores)\nprint(\"Evaluation Results\")\nprint(f\"Best Private Model: {models[best_model_idx]} with Private Score: {private_scores[best_model_idx]:.3f}\")\nprint(f\"Voting Model Public Score: {public_scores[-1]:.3f}\")\nprint(f\"Voting Model Private Score: {private_scores[-1]:.3f}\")\nprint(\"Score Differences (Public - Private):\")\nfor i, model in enumerate(models):\n    print(f\"{model}: {score_diff[i]:.3f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.310794Z","iopub.status.idle":"2024-12-22T15:10:42.311082Z","shell.execute_reply":"2024-12-22T15:10:42.310966Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- The **Voting Model** has a **Public score** of 0.474, close to the scores of the other models, but its **Private score** is 0.442, not significantly better. This shows that the voting model doesn't provide a substantial improvement over individual models, especially compared to **Model 2** (private score of 0.475).\n- The difference between **Public** and **Private** in the voting model is 0.032, indicating a relatively stable performance compared to other models, though still not exceptional.","metadata":{}},{"cell_type":"markdown","source":"\n- **Model 3** has the largest discrepancy (0.070), suggesting it is unstable and may be overfitting or struggling with private data.\n- The other models have smaller discrepancies, but **Model 2** remains the most stable with a difference of 0.025, and it also has the highest private score.","metadata":{}},{"cell_type":"markdown","source":"### Conclusion:\n- The **Voting Model** provides a balanced result between public and private scores but does not significantly outperform **Model 2**, which has the highest private score. Therefore, the voting model can be useful when a balance between public and private data is needed, but if the goal is to optimize private score, **Model 2** is still the better choice.\n","metadata":{}},{"cell_type":"code","source":"Submission2.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:10:42.311763Z","iopub.status.idle":"2024-12-22T15:10:42.312074Z","shell.execute_reply":"2024-12-22T15:10:42.311963Z"}},"outputs":[],"execution_count":null}]}