{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Inport thư viện này kia","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib.gridspec as gridspec\nimport seaborn as sns\nimport warnings\nfrom xgboost import XGBRegressor\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom xgboost import XGBClassifier\n\nimport os\nimport re\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\nimport plotly.subplots as sp\nimport plotly.express as px\nfrom concurrent.futures import ThreadPoolExecutor\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nfrom IPython.display import display\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nfrom sklearn.base import clone, BaseEstimator, RegressorMixin\nfrom sklearn.ensemble import RandomForestClassifier, RandomForestRegressor\nfrom sklearn.ensemble import StackingRegressor\nfrom sklearn.linear_model import Ridge\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, accuracy_score, precision_score, recall_score, f1_score, roc_curve, auc\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor, HistGradientBoostingRegressor, ExtraTreesRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.model_selection import GridSearchCV","metadata":{"execution":{"iopub.status.busy":"2025-06-10T14:55:27.384899Z","iopub.execute_input":"2025-06-10T14:55:27.385176Z","iopub.status.idle":"2025-06-10T14:55:27.397886Z","shell.execute_reply.started":"2025-06-10T14:55:27.385156Z","shell.execute_reply":"2025-06-10T14:55:27.396715Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load dataset ","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ndata_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:55:33.788852Z","iopub.execute_input":"2025-06-10T14:55:33.790057Z","iopub.status.idle":"2025-06-10T14:55:33.918903Z","shell.execute_reply.started":"2025-06-10T14:55:33.790004Z","shell.execute_reply":"2025-06-10T14:55:33.917986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:55:35.554850Z","iopub.execute_input":"2025-06-10T14:55:35.555943Z","iopub.status.idle":"2025-06-10T14:55:35.662686Z","shell.execute_reply.started":"2025-06-10T14:55:35.555909Z","shell.execute_reply":"2025-06-10T14:55:35.661246Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_count = (\n    train.isna().sum()\n    .reset_index()\n    .rename(columns={'index': 'feature', 0: 'null_count'})\n    .sort_values('null_count', ascending=False)\n)\nmissing_count['null_ratio'] = missing_count['null_count'] / len(train)\n\nplt.figure(figsize=(6, 15))\nplt.title('Missing values over the whole training dataset')\nplt.barh(np.arange(len(missing_count)), missing_count['null_ratio'], color='coral', label='missing')\nplt.barh(np.arange(len(missing_count)), \n         1 - missing_count['null_ratio'],\n         left=missing_count['null_ratio'],\n         color='darkseagreen', label='available')\nplt.yticks(np.arange(len(missing_count)), missing_count['feature'])\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0, 1)\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T15:05:46.539499Z","iopub.execute_input":"2025-06-10T15:05:46.539877Z","iopub.status.idle":"2025-06-10T15:05:47.978680Z","shell.execute_reply.started":"2025-06-10T15:05:46.539851Z","shell.execute_reply":"2025-06-10T15:05:47.977665Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Nhận xét ban đầu là dữ liệu bị thiểu rất nhiều ","metadata":{}},{"cell_type":"code","source":"train_cols = set(train.columns)\ntest_cols = set(test.columns)\ncolumns_not_in_test = sorted(list(train_cols - test_cols))\n\ncolumns_to_exclude = ['PCIAT-PCIAT_Total', 'PCIAT-Season', 'sii']\nquestion_columns = [\n    col for col in columns_not_in_test if col not in columns_to_exclude\n]\n\nquestion_columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T15:01:37.220217Z","iopub.execute_input":"2025-06-10T15:01:37.220580Z","iopub.status.idle":"2025-06-10T15:01:37.228749Z","shell.execute_reply.started":"2025-06-10T15:01:37.220556Z","shell.execute_reply":"2025-06-10T15:01:37.227647Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Giải thích \nSau khi dựa đoán 20 cột ở trên với miền giá trị của mỗi cột là [0 .. 5] \n\nTính tổng lại của các cột này ta sẽ có được PCIAT-PCIAT_Total nằm trong [0 .. 100] \n\nCuối cùng sẽ ánh xạ lên đáp án Severity Impairment Index (sii): 0-30=None; 31-49=Mild; 50-79=Moderate; 80-100=Severe","metadata":{}},{"cell_type":"markdown","source":"# Tính tổng Sii\n\nCần một hàm ước lượng các giá trị Sii từ những dữ liệu đang có ","metadata":{}},{"cell_type":"code","source":"def recalculate_sii(row):\n    if pd.isna(row['PCIAT-PCIAT_Total']): # trả về nan khi có cột bị thiếu\n        return np.nan \n    max_possible = row['PCIAT-PCIAT_Total'] + row[question_columns].isna().sum() * 5\n    if row['PCIAT-PCIAT_Total'] <= 30 and max_possible <= 30:\n        return 0\n    elif 31 <= row['PCIAT-PCIAT_Total'] <= 49 and max_possible <= 49:\n        return 1\n    elif 50 <= row['PCIAT-PCIAT_Total'] <= 79 and max_possible <= 79:\n        return 2\n    elif row['PCIAT-PCIAT_Total'] >= 80 and max_possible >= 80:\n        return 3\n    return np.nan\n\ntrain['recalc_sii'] = train.apply(recalculate_sii, axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T15:01:41.592394Z","iopub.execute_input":"2025-06-10T15:01:41.592747Z","iopub.status.idle":"2025-06-10T15:01:43.009167Z","shell.execute_reply.started":"2025-06-10T15:01:41.592724Z","shell.execute_reply":"2025-06-10T15:01:43.008222Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mismatch_rows = train[\n    (train['recalc_sii'] != train['sii']) & train['sii'].notna()\n]\n\nmismatch_rows[question_columns + ['recalc_sii'] + ['sii']].style.map(\n    lambda x: 'background-color: #FFC0CB' if pd.isna(x) else ''\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T15:01:47.854296Z","iopub.execute_input":"2025-06-10T15:01:47.855750Z","iopub.status.idle":"2025-06-10T15:01:47.893926Z","shell.execute_reply.started":"2025-06-10T15:01:47.855709Z","shell.execute_reply":"2025-06-10T15:01:47.892769Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Những người không thể đưa ra giá trị dự đoán nhưng lại có dữ liệu trong data gốc\n\nTa thấy ở đây có 17 người mà giá trị ước lượng recalc_sii khác với giá trị sii trong dữ liệu ban đầu\n\nSau khi xóa những người này thì ta có thể đồng bộ lại giữa recalc_sii và sii","metadata":{}},{"cell_type":"code","source":"train['sii'] = train['recalc_sii']\ntrain = train.drop(mismatch_rows.index)\n\ntrain[columns_not_in_test + ['recalc_sii']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:48.531347Z","iopub.execute_input":"2025-06-10T14:44:48.531753Z","iopub.status.idle":"2025-06-10T14:44:48.584390Z","shell.execute_reply.started":"2025-06-10T14:44:48.531716Z","shell.execute_reply":"2025-06-10T14:44:48.581726Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"na_total_rows = train[train['sii'].isna()]\nna_total_rows","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:48.586485Z","iopub.execute_input":"2025-06-10T14:44:48.587396Z","iopub.status.idle":"2025-06-10T14:44:48.669195Z","shell.execute_reply.started":"2025-06-10T14:44:48.587345Z","shell.execute_reply":"2025-06-10T14:44:48.668225Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## NaN Sii\nCó khoảng 1224 người mà ta không thể ước tính được giá trị của sii \n\nChưa có cách nào hiệu quả để có thể xử lý những người này \n\nGiải pháp là chấp nhận bỏ qua những người này","metadata":{}},{"cell_type":"code","source":"train = train.dropna(subset=['PCIAT-PCIAT_Total'])\ntrain","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:48.670301Z","iopub.execute_input":"2025-06-10T14:44:48.670570Z","iopub.status.idle":"2025-06-10T14:44:48.769748Z","shell.execute_reply.started":"2025-06-10T14:44:48.670549Z","shell.execute_reply":"2025-06-10T14:44:48.767677Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Xử lý các giá trị NaN trong các cột còn lại trong nhóm PCIAT","metadata":{}},{"cell_type":"code","source":"for column in question_columns:\n    if train[column].isna().any():\n        mode_value = train[column].mode()[0]\n        train[column] = train[column].fillna(mode_value)\n\ntrain[columns_not_in_test + ['recalc_sii']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:48.771064Z","iopub.execute_input":"2025-06-10T14:44:48.771338Z","iopub.status.idle":"2025-06-10T14:44:48.833287Z","shell.execute_reply.started":"2025-06-10T14:44:48.771317Z","shell.execute_reply":"2025-06-10T14:44:48.831893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.drop(columns='recalc_sii', inplace=True)\n\ntrain[columns_not_in_test]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:48.834574Z","iopub.execute_input":"2025-06-10T14:44:48.834989Z","iopub.status.idle":"2025-06-10T14:44:48.891494Z","shell.execute_reply.started":"2025-06-10T14:44:48.834958Z","shell.execute_reply":"2025-06-10T14:44:48.889752Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Các hàm hỗ trợ","metadata":{}},{"cell_type":"markdown","source":"## Hàm tính tỉ lệ và các thông tin khác về giá trị của một cột","metadata":{}},{"cell_type":"code","source":"def calculate_stats(data, columns):\n    if isinstance(columns, str):\n        columns = [columns]\n\n    stats = []\n    for col in columns:\n        if data[col].dtype in ['object', 'category']:\n            counts = data[col].value_counts(dropna=False, sort=False)\n            percents = data[col].value_counts(normalize=True, dropna=False, sort=False) * 100\n            formatted = counts.astype(str) + ' (' + percents.round(2).astype(str) + '%)'\n            stats_col = pd.DataFrame({'count (%)': formatted})\n            stats.append(stats_col)\n        else:\n            stats_col = data[col].describe().to_frame().transpose()\n            stats_col['missing'] = data[col].isnull().sum()\n            stats_col.index.name = col\n            stats.append(stats_col)\n\n    return pd.concat(stats, axis=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:48.892512Z","iopub.execute_input":"2025-06-10T14:44:48.892867Z","iopub.status.idle":"2025-06-10T14:44:48.902442Z","shell.execute_reply.started":"2025-06-10T14:44:48.892837Z","shell.execute_reply":"2025-06-10T14:44:48.901101Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data cleaning","metadata":{}},{"cell_type":"markdown","source":"## Thực hiện gom tuổi thành các nhóm tuổi \n* 0 : Children\n* 1 : Adolescents\n* 2 : Adults","metadata":{}},{"cell_type":"code","source":"calculate_stats(train, 'Basic_Demos-Age')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T15:19:35.189494Z","iopub.execute_input":"2025-06-10T15:19:35.190215Z","iopub.status.idle":"2025-06-10T15:19:35.209512Z","shell.execute_reply.started":"2025-06-10T15:19:35.190184Z","shell.execute_reply":"2025-06-10T15:19:35.208385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def apply_age_group(df):\n    df['Age_Group'] = pd.cut(\n        df['Basic_Demos-Age'],\n        bins=[4, 12, 18, 22],\n        labels=['0', '1', '2'],\n    )\n    return df\n\ntrain = apply_age_group(train)\ntest = apply_age_group(test)\n\ntrain['Age_Group']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T15:19:37.420780Z","iopub.execute_input":"2025-06-10T15:19:37.421105Z","iopub.status.idle":"2025-06-10T15:19:37.436761Z","shell.execute_reply.started":"2025-06-10T15:19:37.421081Z","shell.execute_reply":"2025-06-10T15:19:37.435887Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## CGAS score","metadata":{}},{"cell_type":"markdown","source":"## CGAS - Season \nDo không có cơ sở khi dự đoán các mùa nên giải pháp sẽ là điền ngẫu nhiên theo tỉ lệ trong data","metadata":{}},{"cell_type":"code","source":"calculate_stats(train, 'CGAS-Season')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:48.983699Z","iopub.execute_input":"2025-06-10T14:44:48.984047Z","iopub.status.idle":"2025-06-10T14:44:49.006704Z","shell.execute_reply.started":"2025-06-10T14:44:48.984021Z","shell.execute_reply":"2025-06-10T14:44:49.005299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"season_counts = train['CGAS-Season'].value_counts(normalize=True)\n\ntrain['CGAS-Season'] = train['CGAS-Season'].apply(\n    lambda x: np.random.choice(season_counts.index, p=season_counts.values) if pd.isna(x) else x\n)\ntest['CGAS-Season'] = test['CGAS-Season'].apply(\n    lambda x: np.random.choice(season_counts.index, p=season_counts.values) if pd.isna(x) else x\n)\ncalculate_stats(train, 'CGAS-Season')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:49.007676Z","iopub.execute_input":"2025-06-10T14:44:49.008013Z","iopub.status.idle":"2025-06-10T14:44:49.076587Z","shell.execute_reply.started":"2025-06-10T14:44:49.007984Z","shell.execute_reply":"2025-06-10T14:44:49.075680Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## CGAS-CGAS_Score : Children's Global Assessment Scale","metadata":{}},{"cell_type":"code","source":"calculate_stats(train, 'CGAS-CGAS_Score')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:49.077745Z","iopub.execute_input":"2025-06-10T14:44:49.078562Z","iopub.status.idle":"2025-06-10T14:44:49.110672Z","shell.execute_reply.started":"2025-06-10T14:44:49.078528Z","shell.execute_reply":"2025-06-10T14:44:49.109690Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.scatterplot(data=train, x='Basic_Demos-Age', y='CGAS-CGAS_Score', palette='viridis')\nplt.title(\"Relationship between Age and CGAS Score\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:49.112023Z","iopub.execute_input":"2025-06-10T14:44:49.113169Z","iopub.status.idle":"2025-06-10T14:44:49.531984Z","shell.execute_reply.started":"2025-06-10T14:44:49.113134Z","shell.execute_reply":"2025-06-10T14:44:49.530911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.boxplot(data=train, x='Age_Group', y='CGAS-CGAS_Score')\nplt.title(\"CGAS Score Distribution by Age Group\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:49.533282Z","iopub.execute_input":"2025-06-10T14:44:49.533568Z","iopub.status.idle":"2025-06-10T14:44:49.739469Z","shell.execute_reply.started":"2025-06-10T14:44:49.533547Z","shell.execute_reply":"2025-06-10T14:44:49.738153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"age_group_cgas_stats = train.groupby('Age_Group')['CGAS-CGAS_Score'].describe()\nage_group_cgas_stats","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:49.744883Z","iopub.execute_input":"2025-06-10T14:44:49.745476Z","iopub.status.idle":"2025-06-10T14:44:49.783268Z","shell.execute_reply.started":"2025-06-10T14:44:49.745452Z","shell.execute_reply":"2025-06-10T14:44:49.782364Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Nhận xét không thực sự có mối liên hệ nào rõ ràng ở đây => Không thể dự đoán thông tin về CGAS-CGAS_Score từ độ tuổi ","metadata":{}},{"cell_type":"code","source":"sns.boxplot(data=train, x='Basic_Demos-Sex', y='CGAS-CGAS_Score')\nplt.title(\"CGAS Score Distribution by Sex\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:49.784099Z","iopub.execute_input":"2025-06-10T14:44:49.784349Z","iopub.status.idle":"2025-06-10T14:44:49.974799Z","shell.execute_reply.started":"2025-06-10T14:44:49.784330Z","shell.execute_reply":"2025-06-10T14:44:49.973783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sex_cgas_stats = train.groupby('Basic_Demos-Sex')['CGAS-CGAS_Score'].describe()\nsex_cgas_stats","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:49.975542Z","iopub.execute_input":"2025-06-10T14:44:49.975890Z","iopub.status.idle":"2025-06-10T14:44:49.998416Z","shell.execute_reply.started":"2025-06-10T14:44:49.975862Z","shell.execute_reply":"2025-06-10T14:44:49.997268Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Nhận xét không thực sự có mối liên hệ nào rõ ràng ở đây => Không thể dự đoán thông tin về CGAS-CGAS_Score từ giới tính","metadata":{}},{"cell_type":"markdown","source":"## Mối quan hệ của PCIAT Total và CGAS Score","metadata":{}},{"cell_type":"code","source":"valid_data = train.dropna(subset=['CGAS-CGAS_Score', 'PCIAT-PCIAT_Total'])\n\nplt.figure(figsize=(10, 6))\nsns.scatterplot(\n    data=valid_data, \n    x='CGAS-CGAS_Score', \n    y='PCIAT-PCIAT_Total',\n    hue=valid_data['PCIAT-PCIAT_Total'] > 80,  # PCIAT severe group\n    alpha=0.3\n)\n\nplt.title(\"Scatter Plot of CGAS Score vs PCIAT Total\", fontsize=16)\nplt.xlabel(\"CGAS-CGAS_Score\", fontsize=12)\nplt.ylabel(\"PCIAT-PCIAT_Total (PIU Severity)\", fontsize=12)\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:49.999459Z","iopub.execute_input":"2025-06-10T14:44:49.999957Z","iopub.status.idle":"2025-06-10T14:44:50.387966Z","shell.execute_reply.started":"2025-06-10T14:44:49.999930Z","shell.execute_reply":"2025-06-10T14:44:50.386700Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Nhận xét \nCó thể thấy rằng những người có chỉ số CGAS Score cao (>80) thì sẽ không gặp vấn đề lớn với việc sử dụng Internet (Sii = 3)\n\ntừ nhận xét trên chúng ta tạo một hàm trọng số tập trung vào nhóm CGAS > 80 (vì nhóm này không có PIU nghiêm trọng). Sử dụng hàm sigmoid nghịch đảo để gán trọng số cao hơn cho các trường hợp CGAS > 80.","metadata":{}},{"cell_type":"code","source":"def sigmoid_weight_cgas_high(cgas, a=0.2, b=80):\n    return 1 / (1 + np.exp(-a * (cgas - b)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:50.389351Z","iopub.execute_input":"2025-06-10T14:44:50.389712Z","iopub.status.idle":"2025-06-10T14:44:50.396710Z","shell.execute_reply.started":"2025-06-10T14:44:50.389685Z","shell.execute_reply":"2025-06-10T14:44:50.395664Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Các chỉ số vật lý ","metadata":{}},{"cell_type":"code","source":"physical_columns = [\n 'Physical-BMI',\n 'Physical-Height',\n 'Physical-Weight',\n 'Physical-Waist_Circumference',\n 'Physical-Diastolic_BP',\n 'Physical-HeartRate',\n 'Physical-Systolic_BP'\n]\n\nwh_cols = [\n    'Physical-BMI', 'Physical-Height',\n    'Physical-Weight', 'Physical-Waist_Circumference'\n]\n\nheart_cols = [\n 'Physical-Diastolic_BP',\n 'Physical-HeartRate',\n 'Physical-Systolic_BP'\n]\n\ncalculate_stats(train, wh_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:50.398102Z","iopub.execute_input":"2025-06-10T14:44:50.398542Z","iopub.status.idle":"2025-06-10T14:44:50.443394Z","shell.execute_reply.started":"2025-06-10T14:44:50.398514Z","shell.execute_reply":"2025-06-10T14:44:50.441923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(train[wh_cols] == 0).sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:50.444431Z","iopub.execute_input":"2025-06-10T14:44:50.444802Z","iopub.status.idle":"2025-06-10T14:44:50.457461Z","shell.execute_reply.started":"2025-06-10T14:44:50.444772Z","shell.execute_reply":"2025-06-10T14:44:50.456455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[wh_cols] = train[wh_cols].replace(0, np.nan)\ntest[wh_cols] = test[wh_cols].replace(0, np.nan)\ncalculate_stats(train, wh_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:50.458344Z","iopub.execute_input":"2025-06-10T14:44:50.458591Z","iopub.status.idle":"2025-06-10T14:44:50.510801Z","shell.execute_reply.started":"2025-06-10T14:44:50.458572Z","shell.execute_reply":"2025-06-10T14:44:50.509216Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Phân tích về mối liên hệ giữa cân nặng - chiều cao và các mùa","metadata":{}},{"cell_type":"code","source":"encoded_season_train = pd.get_dummies(train, columns=['Basic_Demos-Enroll_Season'], prefix='Season', drop_first=False)\nencoded_season_test = pd.get_dummies(test, columns=['Basic_Demos-Enroll_Season'], prefix='Season', drop_first=False)\ntrain = train.join(encoded_season_train[['Season_Fall', 'Season_Spring', 'Season_Summer', 'Season_Winter']])\ntest = test.join(encoded_season_test[['Season_Fall', 'Season_Spring', 'Season_Summer', 'Season_Winter']])\ntrain.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:50.512932Z","iopub.execute_input":"2025-06-10T14:44:50.513307Z","iopub.status.idle":"2025-06-10T14:44:50.552904Z","shell.execute_reply.started":"2025-06-10T14:44:50.513278Z","shell.execute_reply":"2025-06-10T14:44:50.551868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.groupby('Basic_Demos-Enroll_Season')[['Physical-Weight', 'Physical-Height']].mean()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:50.553860Z","iopub.execute_input":"2025-06-10T14:44:50.554191Z","iopub.status.idle":"2025-06-10T14:44:50.572823Z","shell.execute_reply.started":"2025-06-10T14:44:50.554164Z","shell.execute_reply":"2025-06-10T14:44:50.571481Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.groupby(['Basic_Demos-Enroll_Season', 'Basic_Demos-Sex'])[['Physical-Weight', 'Physical-Height']].mean()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:50.574068Z","iopub.execute_input":"2025-06-10T14:44:50.574416Z","iopub.status.idle":"2025-06-10T14:44:50.596038Z","shell.execute_reply.started":"2025-06-10T14:44:50.574385Z","shell.execute_reply":"2025-06-10T14:44:50.594960Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Nhận xét\n\nCân nặng:\n\nTrong từng mùa, nữ giới thường có cân nặng thấp hơn hoặc tương đương nam giới.\nVí dụ:\n\nMùa thu: 87.21 pound (nữ) so với 90.43 pound (nam)\n\nMùa đông: 92.12 pound (nữ) ≈ 91.98 pound (nam)\n\nCân nặng có xu hướng cao hơn vào mùa hè và đông, thấp hơn vào mùa thu và xuân, nhưng chênh lệch giữa các mùa không đáng kể (1–5 kg).\n\nChiều cao:\n\nNam giới luôn cao hơn nữ giới ở tất cả các mùa, nhưng khác biệt rất nhỏ (0.1–1 cm).\nVí dụ:\n\nMùa xuân: 55.18 inch (nữ) so với 56.15 inch (nam)\n\nMùa đông: 56.16 inch (nữ) so với 56.39 inch (nam)\n\nSự thay đổi chiều cao theo mùa gần như không đáng kể, chênh lệch tối đa khoảng 1 cm giữa các mùa.\n\n=> Kết luận:\n\nCân nặng: Có sự khác biệt nhỏ theo mùa và giới tính, đặc biệt cao hơn vào mùa hè và đông. Tuy nhiên, các khác biệt này rất nhỏ và có thể bị ảnh hưởng bởi yếu tố khác như tuổi tác hoặc lối sống.\n\nChiều cao: Ảnh hưởng của mùa lên chiều cao gần như không đáng kể (chênh lệch trung bình < 1 cm).\n\nLưu ý: Do dữ liệu rất ít, chúng tôi vẫn sẽ sử dụng cả 3 cột mùa (season), tuổi (age) và giới tính (sex) để dự đoán giá trị thiếu của cân nặng (w) và chiều cao (h).","metadata":{}},{"cell_type":"markdown","source":"### Chuyển đổi sang cm và kg \nChuyển đổi dữ liệu từ inch và pound sang cm và kg","metadata":{}},{"cell_type":"code","source":"lbs_to_kg = 0.453592\ninches_to_cm = 2.54\n\ndef process_physical_BMI(df):\n    df['Physical-Weight'] = df['Physical-Weight'] * lbs_to_kg\n    df['Physical-Height'] = df['Physical-Height'] * inches_to_cm\n    df['Physical-Waist_Circumference'] = df['Physical-Waist_Circumference'] * inches_to_cm\n    \n    df['Physical-BMI'] = np.where(\n        df['Physical-Weight'].notna() & df['Physical-Height'].notna(),\n        df['Physical-Weight'] / ((df['Physical-Height'] / 100) ** 2),\n        np.nan\n    )\n    \n    return df\n\ntrain = process_physical_BMI(train)\ntest = process_physical_BMI(test)\n\ncalculate_stats(train, wh_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:50.596976Z","iopub.execute_input":"2025-06-10T14:44:50.597256Z","iopub.status.idle":"2025-06-10T14:44:50.635183Z","shell.execute_reply.started":"2025-06-10T14:44:50.597236Z","shell.execute_reply":"2025-06-10T14:44:50.633548Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Thực hiện điền các giá trị bị thiếu của cân nặng và chiều cao bằng phương pháp KNN \nselected_features = ['Basic_Demos-Age', 'Season_Fall', 'Season_Winter', 'Basic_Demos-Sex', 'Physical-Weight', 'Physical-Height']\n\nChọn 2 mùa có ảnh hưởng nhiều nhất để tham gia dự đoán \n","metadata":{}},{"cell_type":"code","source":"imputer = KNNImputer(n_neighbors=10)\n\nselected_features = ['Basic_Demos-Age', 'Season_Fall', 'Season_Winter', 'Basic_Demos-Sex', 'Physical-Weight', 'Physical-Height']\n\nimputed_data = imputer.fit_transform(train[selected_features])\ntrain_imputed = pd.DataFrame(imputed_data, columns=selected_features)\ntrain = train.drop(columns=selected_features).reset_index()\n\nimputed_test_data = imputer.transform(test[selected_features])\ntest_imputed = pd.DataFrame(imputed_test_data, columns=selected_features)\ntest = test.drop(columns=selected_features).reset_index()\n\ntrain = pd.concat([train, train_imputed], axis=1)\ntest = pd.concat([test, test_imputed], axis=1)\n\ncalculate_stats(train, ['Physical-Weight', 'Physical-Height'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:50.636804Z","iopub.execute_input":"2025-06-10T14:44:50.637200Z","iopub.status.idle":"2025-06-10T14:44:50.808175Z","shell.execute_reply.started":"2025-06-10T14:44:50.637129Z","shell.execute_reply":"2025-06-10T14:44:50.807195Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Tính lại chỉ số BMI dựa trên cân nặng và chiều cao ","metadata":{}},{"cell_type":"code","source":"train['Physical-BMI'] = train.apply(\n    lambda row: row['Physical-Weight'] / (row['Physical-Height'] / 100) ** 2 \n    if pd.isnull(row['Physical-BMI']) else row['Physical-BMI'], axis=1\n)\ntest['Physical-BMI'] = test.apply(\n    lambda row: row['Physical-Weight'] / (row['Physical-Height'] / 100) ** 2 \n    if pd.isnull(row['Physical-BMI']) else row['Physical-BMI'], axis=1\n)\ncalculate_stats(train, ['Physical-BMI'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:50.809276Z","iopub.execute_input":"2025-06-10T14:44:50.809627Z","iopub.status.idle":"2025-06-10T14:44:50.880542Z","shell.execute_reply.started":"2025-06-10T14:44:50.809580Z","shell.execute_reply":"2025-06-10T14:44:50.879540Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"So sánh với cột BIA-BIA_BMI ","metadata":{}},{"cell_type":"code","source":"bmi_ratio = train['Physical-BMI'] / train['BIA-BIA_BMI']\ncolor = (bmi_ratio < 0.8) | (bmi_ratio > 1.2)  # red if difference > 30%\n\nplt.scatter(\n    train['Physical-BMI'],\n    train['BIA-BIA_BMI'],\n    s=6,\n    c=color,\n    cmap='coolwarm'\n)\nplt.gca().set_aspect('equal')\nplt.xlabel('Physical-BMI')\nplt.ylabel('BIA-BIA_BMI')\nplt.title('Physical-BMI vs VIA-BIA_BMI')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:50.881482Z","iopub.execute_input":"2025-06-10T14:44:50.882398Z","iopub.status.idle":"2025-06-10T14:44:51.129351Z","shell.execute_reply.started":"2025-06-10T14:44:50.882364Z","shell.execute_reply":"2025-06-10T14:44:51.127663Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Ta thấy rằng những điểm màu đỏ có thể là những dữ liệu thiếu tin cậy ta tiến hành thay thế những điểm dữ liệu đó","metadata":{}},{"cell_type":"code","source":"train.loc[color, 'BIA-BIA_BMI'] = train.loc[color, 'Physical-BMI']\nplt.scatter(\n    train['Physical-BMI'],\n    train['BIA-BIA_BMI'],\n    s=6,\n    c=color,\n    cmap='coolwarm'\n)\nplt.gca().set_aspect('equal')\nplt.xlabel('Physical-BMI')\nplt.ylabel('BIA-BIA_BMI')\nplt.title('Physical-BMI vs VIA-BIA_BMI')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:51.130899Z","iopub.execute_input":"2025-06-10T14:44:51.131307Z","iopub.status.idle":"2025-06-10T14:44:51.409797Z","shell.execute_reply.started":"2025-06-10T14:44:51.131240Z","shell.execute_reply":"2025-06-10T14:44:51.407974Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Điền các giá trị bị thiếu của Physical-Waist_Circumference bằng LinearRegression","metadata":{}},{"cell_type":"code","source":"waist_data = train.dropna(subset=['Physical-Waist_Circumference'])\n\nX = waist_data[['Physical-BMI', 'Physical-Weight']]\ny = waist_data['Physical-Waist_Circumference']\n\nfrom sklearn.linear_model import LinearRegression\nwaist_model = LinearRegression()\nwaist_model.fit(X, y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:51.410698Z","iopub.execute_input":"2025-06-10T14:44:51.411153Z","iopub.status.idle":"2025-06-10T14:44:51.471836Z","shell.execute_reply.started":"2025-06-10T14:44:51.411116Z","shell.execute_reply":"2025-06-10T14:44:51.470335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_waist = train['Physical-Waist_Circumference'].isnull()\ntrain.loc[missing_waist, 'Physical-Waist_Circumference'] = waist_model.predict(\n    train.loc[missing_waist, ['Physical-BMI', 'Physical-Weight']]\n)\n\nmissing_waist_test = test['Physical-Waist_Circumference'].isnull()\ntest.loc[missing_waist_test, 'Physical-Waist_Circumference'] = waist_model.predict(\n    test.loc[missing_waist_test, ['Physical-BMI', 'Physical-Weight']]\n)\n\ncalculate_stats(train, wh_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:51.473213Z","iopub.execute_input":"2025-06-10T14:44:51.473573Z","iopub.status.idle":"2025-06-10T14:44:51.521007Z","shell.execute_reply.started":"2025-06-10T14:44:51.473545Z","shell.execute_reply":"2025-06-10T14:44:51.519977Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Huyết áp và nhịp tim ","metadata":{}},{"cell_type":"code","source":"bp_hr_cols = [\n    'Physical-Diastolic_BP', 'Physical-Systolic_BP',\n    'Physical-HeartRate'\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:51.521981Z","iopub.execute_input":"2025-06-10T14:44:51.522317Z","iopub.status.idle":"2025-06-10T14:44:51.528766Z","shell.execute_reply.started":"2025-06-10T14:44:51.522278Z","shell.execute_reply":"2025-06-10T14:44:51.527567Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(train[bp_hr_cols] < 50).sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:51.530774Z","iopub.execute_input":"2025-06-10T14:44:51.531188Z","iopub.status.idle":"2025-06-10T14:44:51.569081Z","shell.execute_reply.started":"2025-06-10T14:44:51.531156Z","shell.execute_reply":"2025-06-10T14:44:51.567698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[train['Physical-Systolic_BP'] <= train['Physical-Diastolic_BP']][bp_hr_cols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:51.570089Z","iopub.execute_input":"2025-06-10T14:44:51.570424Z","iopub.status.idle":"2025-06-10T14:44:51.599120Z","shell.execute_reply.started":"2025-06-10T14:44:51.570397Z","shell.execute_reply":"2025-06-10T14:44:51.598014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[bp_hr_cols] = train[bp_hr_cols].replace(0, np.nan)\ntrain.loc[train['Physical-Systolic_BP'] <= train['Physical-Diastolic_BP'], bp_hr_cols] = np.nan","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:51.600696Z","iopub.execute_input":"2025-06-10T14:44:51.601598Z","iopub.status.idle":"2025-06-10T14:44:51.640595Z","shell.execute_reply.started":"2025-06-10T14:44:51.601451Z","shell.execute_reply":"2025-06-10T14:44:51.639124Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Thang đo về giấc ngủ ","metadata":{}},{"cell_type":"code","source":"SDS_columns = ['SDS-Season', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T']\n\nSDS_number_columns = ['SDS-SDS_Total_Raw', 'SDS-SDS_Total_T']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:51.641967Z","iopub.execute_input":"2025-06-10T14:44:51.642332Z","iopub.status.idle":"2025-06-10T14:44:51.664213Z","shell.execute_reply.started":"2025-06-10T14:44:51.642303Z","shell.execute_reply":"2025-06-10T14:44:51.662953Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calculate_stats(train, 'SDS-Season')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:51.665700Z","iopub.execute_input":"2025-06-10T14:44:51.666452Z","iopub.status.idle":"2025-06-10T14:44:51.693872Z","shell.execute_reply.started":"2025-06-10T14:44:51.666419Z","shell.execute_reply":"2025-06-10T14:44:51.692683Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calculate_stats(train, SDS_number_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:51.695099Z","iopub.execute_input":"2025-06-10T14:44:51.695542Z","iopub.status.idle":"2025-06-10T14:44:51.744650Z","shell.execute_reply.started":"2025-06-10T14:44:51.695519Z","shell.execute_reply":"2025-06-10T14:44:51.742594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(18, 5))\n\n# SDS-SDS_Total_Raw\nplt.subplot(1, 3, 2)\nsns.histplot(train['SDS-SDS_Total_Raw'].dropna(), bins=20, kde=True)\nplt.title('SDS-SDS_Total_Raw')\nplt.xlabel('Value')\n\n# SDS-SDS_Total_T\nplt.subplot(1, 3, 3)\nsns.histplot(train['SDS-SDS_Total_T'].dropna(), bins=20, kde=True)\nplt.title('SDS-SDS_Total_T')\nplt.xlabel('Value')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:51.746385Z","iopub.execute_input":"2025-06-10T14:44:51.747387Z","iopub.status.idle":"2025-06-10T14:44:52.396879Z","shell.execute_reply.started":"2025-06-10T14:44:51.747344Z","shell.execute_reply":"2025-06-10T14:44:52.395925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.title('Sleep disturbance scale: conversion from raw to t score')\nplt.scatter(train['SDS-SDS_Total_Raw'],\n            train['SDS-SDS_Total_T'],\n            color='brown')\nplt.xlabel('SDS-SDS_Total_Raw')\nplt.ylabel('SDS-SDS_Total_T')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:52.397823Z","iopub.execute_input":"2025-06-10T14:44:52.398316Z","iopub.status.idle":"2025-06-10T14:44:52.624188Z","shell.execute_reply.started":"2025-06-10T14:44:52.398284Z","shell.execute_reply":"2025-06-10T14:44:52.623152Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Hai cột SDS-SDS_Total_Raw và SDS-SDS_Total_T có nhiều điểm tương đồng. \n\nCột SDS-SDS_Total_Raw sẽ mang nhiều thông tin hơn nên sẽ chỉ giữ lại cột này ","metadata":{}},{"cell_type":"code","source":"train = train.drop(columns=['SDS-SDS_Total_T'])\ntest = test.drop(columns=['SDS-SDS_Total_T'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:52.625279Z","iopub.execute_input":"2025-06-10T14:44:52.625643Z","iopub.status.idle":"2025-06-10T14:44:52.635104Z","shell.execute_reply.started":"2025-06-10T14:44:52.625587Z","shell.execute_reply":"2025-06-10T14:44:52.634011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cols=['Basic_Demos-Age', 'Age_Group', 'SDS-SDS_Total_Raw', 'PCIAT-PCIAT_Total']\ntrain[train['SDS-SDS_Total_Raw'] > 70][cols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:52.636561Z","iopub.execute_input":"2025-06-10T14:44:52.636890Z","iopub.status.idle":"2025-06-10T14:44:52.671776Z","shell.execute_reply.started":"2025-06-10T14:44:52.636868Z","shell.execute_reply":"2025-06-10T14:44:52.670762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[train['SDS-SDS_Total_Raw'] > 70].groupby('Age_Group').size()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:52.672748Z","iopub.execute_input":"2025-06-10T14:44:52.672985Z","iopub.status.idle":"2025-06-10T14:44:52.696027Z","shell.execute_reply.started":"2025-06-10T14:44:52.672966Z","shell.execute_reply":"2025-06-10T14:44:52.694973Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"valid_data = train.dropna(subset=['SDS-SDS_Total_Raw', 'PCIAT-PCIAT_Total'])\n\nplt.figure(figsize=(10, 6))\nsns.scatterplot(\n    data=valid_data, \n    x='SDS-SDS_Total_Raw', \n    y='PCIAT-PCIAT_Total',\n    hue=valid_data['PCIAT-PCIAT_Total'] > 80,  # PCIAT severe group\n    alpha=0.3\n)\n\nplt.title(\"Scatter Plot of SDS Total Raw vs PCIAT Total\", fontsize=16)\nplt.xlabel(\"SDS-SDS_Total_Raw (Sleep Disturbance Score)\", fontsize=12)\nplt.ylabel(\"PCIAT-PCIAT_Total (PIU Severity)\", fontsize=12)\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:52.697044Z","iopub.execute_input":"2025-06-10T14:44:52.697303Z","iopub.status.idle":"2025-06-10T14:44:53.103160Z","shell.execute_reply.started":"2025-06-10T14:44:52.697283Z","shell.execute_reply":"2025-06-10T14:44:53.101295Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Nhận xét \n\nKhông có gì thực sự rõ ràng \n\nNhững người không bị rối loạn giấc ngủ thì sẽ không bị ảnh hưởng nghiêm trọng bởi internet","metadata":{}},{"cell_type":"markdown","source":"## SDS-Season ","metadata":{}},{"cell_type":"code","source":"season_stats = train.groupby('SDS-Season')['SDS-SDS_Total_Raw'].agg(['mean', 'median'])\nseason_stats","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:53.104576Z","iopub.execute_input":"2025-06-10T14:44:53.105041Z","iopub.status.idle":"2025-06-10T14:44:53.124068Z","shell.execute_reply.started":"2025-06-10T14:44:53.105008Z","shell.execute_reply":"2025-06-10T14:44:53.122406Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Thực hiện điền các giá trị bị thiếu theo phân phối của dữ liệu ","metadata":{}},{"cell_type":"code","source":"season_counts = train['SDS-Season'].value_counts(normalize=True)\n\ntrain['SDS-Season'] = train['SDS-Season'].apply(\n    lambda x: np.random.choice(season_counts.index, p=season_counts.values) if pd.isna(x) else x\n)\ntest['SDS-Season'] = test['SDS-Season'].apply(\n    lambda x: np.random.choice(season_counts.index, p=season_counts.values) if pd.isna(x) else x\n)\n\ncalculate_stats(train, 'SDS-Season')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:53.125493Z","iopub.execute_input":"2025-06-10T14:44:53.125904Z","iopub.status.idle":"2025-06-10T14:44:53.165987Z","shell.execute_reply.started":"2025-06-10T14:44:53.125873Z","shell.execute_reply":"2025-06-10T14:44:53.164623Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Kiểm tra mối quan hệ của giới tính, độ tuổi và chỉ số SDS","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nsns.boxplot(x='Basic_Demos-Sex', y='SDS-SDS_Total_Raw', data=train)\nplt.title('Distribution of SDS by Sex')\nplt.xlabel('Sex (0 = Male, 1 = Female)')\nplt.ylabel('SDS-SDS_Total_Raw')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:53.167041Z","iopub.execute_input":"2025-06-10T14:44:53.167456Z","iopub.status.idle":"2025-06-10T14:44:53.403810Z","shell.execute_reply.started":"2025-06-10T14:44:53.167424Z","shell.execute_reply":"2025-06-10T14:44:53.402693Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nsns.boxplot(x='Age_Group', y='SDS-SDS_Total_Raw', data=train)\nplt.title('Boxplot of SDS by Age Group')\nplt.xlabel('Age Group')\nplt.ylabel('SDS-SDS_Total_Raw')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:53.405851Z","iopub.execute_input":"2025-06-10T14:44:53.406198Z","iopub.status.idle":"2025-06-10T14:44:53.612059Z","shell.execute_reply.started":"2025-06-10T14:44:53.406175Z","shell.execute_reply":"2025-06-10T14:44:53.611237Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":" Chúng ta có thể thấy nhóm tuổi thanh niên 1 có trung bình cao nhất \n\n Chúng ta sẽ xử dụng tuổi và mùa để dự đoán chỉ số SDS","metadata":{}},{"cell_type":"markdown","source":"## Thực hiện điền các giá trị bị thiếu của SDS-SDS_Total_Raw bằng phương pháp KNN","metadata":{}},{"cell_type":"code","source":"encoded_season_df = pd.get_dummies(train, columns=['SDS-Season'], prefix='SDS-Season', drop_first=False)\ntrain = train.join(encoded_season_df[['SDS-Season_Fall', 'SDS-Season_Spring', 'SDS-Season_Summer', 'SDS-Season_Winter']])\n\nencoded_season_df = pd.get_dummies(test, columns=['SDS-Season'], prefix='SDS-Season', drop_first=False)\ntest = test.join(encoded_season_df[['SDS-Season_Fall', 'SDS-Season_Spring', 'SDS-Season_Summer', 'SDS-Season_Winter']])\n\ntrain.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:53.613166Z","iopub.execute_input":"2025-06-10T14:44:53.613458Z","iopub.status.idle":"2025-06-10T14:44:53.642557Z","shell.execute_reply.started":"2025-06-10T14:44:53.613438Z","shell.execute_reply":"2025-06-10T14:44:53.641555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train_SDS_Score = train[['Basic_Demos-Age', 'SDS-Season_Fall', 'SDS-Season_Spring', 'SDS-Season_Winter']]\ny_train_SDS_Score = train['SDS-SDS_Total_Raw']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:53.643576Z","iopub.execute_input":"2025-06-10T14:44:53.643918Z","iopub.status.idle":"2025-06-10T14:44:53.650444Z","shell.execute_reply.started":"2025-06-10T14:44:53.643892Z","shell.execute_reply":"2025-06-10T14:44:53.649206Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sds_imputer = KNNImputer(n_neighbors=7)\n\ndata_with_sds = pd.concat([X_train_SDS_Score, y_train_SDS_Score], axis=1)\n\nfilled_data = sds_imputer.fit_transform(data_with_sds)\n\ncolumns = data_with_sds.columns.tolist()\nfilled_df = pd.DataFrame(filled_data, columns=columns)\n\ntrain['SDS-SDS_Total_Raw'] = filled_df['SDS-SDS_Total_Raw']\ncalculate_stats(train, ['SDS-SDS_Total_Raw'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:53.651511Z","iopub.execute_input":"2025-06-10T14:44:53.652182Z","iopub.status.idle":"2025-06-10T14:44:53.728679Z","shell.execute_reply.started":"2025-06-10T14:44:53.652150Z","shell.execute_reply":"2025-06-10T14:44:53.727158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test_SDS_Score = test[['Basic_Demos-Age', 'SDS-Season_Fall', 'SDS-Season_Spring', 'SDS-Season_Winter']]\ny_test_SDS_Score = test['SDS-SDS_Total_Raw']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:53.729642Z","iopub.execute_input":"2025-06-10T14:44:53.729925Z","iopub.status.idle":"2025-06-10T14:44:53.736175Z","shell.execute_reply.started":"2025-06-10T14:44:53.729904Z","shell.execute_reply":"2025-06-10T14:44:53.734718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_test_with_sds = pd.concat([X_test_SDS_Score, y_test_SDS_Score], axis=1)\nfilled_test_data = sds_imputer.transform(data_test_with_sds)\nfilled_test_df = pd.DataFrame(filled_test_data, columns=data_test_with_sds.columns)\ntest['SDS-SDS_Total_Raw'] = filled_test_df['SDS-SDS_Total_Raw']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:53.737959Z","iopub.execute_input":"2025-06-10T14:44:53.738362Z","iopub.status.idle":"2025-06-10T14:44:53.772316Z","shell.execute_reply.started":"2025-06-10T14:44:53.738330Z","shell.execute_reply":"2025-06-10T14:44:53.771043Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Thêm vào trọng số của SDS","metadata":{}},{"cell_type":"code","source":"def sigmoid_weight(sds, a=0.1, b=35):\n    return 1 / (1 + np.exp(a * (sds - b)))\n\ntrain['SDS_Weight'] = train['SDS-SDS_Total_Raw'].apply(sigmoid_weight)\ntest['SDS_Weight'] = test['SDS-SDS_Total_Raw'].apply(sigmoid_weight)\ntrain[['SDS_Weight', 'SDS-SDS_Total_Raw']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:53.773831Z","iopub.execute_input":"2025-06-10T14:44:53.774674Z","iopub.status.idle":"2025-06-10T14:44:53.806596Z","shell.execute_reply.started":"2025-06-10T14:44:53.774636Z","shell.execute_reply":"2025-06-10T14:44:53.805006Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Internet Use","metadata":{}},{"cell_type":"code","source":"calculate_stats(train, 'PreInt_EduHx-Season')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:53.807786Z","iopub.execute_input":"2025-06-10T14:44:53.808132Z","iopub.status.idle":"2025-06-10T14:44:53.833537Z","shell.execute_reply.started":"2025-06-10T14:44:53.808102Z","shell.execute_reply":"2025-06-10T14:44:53.832521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calculate_stats(train, ['PreInt_EduHx-computerinternet_hoursday'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:53.835160Z","iopub.execute_input":"2025-06-10T14:44:53.836230Z","iopub.status.idle":"2025-06-10T14:44:53.871000Z","shell.execute_reply.started":"2025-06-10T14:44:53.836194Z","shell.execute_reply":"2025-06-10T14:44:53.869361Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Mối quan hệ dữ việc sử dụng Internet và độ tuổi, giới tính","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(18, 5))\n\n# Hours of Internet Use by Age\nsns.boxplot(y=train['Basic_Demos-Age'], x=train['PreInt_EduHx-computerinternet_hoursday'], ax=axes[0], palette=\"Set3\")\naxes[0].set_title('Hours of Internet Use by Age')\naxes[0].set_ylabel('Age')\naxes[0].set_xlabel('Hours per Day Group')\n\n# Hours of Internet Use by Age Group\nsns.boxplot(y='PreInt_EduHx-computerinternet_hoursday', x='Age_Group', data=train, ax=axes[1], palette=\"Set3\")\naxes[1].set_title('Internet Hours by Age Group')\naxes[1].set_ylabel('Hours per Day (Numeric)')\naxes[1].set_xlabel('Age Group')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:53.872004Z","iopub.execute_input":"2025-06-10T14:44:53.872436Z","iopub.status.idle":"2025-06-10T14:44:54.322798Z","shell.execute_reply.started":"2025-06-10T14:44:53.872410Z","shell.execute_reply":"2025-06-10T14:44:54.321770Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stats = train.groupby(['Basic_Demos-Sex', 'PreInt_EduHx-computerinternet_hoursday']\n).size().unstack(fill_value=0)\nstats_prop = stats.div(stats.sum(axis=1), axis=0) * 100\n\nstats = stats.astype(str) +' (' + stats_prop.round(1).astype(str) + '%)'\nstats","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:54.323818Z","iopub.execute_input":"2025-06-10T14:44:54.324152Z","iopub.status.idle":"2025-06-10T14:44:54.341415Z","shell.execute_reply.started":"2025-06-10T14:44:54.324127Z","shell.execute_reply":"2025-06-10T14:44:54.340410Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Nhận xét \n\nNhóm tuổi càng lớn càng có thời gian xử dụng Internet nhiều hơn \n\nGiới tính không ảnh hưởng quá nhiều","metadata":{}},{"cell_type":"markdown","source":"### Mối quan hệ với chỉ số Sii","metadata":{}},{"cell_type":"code","source":"stats = train.groupby(\n    ['sii', 'PreInt_EduHx-computerinternet_hoursday']\n).size().unstack(fill_value=0)\nstats_prop = stats.div(stats.sum(axis=1), axis=0) * 100\n\nstats = stats.astype(str) +' (' + stats_prop.round(1).astype(str) + '%)'\nstats","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:54.349261Z","iopub.execute_input":"2025-06-10T14:44:54.349554Z","iopub.status.idle":"2025-06-10T14:44:54.370978Z","shell.execute_reply.started":"2025-06-10T14:44:54.349533Z","shell.execute_reply":"2025-06-10T14:44:54.369103Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nsns.boxplot(\n    x='PreInt_EduHx-computerinternet_hoursday', y='PCIAT-PCIAT_Total',\n    data=train,\n    hue='Age_Group', palette=\"Set3\"\n)\nplt.title('PCIAT_Total vs Hours of Internet Use by Age Group')\nplt.ylabel('PCIAT_Total')\nplt.xlabel('Hours per Day Group')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:54.372641Z","iopub.execute_input":"2025-06-10T14:44:54.373015Z","iopub.status.idle":"2025-06-10T14:44:54.834917Z","shell.execute_reply.started":"2025-06-10T14:44:54.372976Z","shell.execute_reply":"2025-06-10T14:44:54.833565Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Không thực sự có mối quan hệ rõ ràng ở đây. Vì có những người ít xử dụng internet nhưng vẫn có mức sii cao\n\nNgược lại có những người dùng internet nhiều hơn 3 giờ một ngày nhưng vẫn có sii thấp","metadata":{}},{"cell_type":"markdown","source":"## Phân tích nhóm người xử dụng nhiều internet","metadata":{}},{"cell_type":"code","source":"high_sii_high_internet = train[(train['sii'] >= 2) & (train['PreInt_EduHx-computerinternet_hoursday'] >= 2)]\nlow_sii_high_internet = train[(train['sii'] < 2) & (train['PreInt_EduHx-computerinternet_hoursday'] >= 2)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:54.836094Z","iopub.execute_input":"2025-06-10T14:44:54.836409Z","iopub.status.idle":"2025-06-10T14:44:54.849496Z","shell.execute_reply.started":"2025-06-10T14:44:54.836386Z","shell.execute_reply":"2025-06-10T14:44:54.846761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"high_sii_means = [\n    high_sii_high_internet['Basic_Demos-Age'].mean(),\n    high_sii_high_internet['Basic_Demos-Sex'].mean(),\n    high_sii_high_internet['Physical-Weight'].mean(),\n    high_sii_high_internet['Physical-Height'].mean(),\n    high_sii_high_internet['SDS-SDS_Total_Raw'].mean(),\n    high_sii_high_internet['CGAS-CGAS_Score'].mean(),\n    len(high_sii_high_internet),\n]\nlow_sii_means = [\n    low_sii_high_internet['Basic_Demos-Age'].mean(),\n    low_sii_high_internet['Basic_Demos-Sex'].mean(),\n    low_sii_high_internet['Physical-Weight'].mean(),\n    low_sii_high_internet['Physical-Height'].mean(),\n    low_sii_high_internet['SDS-SDS_Total_Raw'].mean(),\n    low_sii_high_internet['CGAS-CGAS_Score'].mean(),\n    len(low_sii_high_internet),\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:54.850911Z","iopub.execute_input":"2025-06-10T14:44:54.851209Z","iopub.status.idle":"2025-06-10T14:44:54.878738Z","shell.execute_reply.started":"2025-06-10T14:44:54.851186Z","shell.execute_reply":"2025-06-10T14:44:54.877499Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels = ['Age', 'Sex (0=Male)', 'Weight', 'Height', 'SDS', 'CGAS', 'Total Individuals']\n\nfig, axes = plt.subplots(1, 7, figsize=(20, 5), sharey=False)\n\nfor i, ax in enumerate(axes):\n    ax.bar(['High SII', 'Low SII'], [high_sii_means[i], low_sii_means[i]], color=['blue', 'orange'])\n    ax.set_title(labels[i])\n    ax.set_ylabel('Value' if i != 6 else 'Count')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:54.879936Z","iopub.execute_input":"2025-06-10T14:44:54.880280Z","iopub.status.idle":"2025-06-10T14:44:56.094078Z","shell.execute_reply.started":"2025-06-10T14:44:54.880251Z","shell.execute_reply":"2025-06-10T14:44:56.093105Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Nhận xét \n\nTa thấy rằng những người có độ tuổi lớn hơn sẽ có thời gian xử dụng Internet nhiều hơn \n\n","metadata":{}},{"cell_type":"markdown","source":"## Phân tích nhóm người xử dụng nhiều internet","metadata":{}},{"cell_type":"code","source":"high_sii_low_internet = train[(train['sii'] >= 2) & (train['PreInt_EduHx-computerinternet_hoursday'] < 2)]\nlow_sii_low_internet = train[(train['sii'] < 2) & (train['PreInt_EduHx-computerinternet_hoursday'] < 2)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:56.095237Z","iopub.execute_input":"2025-06-10T14:44:56.095575Z","iopub.status.idle":"2025-06-10T14:44:56.109127Z","shell.execute_reply.started":"2025-06-10T14:44:56.095547Z","shell.execute_reply":"2025-06-10T14:44:56.107424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"high_sii_means = [\n    high_sii_low_internet['Basic_Demos-Age'].mean(),\n    high_sii_low_internet['Basic_Demos-Sex'].mean(),\n    high_sii_low_internet['Physical-Weight'].mean(),\n    high_sii_low_internet['Physical-Height'].mean(),\n    high_sii_low_internet['SDS-SDS_Total_Raw'].mean(),\n    high_sii_low_internet['CGAS-CGAS_Score'].mean(),\n    len(high_sii_low_internet)\n]\nlow_sii_means = [\n    low_sii_low_internet['Basic_Demos-Age'].mean(),\n    low_sii_low_internet['Basic_Demos-Sex'].mean(),\n    low_sii_low_internet['Physical-Weight'].mean(),\n    low_sii_low_internet['Physical-Height'].mean(),\n    low_sii_low_internet['SDS-SDS_Total_Raw'].mean(),\n    low_sii_low_internet['CGAS-CGAS_Score'].mean(),\n    len(low_sii_low_internet)\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:56.110760Z","iopub.execute_input":"2025-06-10T14:44:56.111255Z","iopub.status.idle":"2025-06-10T14:44:56.130761Z","shell.execute_reply.started":"2025-06-10T14:44:56.111221Z","shell.execute_reply":"2025-06-10T14:44:56.129202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels = ['Age', 'Sex (1=Male)', 'Weight', 'Height', 'SDS_Total_Raw', 'CGAS_Score', 'Total Individuals']\n\ngroups = ['High SII', 'Low SII']\ndata = [high_sii_means, low_sii_means]\n\nfig, axes = plt.subplots(1, 7, figsize=(24, 6), sharey=False)\n\nfor i, ax in enumerate(axes):\n    ax.bar(groups, [data[0][i], data[1][i]], color=['blue', 'orange'])\n    ax.set_title(labels[i])\n    ax.set_ylabel('Value' if i != 6 else 'Count')\n\nplt.title('Low Internet')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:56.131944Z","iopub.execute_input":"2025-06-10T14:44:56.132315Z","iopub.status.idle":"2025-06-10T14:44:57.237810Z","shell.execute_reply.started":"2025-06-10T14:44:56.132286Z","shell.execute_reply":"2025-06-10T14:44:57.236433Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Nhận xét \n\nTrong các trường hợp thì số người gặp vấn đề với chỉ số sii có vẻ thấp hơn \n\n","metadata":{}},{"cell_type":"markdown","source":"Chúng ta có thể tạo những đặc trưng với các cặp chỉ số như giới tính + thời gian sử dụng internet. Hay độ tuổi + thời gian sử dụng internet","metadata":{}},{"cell_type":"markdown","source":"## Điền các giá trị bị thiếu của cột PreInt_EduHx-Season","metadata":{}},{"cell_type":"code","source":"season_counts = train['PreInt_EduHx-Season'].value_counts(normalize=True)\n\ntrain['PreInt_EduHx-Season'] = train['PreInt_EduHx-Season'].apply(lambda x: np.random.choice(season_counts.index, p=season_counts.values) if pd.isna(x) else x)\ntest['PreInt_EduHx-Season'] = test['PreInt_EduHx-Season'].apply(lambda x: np.random.choice(season_counts.index, p=season_counts.values) if pd.isna(x) else x)\n\ncalculate_stats(train, 'PreInt_EduHx-Season')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:57.238801Z","iopub.execute_input":"2025-06-10T14:44:57.239213Z","iopub.status.idle":"2025-06-10T14:44:57.262784Z","shell.execute_reply.started":"2025-06-10T14:44:57.239181Z","shell.execute_reply":"2025-06-10T14:44:57.261542Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Điền các giá trị bị thiếu của cột PreInt_EduHx-computerinternet_hoursday bằng logistic regression","metadata":{}},{"cell_type":"code","source":"features = ['Basic_Demos-Age', 'Basic_Demos-Sex', 'SDS-SDS_Total_Raw', 'PreInt_EduHx-Season']\ntarget = 'PreInt_EduHx-computerinternet_hoursday'\n\ntrain_data = train[train[target].notna()]\n\nX_train = train_data[features]\ny_train = train_data[target]\n\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('season', OneHotEncoder(), ['PreInt_EduHx-Season']),  # Mùa\n        ('num', 'passthrough', ['Basic_Demos-Age', 'Basic_Demos-Sex', 'SDS-SDS_Total_Raw'])  # Các cột số\n    ])\n\nmodel = Pipeline(steps=[\n    ('preprocessor', preprocessor),\n    ('classifier', LogisticRegression(max_iter=1000, multi_class='ovr'))\n])\n\nmodel.fit(X_train, y_train)\n\nX_missing = train[train[target].isna()][features]\npredicted_values = model.predict(X_missing)\n\ntrain.loc[train[target].isna(), target] = predicted_values\n\nX_missing_test = test[test[target].isna()][features]\npredicted_values_test = model.predict(X_missing_test)\n\ntest.loc[test[target].isna(), target] = predicted_values_test\n\ncalculate_stats(train, ['PreInt_EduHx-computerinternet_hoursday'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:57.263891Z","iopub.execute_input":"2025-06-10T14:44:57.264323Z","iopub.status.idle":"2025-06-10T14:44:58.504998Z","shell.execute_reply.started":"2025-06-10T14:44:57.264281Z","shell.execute_reply":"2025-06-10T14:44:58.503938Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Hoạt động thể chất ","metadata":{}},{"cell_type":"code","source":"PAQ_Adolescents_columns = ['PAQ_A-Season', 'PAQ_A-PAQ_A_Total']\ncalculate_stats(train, 'PAQ_A-Season')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:58.505683Z","iopub.execute_input":"2025-06-10T14:44:58.505936Z","iopub.status.idle":"2025-06-10T14:44:58.520215Z","shell.execute_reply.started":"2025-06-10T14:44:58.505916Z","shell.execute_reply":"2025-06-10T14:44:58.519380Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calculate_stats(train, 'PAQ_A-PAQ_A_Total')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:58.522008Z","iopub.execute_input":"2025-06-10T14:44:58.525001Z","iopub.status.idle":"2025-06-10T14:44:58.558359Z","shell.execute_reply.started":"2025-06-10T14:44:58.524959Z","shell.execute_reply":"2025-06-10T14:44:58.557531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PAQ_Children_columns = ['PAQ_C-Season', 'PAQ_C-PAQ_C_Total']\ncalculate_stats(train, 'PAQ_C-Season')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:58.559397Z","iopub.execute_input":"2025-06-10T14:44:58.559701Z","iopub.status.idle":"2025-06-10T14:44:58.575315Z","shell.execute_reply.started":"2025-06-10T14:44:58.559680Z","shell.execute_reply":"2025-06-10T14:44:58.573736Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calculate_stats(train, 'PAQ_C-PAQ_C_Total')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:58.577055Z","iopub.execute_input":"2025-06-10T14:44:58.577368Z","iopub.status.idle":"2025-06-10T14:44:58.613508Z","shell.execute_reply.started":"2025-06-10T14:44:58.577344Z","shell.execute_reply":"2025-06-10T14:44:58.612623Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Nhận xét: \nĐây là nhóm dữ liệu khá tệ vì thiết hụt quá nhiều ","metadata":{}},{"cell_type":"markdown","source":"## Phân tích trở kháng điện sinh học","metadata":{}},{"cell_type":"code","source":"bia_data_dict = data_dict[data_dict['Instrument'] == 'Bio-electric Impedance Analysis']\ncategorical_columns = bia_data_dict[bia_data_dict['Type'] == 'categorical int']['Field'].tolist()\ncontinuous_columns = bia_data_dict[bia_data_dict['Type'] == 'float']['Field'].tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:58.614793Z","iopub.execute_input":"2025-06-10T14:44:58.615658Z","iopub.status.idle":"2025-06-10T14:44:58.623341Z","shell.execute_reply.started":"2025-06-10T14:44:58.615623Z","shell.execute_reply":"2025-06-10T14:44:58.622296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(24, 20))\n\nfor idx, col in enumerate(continuous_columns):\n    plt.subplot(4, 4, idx + 1)\n    sns.histplot(train[col].dropna(), bins=20, kde=True)\n    plt.title(data_dict[data_dict['Field'] == col]['Description'].values[0])\n    plt.xlabel('Value')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:44:58.624413Z","iopub.execute_input":"2025-06-10T14:44:58.624764Z","iopub.status.idle":"2025-06-10T14:45:02.675790Z","shell.execute_reply.started":"2025-06-10T14:44:58.624737Z","shell.execute_reply":"2025-06-10T14:45:02.674717Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Nhận xét: \nDường như chỉ có chỉ số BMI có phân bố rõ ràng và đang tin cậy, chúng ta sẽ chỉ dùng dữ liệu này","metadata":{}},{"cell_type":"code","source":"for col in continuous_columns:\n    if (col != 'BIA-BIA_BMI'):\n        train = train.drop(columns=col)\n        test = test.drop(columns=col)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:45:02.676797Z","iopub.execute_input":"2025-06-10T14:45:02.677104Z","iopub.status.idle":"2025-06-10T14:45:02.712925Z","shell.execute_reply.started":"2025-06-10T14:45:02.677079Z","shell.execute_reply":"2025-06-10T14:45:02.711692Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Time series","metadata":{}},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n\n    stats, indexes = zip(*results)\n\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\n# train = train.drop('id', axis=1)\n# test = test.drop('id', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:45:02.713872Z","iopub.execute_input":"2025-06-10T14:45:02.714116Z","iopub.status.idle":"2025-06-10T14:46:43.035265Z","shell.execute_reply.started":"2025-06-10T14:45:02.714099Z","shell.execute_reply":"2025-06-10T14:46:43.034291Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#  Feature engineering","metadata":{}},{"cell_type":"markdown","source":"Encode unencoded season columns và xóa String columns","metadata":{}},{"cell_type":"code","source":"encoded_season_cols = ['Basic_Demos-Enroll_Season', 'SDS-Season']\nnon_encoded_season_cols = ['CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n          'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'PreInt_EduHx-Season']\n\ndef remove_encoded_cols(df):\n    return df.drop(columns=encoded_season_cols + ['Age_Group', 'index'])\n\ntrain = remove_encoded_cols(train)\ntest = remove_encoded_cols(test)\n\ndef fillna_season(df):\n    for c in non_encoded_season_cols: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = fillna_season(train)\ntest = fillna_season(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in non_encoded_season_cols:\n    mapping_train = create_mapping(col, train)\n    mapping_test = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping_train).astype(int)\n    test[col] = test[col].replace(mapping_test).astype(int)\n\n\ntrain = train.drop(columns=question_columns + ['PCIAT-Season', 'PCIAT-PCIAT_Total'])\n\nprint(f'Train Shape : {train.shape} || Test Shape : {test.shape}')\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:46:43.036642Z","iopub.execute_input":"2025-06-10T14:46:43.036912Z","iopub.status.idle":"2025-06-10T14:46:43.216739Z","shell.execute_reply.started":"2025-06-10T14:46:43.036891Z","shell.execute_reply":"2025-06-10T14:46:43.215242Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n    #df['id'] = df['id']\n    \n    #Weight\n    df['CGAS_Weight'] = df['CGAS-CGAS_Score'].apply(sigmoid_weight_cgas_high)\n    df['SDS_Score_Weighted'] = df['SDS-SDS_Total_Raw'] * df['SDS_Weight']\n    df['CGAS_Score_Weighted'] = df['CGAS-CGAS_Score'] * df['CGAS_Weight']\n    df = df.drop(columns=['SDS-SDS_Total_Raw', 'SDS_Weight', 'CGAS-CGAS_Score', 'CGAS_Weight'])\n    \n    #Age\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['Physical-Waist_Age'] = df['Basic_Demos-Age'] * df['Physical-Waist_Circumference']\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Physical-Height_Age'] = df['Basic_Demos-Age'] * df['Physical-Height']\n\n    #SDS\n    df['SDS_BMI'] = df['Physical-BMI'] * df['SDS_Score_Weighted']\n    df['CGAS_SDS'] = df['CGAS_Score_Weighted'] * df['SDS_Score_Weighted']\n    df['CGAS_Endurance_Mins'] = df['CGAS_Score_Weighted'] * df['Fitness_Endurance-Time_Mins']\n    df['SDS_Activity'] = df['BIA-BIA_Activity_Level_num'] * df['SDS_Score_Weighted']\n    df['SDS_InternetHours'] = df['SDS_Score_Weighted'] * df['PreInt_EduHx-computerinternet_hoursday']\n\n    df['BMI_Systolic_BP'] = df['Physical-BMI'] * df['Physical-Systolic_BP']\n    df['Age_Systolic_BP'] = df['Basic_Demos-Age'] * df['Physical-Systolic_BP']\n    df['PreInt_Systolic_BP'] = df['Physical-Systolic_BP'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['PAQ_A_Activity'] = df['BIA-BIA_Activity_Level_num'] * df['PAQ_A-PAQ_A_Total']\n    df['Activity_CU_PU'] = df['BIA-BIA_Activity_Level_num'] * df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n\n    #FGC\n    df['FGC_CU_PU'] = df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n    df['FGC_CU_PU_Age'] = df['FGC-FGC_CU'] * df['FGC-FGC_PU'] * df['Basic_Demos-Age']\n    df['FGC_GSND_GSD'] = df['FGC-FGC_GSND'] * df['FGC-FGC_GSD']\n    df['FGC_GSND_GSD_Age'] = df['FGC-FGC_GSND'] * df['FGC-FGC_GSD'] * df['Basic_Demos-Age']\n    df['CGAS_CU_PU'] = df['CGAS_Score_Weighted'] * df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n    df['PreInt_FGC_CU_PU'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n    df['Endurance_CU_PU'] = df['Fitness_Endurance-Time_Mins'] * df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n    return df\n\ntrain = feature_engineering(train)\ntest = feature_engineering(test)\n\nnew_features = ['Internet_Hours_Age', 'Physical-Waist_Age', 'BMI_Age', 'Physical-Height_Age', 'SDS_InternetHours', 'SDS_BMI', 'CGAS_SDS', 'CGAS_Endurance_Mins', 'SDS_Activity', 'BMI_Systolic_BP', 'Age_Systolic_BP', 'PreInt_Systolic_BP', 'PAQ_A_Activity', 'Activity_CU_PU', 'FGC_CU_PU', 'FGC_CU_PU_Age', 'FGC_GSND_GSD', 'FGC_GSND_GSD_Age', 'CGAS_CU_PU', 'PreInt_FGC_CU_PU', 'Endurance_CU_PU', 'CGAS_Weight', 'SDS_Score_Weighted', 'CGAS_Score_Weighted']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:46:43.217942Z","iopub.execute_input":"2025-06-10T14:46:43.218345Z","iopub.status.idle":"2025-06-10T14:46:43.265148Z","shell.execute_reply.started":"2025-06-10T14:46:43.218311Z","shell.execute_reply":"2025-06-10T14:46:43.264201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"simple_features = ['SDS_Score_Weighted', 'Internet_Hours_Age', 'Physical-Height_Age', \n                  'Basic_Demos-Sex', 'Physical-Waist_Age', 'PreInt_FGC_CU_PU']\n\ntrain = train[simple_features + ['sii']]\ntest = test[simple_features]\n\ntest","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:46:43.266048Z","iopub.execute_input":"2025-06-10T14:46:43.266295Z","iopub.status.idle":"2025-06-10T14:46:43.288410Z","shell.execute_reply.started":"2025-06-10T14:46:43.266276Z","shell.execute_reply":"2025-06-10T14:46:43.287363Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Một số hàm hỗ trợ khác \n* quadratic_weigthed_kappa: Hàm tính điểm có trọng số. Đây là hàm được dùng tính điểm trong yêu cầu đề bài\n* threshold_Rounder: Hàm làm tròn dựa trên thresholds đã được customize. Thay vì làm tròn ở 0.5, 1.5 và 2.5, chúng ta có thể tự tính ngưỡng làm tròn sao cho mô hình hoạt động tốt nhất.\n* evaluate_predictions: Hàm này đánh giá điểm của mô hình dựa trên thresholds, sử dụng hàm threshold_Rounder để làm tròn giá trị dự đoán, sau đó tính điểm Kappa. Dùng để tối ưu hóa trong quá trình cải thiện mô hình.\n* trainML: Hàm huấn luyện mô hình với StratifiedKFold. Fit từng fold, tính ngưỡng làm tròn tối ưu cho sau khi huấn luyện qua tất cả các fold, và dự đoán trên tập test, trả về kết quả dự đoán và ô hình đã fit","metadata":{}},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, train_data, test_data):\n    \n    X = train_data.drop(['sii'], axis=1)\n    y = train_data['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        #model = model_ok\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead') # Nelder-Mead | # Powell\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission,model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:46:43.289928Z","iopub.execute_input":"2025-06-10T14:46:43.290366Z","iopub.status.idle":"2025-06-10T14:46:43.321559Z","shell.execute_reply.started":"2025-06-10T14:46:43.290331Z","shell.execute_reply":"2025-06-10T14:46:43.320401Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Huấn luyện mô hình và đưa ra kết quả","metadata":{}},{"cell_type":"code","source":"SEED = 42\nn_splits = 20\n\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:46:43.322495Z","iopub.execute_input":"2025-06-10T14:46:43.322953Z","iopub.status.idle":"2025-06-10T14:46:43.355724Z","shell.execute_reply.started":"2025-06-10T14:46:43.322918Z","shell.execute_reply":"2025-06-10T14:46:43.354399Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LGB_Params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,\n    'lambda_l2': 0.01\n}\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 400,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10,\n    'task_type': 'GPU'\n}\n\nXGB_Params = {\n    'learning_rate': 0.01,\n    'max_depth': 5,\n    'n_estimators': 500,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 5,  \n    'reg_lambda': 10,  \n    'random_state': SEED,\n    'tree_method': 'gpu_hist',\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:48:40.334905Z","iopub.execute_input":"2025-06-10T14:48:40.336954Z","iopub.status.idle":"2025-06-10T14:48:40.346567Z","shell.execute_reply.started":"2025-06-10T14:48:40.336913Z","shell.execute_reply":"2025-06-10T14:48:40.345580Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import Ridge\nfrom sklearn.svm import SVR\nfrom lightgbm import LGBMRegressor\n\nLGB_Model = LGBMRegressor(**LGB_Params, random_state=SEED, verbose=-1, n_estimators=300)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\nXGB_Model = XGBRegressor(**XGB_Params)\n\nestimators = [\n    ('catboost', CatBoost_Model),\n    ('lightgbm', LGB_Model),\n    ('xgboost', XGB_Model),\n]\n\nensemble_model = StackingRegressor(\n    estimators=estimators,\n)\n\nfinal_submission,new_ensemble_model = TrainML(LGB_Model, train, test)\n\nfinal_submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:49:02.744239Z","iopub.execute_input":"2025-06-10T14:49:02.744576Z","iopub.status.idle":"2025-06-10T14:49:08.016926Z","shell.execute_reply.started":"2025-06-10T14:49:02.744553Z","shell.execute_reply":"2025-06-10T14:49:08.015624Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_submission[['id', 'sii']].to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T14:49:12.004246Z","iopub.execute_input":"2025-06-10T14:49:12.004626Z","iopub.status.idle":"2025-06-10T14:49:12.019814Z","shell.execute_reply.started":"2025-06-10T14:49:12.004582Z","shell.execute_reply":"2025-06-10T14:49:12.018483Z"}},"outputs":[],"execution_count":null}]}