{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":31011,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nfrom concurrent.futures import ThreadPoolExecutor\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split, StratifiedKFold, cross_val_score\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.compose import ColumnTransformer\nimport warnings\n\nwarnings.filterwarnings('ignore')\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:37:36.917159Z","iopub.execute_input":"2025-06-06T11:37:36.917447Z","iopub.status.idle":"2025-06-06T11:37:36.922129Z","shell.execute_reply.started":"2025-06-06T11:37:36.917411Z","shell.execute_reply":"2025-06-06T11:37:36.921464Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"seed = 2023","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:37:36.923263Z","iopub.execute_input":"2025-06-06T11:37:36.923612Z","iopub.status.idle":"2025-06-06T11:37:36.940110Z","shell.execute_reply.started":"2025-06-06T11:37:36.923590Z","shell.execute_reply":"2025-06-06T11:37:36.939407Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"markdown","source":"### Load timeseries","metadata":{}},{"cell_type":"code","source":"def process_file(filename, dirname):\n    \"\"\"Process a single parquet file and extract time-based features\"\"\"\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    \n    # Drop 'step' column if it exists\n    if 'step' in df.columns:\n        df.drop('step', axis=1, inplace=True)\n    \n    # Convert time_of_day to hours\n    df[\"hours\"] = df[\"time_of_day\"] // (3_600 * 1_000_000_000)\n    \n    # Define time periods\n    night = ((df[\"hours\"] >= 22) | (df[\"hours\"] <= 5))\n    day = ((df[\"hours\"] <= 20) & (df[\"hours\"] >= 7))\n    \n    # Initialize features dictionary\n    features = {}\n    \n    # Basic activity features\n    features['non_wear_mean'] = df[\"non-wear_flag\"].mean()\n    features['active_enmo_sum'] = df[\"enmo\"][df[\"enmo\"] >= 0.05].sum()\n    \n    # Process each column for different time periods\n    for col in ['enmo', 'anglez', 'light', 'battery_voltage']:\n        # Full day statistics\n        features[f\"{col}_mean\"] = df[col].mean()\n        features[f\"{col}_std\"] = df[col].std()\n        features[f\"{col}_max\"] = df[col].max()\n        features[f\"{col}_min\"] = df[col].min()\n        features[f\"{col}_diff_mean\"] = df[col].diff().mean()\n        features[f\"{col}_diff_std\"] = df[col].diff().std()\n        \n        # Night time statistics\n        night_data = df.loc[night, col]\n        features[f\"{col}_night_mean\"] = night_data.mean()\n        features[f\"{col}_night_std\"] = night_data.std()\n        features[f\"{col}_night_max\"] = night_data.max()\n        features[f\"{col}_night_min\"] = night_data.min()\n        \n        # Day time statistics\n        day_data = df.loc[day, col]\n        features[f\"{col}_day_mean\"] = day_data.mean()\n        features[f\"{col}_day_std\"] = day_data.std()\n        features[f\"{col}_day_max\"] = day_data.max()\n        features[f\"{col}_day_min\"] = day_data.min()\n    \n    return features, filename.split('=')[1]\n\ndef load_data_parquet(dirname) -> pd.DataFrame:\n    \"\"\"Load and process time series data from directory in parallel\"\"\"\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    features_list, indexes = zip(*results)\n    \n    # Create DataFrame with extracted features and IDs\n    df = pd.DataFrame(features_list)\n    df['id'] = indexes\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:37:36.940889Z","iopub.execute_input":"2025-06-06T11:37:36.941133Z","iopub.status.idle":"2025-06-06T11:37:36.954044Z","shell.execute_reply.started":"2025-06-06T11:37:36.941111Z","shell.execute_reply":"2025-06-06T11:37:36.953475Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Load datasets","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n# chia thành 3 nhóm features chính (Bộ dữ liệu khách quan csv)\ndemographicFeatures = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex']\n\nseasonFeatures = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season', 'PCIAT-Season']\n\ntrain_ts = load_data_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet')\ntest_ts = load_data_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet')\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:37:36.955853Z","iopub.execute_input":"2025-06-06T11:37:36.956026Z","iopub.status.idle":"2025-06-06T11:38:20.130090Z","shell.execute_reply.started":"2025-06-06T11:37:36.956013Z","shell.execute_reply":"2025-06-06T11:38:20.129291Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Overview datasets","metadata":{}},{"cell_type":"code","source":"print(train.shape)\nprint(test.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:20.131515Z","iopub.execute_input":"2025-06-06T11:38:20.131749Z","iopub.status.idle":"2025-06-06T11:38:20.135797Z","shell.execute_reply.started":"2025-06-06T11:38:20.131731Z","shell.execute_reply":"2025-06-06T11:38:20.135187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns_not_in_test = sorted(list(set(train.columns) - set(test.columns)))\n\ncolumns_to_exclude = ['PCIAT-PCIAT_Total', 'PCIAT-Season', 'sii']\nquestion_columns = [\n    col for col in columns_not_in_test if col not in columns_to_exclude\n]\n\nquestion_columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:20.136508Z","iopub.execute_input":"2025-06-06T11:38:20.136786Z","iopub.status.idle":"2025-06-06T11:38:20.150843Z","shell.execute_reply.started":"2025-06-06T11:38:20.136744Z","shell.execute_reply":"2025-06-06T11:38:20.150312Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Convert Season to numeric","metadata":{}},{"cell_type":"code","source":"def convert_season_to_numeric(df, season_columns):\n    # Định nghĩa mapping thứ tự cho các mùa\n    season_mapping = {\n        'Spring': 0,\n        'Summer': 1,\n        'Fall': 2,\n        'Winter': 3\n    }\n    \n    # Kiểm tra từng cột trong danh sách\n    for col in season_columns:\n        if col in df.columns:\n            #In ra các giá trị trước khi ánh xạ\n            # print(f\"Giá trị trước khi ánh xạ trong cột {col}:\")\n            # print(df[col].unique())\n            \n            # Áp dụng mapping\n            df[col] = df[col].map(season_mapping)\n            \n            #In ra các giá trị sau khi ánh xạ\n            # print(f\"Giá trị sau khi ánh xạ trong cột {col}:\")\n            # print(df[col].unique())\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:20.151724Z","iopub.execute_input":"2025-06-06T11:38:20.152302Z","iopub.status.idle":"2025-06-06T11:38:20.163112Z","shell.execute_reply.started":"2025-06-06T11:38:20.152276Z","shell.execute_reply":"2025-06-06T11:38:20.162477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"season_columns = [\n    'Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n    'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n    'PAQ_A-Season', 'PAQ_C-Season',  'SDS-Season', \n    'PreInt_EduHx-Season'\n]\n\n# Áp dụng hàm cho tập train và test\ntrain = convert_season_to_numeric(train, season_columns)\ntest = convert_season_to_numeric(test, season_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:20.163751Z","iopub.execute_input":"2025-06-06T11:38:20.164002Z","iopub.status.idle":"2025-06-06T11:38:20.192087Z","shell.execute_reply.started":"2025-06-06T11:38:20.163984Z","shell.execute_reply":"2025-06-06T11:38:20.191600Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\ncols = [\n    'Basic_Demos-Age', 'Physical-Weight', 'Physical-Height', 'Physical-BMI',\n    'Physical-Waist_Circumference', 'Physical-Diastolic_BP', 'Physical-Systolic_BP'\n]\ndata_subset = train[cols]\n\ncorr_matrix = data_subset.corr()\n\nplt.figure(figsize=(10, 8))\nsns.heatmap(corr_matrix, annot=True, cmap='coolwarm', fmt='.2f', vmin=-1, vmax=1)\nplt.title('Correlation Heatmap')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:20.192679Z","iopub.execute_input":"2025-06-06T11:38:20.192858Z","iopub.status.idle":"2025-06-06T11:38:20.586739Z","shell.execute_reply.started":"2025-06-06T11:38:20.192843Z","shell.execute_reply":"2025-06-06T11:38:20.585938Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Dựa vào heatmap ta biết được các feature có sự tương quan mạnh như:\n- Age vs Weight/Height\n- BMI vs Weight/Waist\n- ...","metadata":{}},{"cell_type":"code","source":"cols = [\n    'Basic_Demos-Age', 'FGC-FGC_CU', 'FGC-FGC_GSND', 'FGC-FGC_GSD', 'FGC-FGC_PU', 'FGC-FGC_SRL',\n    'FGC-FGC_SRR', 'FGC-FGC_TL', 'Fitness_Endurance-Time_Mins', 'PreInt_EduHx-computerinternet_hoursday'\n]\ndata_subset = train[cols]\n\ncorr_matrix = data_subset.corr()\n\nplt.figure(figsize=(10, 8))\nsns.heatmap(corr_matrix, annot=True, cmap='coolwarm', fmt='.2f', vmin=-1, vmax=1)\nplt.title('Correlation Heatmap')\nplt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:20.589257Z","iopub.execute_input":"2025-06-06T11:38:20.589493Z","iopub.status.idle":"2025-06-06T11:38:21.003052Z","shell.execute_reply.started":"2025-06-06T11:38:20.589475Z","shell.execute_reply":"2025-06-06T11:38:21.002290Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Dựa vào heatmap ta biết được các feature FGC có sự tương quan mạnh như:\n- GSD vs GSND\n- CU vs PU\n- SRR vs SRL","metadata":{}},{"cell_type":"code","source":"def feature_engineering(df):\n    def assign_age_group(age):\n        thresholds = [7, 9, 11, 13, 15, 22]\n        for i, t in enumerate(thresholds):\n            if age <= t:\n                return i\n        return len(thresholds)\n\n    df['AgeGroup'] = df['Basic_Demos-Age'].apply(assign_age_group)\n\n    # Age\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Height_Age'] = df['Physical-Height'] * df['Basic_Demos-Age']\n    df['Waist_Age'] = df['Physical-Waist_Circumference'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n\n    # Tạo ratio\n    df['Waist_Weight_Ratio'] = df['Physical-Waist_Circumference'] / df['Physical-Weight']\n\n    # Pulse Pressure\n    df['Pulse_Pressure'] = df['Physical-Systolic_BP'] - df['Physical-Diastolic_BP']\n\n    # Interaction với scale khác\n    df['SDS_Height'] = df['SDS-SDS_Total_Raw'] * df['Physical-Height']\n    df['CGAS_BMI'] = df['CGAS-CGAS_Score'] * df['Physical-BMI']\n\n    #FGC\n    df['FGC_GSD_GSND'] = df['FGC-FGC_GSD'] * df['FGC-FGC_GSND']\n    df['FGC_GSD_GSND_Age'] = df['FGC-FGC_GSD'] * df['FGC-FGC_GSD'] * df['Basic_Demos-Age']\n    df['FGC_CU_PU'] = df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n    df['FGC_CU_PU_Age'] = df['FGC-FGC_CU'] * df['FGC-FGC_PU'] * df['Basic_Demos-Age']\n    df['FGC_SRR_SRL'] = df['FGC-FGC_SRR'] * df['FGC-FGC_SRL']\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.003834Z","iopub.execute_input":"2025-06-06T11:38:21.004057Z","iopub.status.idle":"2025-06-06T11:38:21.010133Z","shell.execute_reply.started":"2025-06-06T11:38:21.004035Z","shell.execute_reply":"2025-06-06T11:38:21.009419Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1. Cleaning","metadata":{}},{"cell_type":"code","source":"train.loc[train['CGAS-CGAS_Score'] == 999, 'CGAS-CGAS_Score'] = np.nan","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.011049Z","iopub.execute_input":"2025-06-06T11:38:21.011287Z","iopub.status.idle":"2025-06-06T11:38:21.025089Z","shell.execute_reply.started":"2025-06-06T11:38:21.011266Z","shell.execute_reply":"2025-06-06T11:38:21.024319Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## - Recalc PCIAT-PCIAT_Total","metadata":{}},{"cell_type":"code","source":"# def recalculate_sii(row):\n#     if pd.isna(row['PCIAT-PCIAT_Total']):\n#         return np.nan\n#     max_possible = row['PCIAT-PCIAT_Total'] + row[question_columns].isna().sum() * 5\n#     if row['PCIAT-PCIAT_Total'] <= 30 and max_possible <= 30:\n#         return 0\n#     elif 31 <= row['PCIAT-PCIAT_Total'] <= 49 and max_possible <= 49:\n#         return 1\n#     elif 50 <= row['PCIAT-PCIAT_Total'] <= 79 and max_possible <= 79:\n#         return 2\n#     elif row['PCIAT-PCIAT_Total'] >= 80 and max_possible >= 80:\n#         return 3\n#     return np.nan\n\n# train['recalc_sii'] = train.apply(recalculate_sii, axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.025929Z","iopub.execute_input":"2025-06-06T11:38:21.026139Z","iopub.status.idle":"2025-06-06T11:38:21.036006Z","shell.execute_reply.started":"2025-06-06T11:38:21.026116Z","shell.execute_reply":"2025-06-06T11:38:21.035258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train['recalc_sii'].isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.036871Z","iopub.execute_input":"2025-06-06T11:38:21.037109Z","iopub.status.idle":"2025-06-06T11:38:21.050313Z","shell.execute_reply.started":"2025-06-06T11:38:21.037086Z","shell.execute_reply":"2025-06-06T11:38:21.049626Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# mismatch_rows = train[\n#     (train['recalc_sii'] != train['sii']) & train['sii'].notna()\n# ]\n\n# mismatch_rows[question_columns + ['recalc_sii']].style.map(\n#     lambda x: 'background-color: #FFC0CB' if pd.isna(x) else ''\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.051093Z","iopub.execute_input":"2025-06-06T11:38:21.051341Z","iopub.status.idle":"2025-06-06T11:38:21.062394Z","shell.execute_reply.started":"2025-06-06T11:38:21.051321Z","shell.execute_reply":"2025-06-06T11:38:21.061811Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train['sii'] = train['recalc_sii']\n# train = train.drop(mismatch_rows.index)\n\n# train[columns_not_in_test + ['recalc_sii']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.063168Z","iopub.execute_input":"2025-06-06T11:38:21.063420Z","iopub.status.idle":"2025-06-06T11:38:21.074561Z","shell.execute_reply.started":"2025-06-06T11:38:21.063398Z","shell.execute_reply":"2025-06-06T11:38:21.074013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# na_total_rows = train[train['sii'].isna()]\n# na_total_rows","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.075277Z","iopub.execute_input":"2025-06-06T11:38:21.075496Z","iopub.status.idle":"2025-06-06T11:38:21.085817Z","shell.execute_reply.started":"2025-06-06T11:38:21.075477Z","shell.execute_reply":"2025-06-06T11:38:21.085288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.dropna(subset=['PCIAT-PCIAT_Total'])\ntrain","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.086536Z","iopub.execute_input":"2025-06-06T11:38:21.086762Z","iopub.status.idle":"2025-06-06T11:38:21.130572Z","shell.execute_reply.started":"2025-06-06T11:38:21.086745Z","shell.execute_reply":"2025-06-06T11:38:21.130003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# for column in question_columns:\n#     if train[column].isna().any():\n#         mode_value = train[column].mode()[0]\n#         train[column] = train[column].fillna(mode_value)\n\n# train[columns_not_in_test + ['recalc_sii']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.131139Z","iopub.execute_input":"2025-06-06T11:38:21.131356Z","iopub.status.idle":"2025-06-06T11:38:21.134371Z","shell.execute_reply.started":"2025-06-06T11:38:21.131340Z","shell.execute_reply":"2025-06-06T11:38:21.133616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train.drop(columns='recalc_sii', inplace=True)\n# train[columns_not_in_test]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.135273Z","iopub.execute_input":"2025-06-06T11:38:21.135648Z","iopub.status.idle":"2025-06-06T11:38:21.147200Z","shell.execute_reply.started":"2025-06-06T11:38:21.135625Z","shell.execute_reply":"2025-06-06T11:38:21.146585Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## - Blood Pressure & Heart Rate","metadata":{}},{"cell_type":"code","source":"bp_hr_cols = [\n    'Physical-Diastolic_BP', 'Physical-Systolic_BP',\n    'Physical-HeartRate'\n]\n\n# Thay thế các giá trị = 0 thành NaN (vì không hợp lý cho huyết áp, nhịp tim)\ntrain[bp_hr_cols] = train[bp_hr_cols].replace(0, np.nan)\n\n# Gán NaN cho từng ô < 40 trong các cột liên quan\nfor col in bp_hr_cols:\n    train.loc[train[col] < 40, col] = np.nan\n\n# Huyết áp tâm thu phải lớn hơn tâm trương\ntrain.loc[train['Physical-Systolic_BP'] <= train['Physical-Diastolic_BP'], bp_hr_cols] = np.nan","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.147829Z","iopub.execute_input":"2025-06-06T11:38:21.148026Z","iopub.status.idle":"2025-06-06T11:38:21.163527Z","shell.execute_reply.started":"2025-06-06T11:38:21.148010Z","shell.execute_reply":"2025-06-06T11:38:21.162854Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## - Physical-Height, Physical-Weight","metadata":{}},{"cell_type":"code","source":"wh_cols = [\n    'Physical-BMI', 'Physical-Height',\n    'Physical-Weight', 'Physical-Waist_Circumference'\n]\n\ntrain[wh_cols] = train[wh_cols].replace(0, np.nan)\ntest[wh_cols] = test[wh_cols].replace(0, np.nan)\ntrain[wh_cols].describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.164391Z","iopub.execute_input":"2025-06-06T11:38:21.164618Z","iopub.status.idle":"2025-06-06T11:38:21.191523Z","shell.execute_reply.started":"2025-06-06T11:38:21.164604Z","shell.execute_reply":"2025-06-06T11:38:21.191039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"object_columns = train.select_dtypes(include='object').columns\nprint(\"Các cột kiểu object:\")\nprint(object_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.192059Z","iopub.execute_input":"2025-06-06T11:38:21.192266Z","iopub.status.idle":"2025-06-06T11:38:21.197234Z","shell.execute_reply.started":"2025-06-06T11:38:21.192250Z","shell.execute_reply":"2025-06-06T11:38:21.196419Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lbs_to_kg = 0.453592\ninches_to_cm = 2.54\n\ndef process_physical_BMI(df):\n    df['Physical-Weight'] = df['Physical-Weight'] * lbs_to_kg\n    df['Physical-Height'] = df['Physical-Height'] * inches_to_cm\n    df['Physical-Waist_Circumference'] = df['Physical-Waist_Circumference'] * inches_to_cm\n    \n    df['Physical-BMI'] = np.where(\n        df['Physical-Weight'].notna() & df['Physical-Height'].notna(),\n        df['Physical-Weight'] / ((df['Physical-Height'] / 100) ** 2),\n        np.nan\n    )\n    \n    return df\n\ntrain = process_physical_BMI(train)\ntest = process_physical_BMI(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.198078Z","iopub.execute_input":"2025-06-06T11:38:21.198684Z","iopub.status.idle":"2025-06-06T11:38:21.211632Z","shell.execute_reply.started":"2025-06-06T11:38:21.198658Z","shell.execute_reply":"2025-06-06T11:38:21.211079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_features = ['Physical-Weight', 'Physical-Height', 'Basic_Demos-Age', 'Basic_Demos-Sex']\n\n# Nếu Basic_Demos-Sex chưa numeric, chuyển đổi, ví dụ:\n# train['Basic_Demos-Sex'] = train['Basic_Demos-Sex'].map({'Male':0, 'Female':1})\n# test['Basic_Demos-Sex'] = test['Basic_Demos-Sex'].map({'Male':0, 'Female':1})\n\n# Tách mẫu đầy đủ (các feature cần impute và bổ trợ đều không NaN)\ntrain_complete = train[train[selected_features].notna().all(axis=1)].copy()\ntrain_incomplete = train[train[selected_features].isna().any(axis=1)].copy()\n\nimputer = KNNImputer(n_neighbors=10)\nimputer.fit(train_complete[selected_features])\n\ntrain_incomplete_imputed = pd.DataFrame(\n    imputer.transform(train_incomplete[selected_features]),\n    columns=selected_features,\n    index=train_incomplete.index\n)\n\n# Cập nhật lại 2 cột weight và height sau khi impute\ntrain_incomplete_imputed_subset = train_incomplete_imputed[['Physical-Weight', 'Physical-Height']]\n\ntrain_incomplete = train_incomplete.drop(columns=['Physical-Weight', 'Physical-Height'])\ntrain = pd.concat([\n    train_complete,\n    pd.concat([train_incomplete, train_incomplete_imputed_subset], axis=1)\n]).sort_index()\n\n# Tương tự cho test\ntest_complete = test[test[selected_features].notna().all(axis=1)].copy()\ntest_incomplete = test[test[selected_features].isna().any(axis=1)].copy()\n\nimputer.fit(test_complete[selected_features])\n\ntest_incomplete_imputed = pd.DataFrame(\n    imputer.transform(test_incomplete[selected_features]),\n    columns=selected_features,\n    index=test_incomplete.index\n)\n\ntest_incomplete_imputed_subset = test_incomplete_imputed[['Physical-Weight', 'Physical-Height']]\n\ntest_incomplete = test_incomplete.drop(columns=['Physical-Weight', 'Physical-Height'])\ntest = pd.concat([\n    test_complete,\n    pd.concat([test_incomplete, test_incomplete_imputed_subset], axis=1)\n]).sort_index()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.212263Z","iopub.execute_input":"2025-06-06T11:38:21.212416Z","iopub.status.idle":"2025-06-06T11:38:21.303997Z","shell.execute_reply.started":"2025-06-06T11:38:21.212404Z","shell.execute_reply":"2025-06-06T11:38:21.303262Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## - BMI","metadata":{}},{"cell_type":"code","source":"train['Physical-BMI'] = train.apply(\n    lambda row: row['Physical-Weight'] / (row['Physical-Height'] / 100) ** 2 \n    if pd.isnull(row['Physical-BMI']) else row['Physical-BMI'], axis=1\n)\n\ntest['Physical-BMI'] = test.apply(\n    lambda row: row['Physical-Weight'] / (row['Physical-Height'] / 100) ** 2 \n    if pd.isnull(row['Physical-BMI']) else row['Physical-BMI'], axis=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.304735Z","iopub.execute_input":"2025-06-06T11:38:21.304930Z","iopub.status.idle":"2025-06-06T11:38:21.339336Z","shell.execute_reply.started":"2025-06-06T11:38:21.304908Z","shell.execute_reply":"2025-06-06T11:38:21.338858Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## - Waist","metadata":{}},{"cell_type":"code","source":"waist_data = train.dropna(subset=['Physical-Waist_Circumference'])\n\nX = waist_data[['Physical-BMI', 'Basic_Demos-Age']]\ny = waist_data['Physical-Waist_Circumference']\n\nfrom sklearn.linear_model import LinearRegression\nwaist_model = LinearRegression()\nwaist_model.fit(X, y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.339903Z","iopub.execute_input":"2025-06-06T11:38:21.340081Z","iopub.status.idle":"2025-06-06T11:38:21.371001Z","shell.execute_reply.started":"2025-06-06T11:38:21.340068Z","shell.execute_reply":"2025-06-06T11:38:21.370485Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_waist = train['Physical-Waist_Circumference'].isnull()\ntrain.loc[missing_waist, 'Physical-Waist_Circumference'] = waist_model.predict(\n    train.loc[missing_waist, ['Physical-BMI', 'Basic_Demos-Age']]\n)\n\nmissing_waist_test = test['Physical-Waist_Circumference'].isnull()\ntest.loc[missing_waist_test, 'Physical-Waist_Circumference'] = waist_model.predict(\n    test.loc[missing_waist_test, ['Physical-BMI', 'Basic_Demos-Age']]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.374100Z","iopub.execute_input":"2025-06-06T11:38:21.374620Z","iopub.status.idle":"2025-06-06T11:38:21.384333Z","shell.execute_reply.started":"2025-06-06T11:38:21.374603Z","shell.execute_reply":"2025-06-06T11:38:21.383594Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## - BIA","metadata":{}},{"cell_type":"code","source":"# Danh sách các cột thuộc nhóm BIA\nbia_cols = [\n    'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n    'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_ICW', 'BIA-BIA_LDM',\n    'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW'\n]\n# Các thuộc tính nhóm này không thể là âm => Replace bằng NaN\n\nprint(\"\\n======== Trước khi Replace ========\")\nprint(\"\\nTập train:\")\nprint((train[bia_cols] < 0).sum())\nprint(\"\\nTập test:\")\nprint((test[bia_cols] < 0).sum())\n\ntrain[bia_cols] = train[bia_cols].applymap(lambda x: np.nan if x < 0 else x)\n# Tập test ko có nên ko cần replace\n\n# Kiểm tra lại sau khi Replace\nprint(\"\\n======== Sau khi Replace ========\")\nprint(\"\\nTập train:\")\nprint((train[bia_cols] < 0).sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.385056Z","iopub.execute_input":"2025-06-06T11:38:21.385275Z","iopub.status.idle":"2025-06-06T11:38:21.413290Z","shell.execute_reply.started":"2025-06-06T11:38:21.385259Z","shell.execute_reply":"2025-06-06T11:38:21.412791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def cleaning_BIA(df):\n    # Xóa các dữ liệu bị sai lệch lớn (vô lý)\n    \n    # % mỡ cơ thể\n    df['BIA-BIA_Fat'] = np.where(df['BIA-BIA_Fat'] < 5, np.nan, df['BIA-BIA_Fat'])\n    df['BIA-BIA_Fat'] = np.where(df['BIA-BIA_Fat'] > 60, np.nan, df['BIA-BIA_Fat'])\n    # Bone Mineral Content (Khối lượng khoáng trong xương) (kg)\n    df['BIA-BIA_BMC'] = np.where(df['BIA-BIA_BMC'] < 0.5, np.nan, df['BIA-BIA_BMC'])\n    df['BIA-BIA_BMC'] = np.where(df['BIA-BIA_BMC'] > 5, np.nan, df['BIA-BIA_BMC'])\n    # Body Mass Index (chỉ số khối cơ thể)\n    df['BIA-BIA_BMI'] = np.where(df['BIA-BIA_BMI'] < 10, np.nan, df['BIA-BIA_BMI'])\n    df['BIA-BIA_BMI'] = np.where(df['BIA-BIA_BMI'] > 40, np.nan, df['BIA-BIA_BMI'])\n    # Basal Metabolic Rate (kcal/ngày)\n    df['BIA-BIA_BMR'] = np.where(df['BIA-BIA_BMR'] < 600, np.nan, df['BIA-BIA_BMR'])\n    df['BIA-BIA_BMR'] = np.where(df['BIA-BIA_BMR'] > 3000, np.nan, df['BIA-BIA_BMR'])\n    # Daily Energy Expenditure (kcal/ngày)\n    df['BIA-BIA_DEE'] = np.where(df['BIA-BIA_DEE'] < 800, np.nan, df['BIA-BIA_DEE'])\n    df['BIA-BIA_DEE'] = np.where(df['BIA-BIA_DEE'] > 5000, np.nan, df['BIA-BIA_DEE'])\n    # Extracellular Water – Nước ngoài tế bào (L)\n    df['BIA-BIA_ECW'] = np.where(df['BIA-BIA_ECW'] < 3, np.nan, df['BIA-BIA_ECW'])\n    df['BIA-BIA_ECW'] = np.where(df['BIA-BIA_ECW'] > 15, np.nan, df['BIA-BIA_ECW'])\n    # Fat-Free Mass – Khối lượng không mỡ (kg)\n    df['BIA-BIA_FFM'] = np.where(df['BIA-BIA_FFM'] < 10, np.nan, df['BIA-BIA_FFM'])\n    df['BIA-BIA_FFM'] = np.where(df['BIA-BIA_FFM'] > 60, np.nan, df['BIA-BIA_FFM'])\n    # Fat-Free Mass Index\n    df['BIA-BIA_FFMI'] = np.where(df['BIA-BIA_FFMI'] < 8, np.nan, df['BIA-BIA_FFMI'])\n    df['BIA-BIA_FFMI'] = np.where(df['BIA-BIA_FFMI'] > 25, np.nan, df['BIA-BIA_FFMI'])\n    # Fat Mass Index\n    df['BIA-BIA_FMI'] = np.where(df['BIA-BIA_FMI'] < 1, np.nan, df['BIA-BIA_FMI'])\n    df['BIA-BIA_FMI'] = np.where(df['BIA-BIA_FMI'] > 15, np.nan, df['BIA-BIA_FMI'])\n    # Intracellular Water – Nước trong tế bào (L)\n    df['BIA-BIA_ICW'] = np.where(df['BIA-BIA_ICW'] < 5, np.nan, df['BIA-BIA_ICW'])\n    df['BIA-BIA_ICW'] = np.where(df['BIA-BIA_ICW'] > 20, np.nan, df['BIA-BIA_ICW'])\n    # Lean Dry Mass – Khối lượng nạc không nước (kg)\n    df['BIA-BIA_LDM'] = np.where(df['BIA-BIA_LDM'] < 5, np.nan, df['BIA-BIA_LDM'])\n    df['BIA-BIA_LDM'] = np.where(df['BIA-BIA_LDM'] > 40, np.nan, df['BIA-BIA_LDM'])\n    # Lean Soft Tissue – Mô mềm không mỡ (kg)\n    df['BIA-BIA_LST'] = np.where(df['BIA-BIA_LST'] < 10, np.nan, df['BIA-BIA_LST'])\n    df['BIA-BIA_LST'] = np.where(df['BIA-BIA_LST'] > 55, np.nan, df['BIA-BIA_LST'])\n    # Skeletal Muscle Mass – Khối cơ xương (kg)\n    df['BIA-BIA_SMM'] = np.where(df['BIA-BIA_SMM'] < 5, np.nan, df['BIA-BIA_SMM'])\n    df['BIA-BIA_SMM'] = np.where(df['BIA-BIA_SMM'] > 40, np.nan, df['BIA-BIA_SMM'])\n    # Total Body Water – Tổng lượng nước trong cơ thể (L)\n    df['BIA-BIA_TBW'] = np.where(df['BIA-BIA_TBW'] < 10, np.nan, df['BIA-BIA_TBW'])\n    df['BIA-BIA_TBW'] = np.where(df['BIA-BIA_TBW'] > 40, np.nan, df['BIA-BIA_TBW'])\n\n    return df\n\ntrain = cleaning_BIA(train)\ntest = cleaning_BIA(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.413945Z","iopub.execute_input":"2025-06-06T11:38:21.414173Z","iopub.status.idle":"2025-06-06T11:38:21.436991Z","shell.execute_reply.started":"2025-06-06T11:38:21.414150Z","shell.execute_reply":"2025-06-06T11:38:21.436212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[bia_cols].isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.437717Z","iopub.execute_input":"2025-06-06T11:38:21.437919Z","iopub.status.idle":"2025-06-06T11:38:21.445220Z","shell.execute_reply.started":"2025-06-06T11:38:21.437905Z","shell.execute_reply":"2025-06-06T11:38:21.444605Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## - Imputation","metadata":{}},{"cell_type":"code","source":"# Features to exclude, because they're not in test\nexclude = ['PCIAT-Season', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03',\n           'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07',\n           'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11',\n           'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15',\n           'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19',\n           'PCIAT-PCIAT_20', 'PCIAT-PCIAT_Total', 'sii', 'id']\n\nfeatures = [f for f in train.columns if f not in exclude]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.445972Z","iopub.execute_input":"2025-06-06T11:38:21.446251Z","iopub.status.idle":"2025-06-06T11:38:21.454966Z","shell.execute_reply.started":"2025-06-06T11:38:21.446224Z","shell.execute_reply":"2025-06-06T11:38:21.454308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.base import clone\n\nclass Impute_With_Model:\n    def __init__(self, na_frac=0.5, min_samples=0):\n        self.model_dict = {} # Dictionary storing models for imputation\n        self.mean_dict = {} # Dictionary storing mean of feature\n        self.features = None\n        self.na_frac = na_frac # Maximum fraction of missing values allowed for features to be used\n        self.min_samples = min_samples # Minimum number of samples required for model fitting\n        \n    def find_features(self, data, feature, tmp_features):\n        # Finds valid features where the fraction of missing values is <= na_frac\n        missing_rows = data[feature].isna()\n        na_fraction = data[missing_rows][tmp_features].isna().mean(axis=0)\n        valid_features = np.array(tmp_features)[na_fraction <= self.na_frac]\n        return valid_features\n\n    def fit_models(self, model, data, features):\n        self.features = features\n        n_data = data.shape[0]\n        for feature in features:\n            self.mean_dict[feature] = np.mean(data[feature])\n        # Iterate over all features\n        for feature in tqdm(features):\n            # Impute if there are missing values in the data\n            if data[feature].isna().sum() > 0:\n                model_clone = clone(model)\n                X = data[data[feature].notna()].copy() # Select data where target values are available as trainings data\n                tmp_features = [f for f in features if f != feature]\n                tmp_features = self.find_features(data, feature, tmp_features)\n                if len(tmp_features) >= 1 and X.shape[0] > self.min_samples:\n                    # Fit model if enough features and sufficient samples\n                    for f in tmp_features:\n                        X[f] = X[f].fillna(self.mean_dict[f])\n                    model_clone.fit(X[tmp_features], X[feature])\n                    # Add model and features to dictionary\n                    self.model_dict[feature] = (model_clone, tmp_features.copy())\n                else:\n                    # Revert to mean imputation if too few \n                    self.model_dict[feature] = (\"mean\", np.mean(data[feature]))\n\n    \n    def impute(self, data):\n        imputed_data = data.copy()\n        # Iterate over models\n        for feature, model in self.model_dict.items():\n            missing_rows = imputed_data[feature].isna()  # Identify rows to be imputed\n            if missing_rows.any():\n                if model[0] == \"mean\":\n                    imputed_data[feature].fillna(model[1], inplace=True)\n                else:\n                    tmp_features = [f for f in self.features if f != feature]\n                    X_missing = data.loc[missing_rows, tmp_features].copy()\n                    for f in tmp_features:\n                        X_missing[f] = X_missing[f].fillna(self.mean_dict[f])\n                    imputed_data.loc[missing_rows, feature] = model[0].predict(X_missing[model[1]])\n        return imputed_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.455727Z","iopub.execute_input":"2025-06-06T11:38:21.455968Z","iopub.status.idle":"2025-06-06T11:38:21.467837Z","shell.execute_reply.started":"2025-06-06T11:38:21.455943Z","shell.execute_reply":"2025-06-06T11:38:21.467244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LassoCV\n\nmodel = LassoCV(cv=5, random_state=seed)\nimputer = Impute_With_Model(na_frac=0.4) \n# na_frac is the maximum fraction of missing values until which a feature is imputed with the model\n# if there are more missing values than for example 40% then we revert to mean imputation\nimputer.fit_models(model, train, features)\ntrain = imputer.impute(train)\ntest = imputer.impute(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:21.468723Z","iopub.execute_input":"2025-06-06T11:38:21.469008Z","iopub.status.idle":"2025-06-06T11:38:27.562641Z","shell.execute_reply.started":"2025-06-06T11:38:21.468983Z","shell.execute_reply":"2025-06-06T11:38:27.562037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[train.columns[train.isnull().any()]].isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:27.563200Z","iopub.execute_input":"2025-06-06T11:38:27.563394Z","iopub.status.idle":"2025-06-06T11:38:27.571687Z","shell.execute_reply.started":"2025-06-06T11:38:27.563379Z","shell.execute_reply":"2025-06-06T11:38:27.570944Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Toàn bộ tập dữ liệu đã được làm sạch và fill nan toàn bộ","metadata":{}},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"train = feature_engineering(train)\ntest = feature_engineering(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:27.572493Z","iopub.execute_input":"2025-06-06T11:38:27.572738Z","iopub.status.idle":"2025-06-06T11:38:27.594003Z","shell.execute_reply.started":"2025-06-06T11:38:27.572723Z","shell.execute_reply":"2025-06-06T11:38:27.593555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:27.594587Z","iopub.execute_input":"2025-06-06T11:38:27.594833Z","iopub.status.idle":"2025-06-06T11:38:27.616217Z","shell.execute_reply.started":"2025-06-06T11:38:27.594813Z","shell.execute_reply":"2025-06-06T11:38:27.615676Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"=====================================================================================================================================","metadata":{}},{"cell_type":"code","source":"train = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:27.616831Z","iopub.execute_input":"2025-06-06T11:38:27.617057Z","iopub.status.idle":"2025-06-06T11:38:27.632219Z","shell.execute_reply.started":"2025-06-06T11:38:27.617036Z","shell.execute_reply":"2025-06-06T11:38:27.631636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.drop(columns=question_columns, errors='ignore')\ntrain = train.drop(columns=seasonFeatures, errors='ignore')\ntest = test.drop(columns=seasonFeatures, errors='ignore')\ntrain = train.drop(columns='PCIAT-PCIAT_Total', errors='ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:27.632847Z","iopub.execute_input":"2025-06-06T11:38:27.633030Z","iopub.status.idle":"2025-06-06T11:38:27.640392Z","shell.execute_reply.started":"2025-06-06T11:38:27.633012Z","shell.execute_reply":"2025-06-06T11:38:27.639584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"filtered_features = train.drop(columns=['id', 'sii'])\n\n# X và y\nX = filtered_features\ny = train['sii']\n\n# Chỉ chuẩn hóa (vì dữ liệu đã sạch, không có NaN)\nnum_transformer = Pipeline(steps=[\n    ('scaler', StandardScaler())\n])\n\n# Áp dụng pipeline cho tất cả các cột feature\npreprocessor = ColumnTransformer(transformers=[\n    ('num', num_transformer, filtered_features.columns.tolist())\n])\n\n# Fit và transform dữ liệu\npreprocessor.fit(X)\nX_transformed = pd.DataFrame(preprocessor.transform(X), columns=filtered_features.columns)\n\n# In thử dữ liệu đã transform\nprint(\"Transformed X DataFrame:\")\nprint(X_transformed.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:27.641190Z","iopub.execute_input":"2025-06-06T11:38:27.641479Z","iopub.status.idle":"2025-06-06T11:38:27.681892Z","shell.execute_reply.started":"2025-06-06T11:38:27.641455Z","shell.execute_reply":"2025-06-06T11:38:27.681202Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Split dataset","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_val, y_train, y_val = train_test_split(X_transformed, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:27.682573Z","iopub.execute_input":"2025-06-06T11:38:27.682774Z","iopub.status.idle":"2025-06-06T11:38:27.688903Z","shell.execute_reply.started":"2025-06-06T11:38:27.682752Z","shell.execute_reply":"2025-06-06T11:38:27.688245Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Hyparameter","metadata":{}},{"cell_type":"code","source":"LGB_Param = {\n    \"objective\": \"tweedie\",\n    \"n_estimators\": 343,\n    \"learning_rate\": 0.01491583811648972,\n    \"max_depth\": 11,\n    \"num_leaves\": 133,\n    \"subsample\": 0.5768522149618107,\n    \"colsample_bytree\": 0.8322639703021466,\n    \"min_child_samples\": 93,\n    \"reg_alpha\": 1.5054907940864373e-05,\n    \"reg_lambda\": 6.746281196067214e-07,\n    \"tweedie_variance_power\": 1.6011965627090141\n}\n\nXGBoost_Param = {\n    \"objective\": \"reg:squarederror\",\n    \"n_estimators\": 992,\n    \"learning_rate\": 0.015067889525153925,\n    \"max_depth\": 13,\n    \"subsample\": 0.5515490404431652,\n    \"colsample_bytree\": 0.8914202589303925,\n    \"reg_alpha\": 0.0002210307381259698,\n    \"reg_lambda\": 0.003196949907054426,\n    \"gamma\": 2.639848574832164,\n    \"min_child_weight\": 10\n}\n\nCatBoost_Param = {\n    \"loss_function\": \"RMSE\",\n    \"iterations\": 964,\n    \"learning_rate\": 0.010836158187714499,\n    \"depth\": 10,\n    \"l2_leaf_reg\": 2.0866426311980947e-07,\n    \"random_strength\": 9.773272168320332,\n    \"bagging_temperature\": 0.442580577573594,\n    \"od_type\": \"IncToDec\",\n    \"od_wait\": 16\n}\n\nGradientBoost_Param = {\n    \"n_estimators\": 344,\n    \"learning_rate\": 0.012940913108396936,\n    \"max_depth\": 4,\n    \"min_samples_split\": 18,\n    \"min_samples_leaf\": 18,\n    \"subsample\": 0.5670486908558695,\n    \"max_features\": None\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:27.690026Z","iopub.execute_input":"2025-06-06T11:38:27.690260Z","iopub.status.idle":"2025-06-06T11:38:27.699686Z","shell.execute_reply.started":"2025-06-06T11:38:27.690239Z","shell.execute_reply":"2025-06-06T11:38:27.698909Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Train Model","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nfrom sklearn.metrics import cohen_kappa_score, make_scorer\nfrom sklearn.model_selection import cross_val_score, StratifiedKFold\nfrom sklearn.ensemble import (\n    RandomForestRegressor,\n    GradientBoostingRegressor, \n    AdaBoostRegressor,\n    ExtraTreesRegressor,\n    BaggingRegressor,\n    StackingRegressor,\n    VotingRegressor\n)\nfrom lightgbm import LGBMRegressor\nfrom catboost import CatBoostRegressor\nfrom xgboost import XGBRegressor\nfrom scipy.optimize import minimize\n\n# Cấu hình\nnp.random.seed(seed)\nwarnings.filterwarnings(\"ignore\")\n\n# Khởi tạo các mô hình Regressor\nLGB_Model = LGBMRegressor(**LGB_Param,random_state=seed, verbose=-1)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Param,random_state=seed, verbose=0)\nXGB_Model = XGBRegressor(**XGBoost_Param,random_state=seed)\n\n# Hàm tính Quadratic Weighted Kappa\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred.round().astype(int), weights='quadratic')\n\n# Hàm tối ưu Threshold Rounding\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\n# Tạo mô hình StackingRegressor\nstacking_model = StackingRegressor(\n    estimators=[\n        ('lgb', LGB_Model),\n        ('cat', CatBoost_Model),\n        ('xgb', XGB_Model)\n    ]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:27.700299Z","iopub.execute_input":"2025-06-06T11:38:27.700546Z","iopub.status.idle":"2025-06-06T11:38:32.331939Z","shell.execute_reply.started":"2025-06-06T11:38:27.700516Z","shell.execute_reply":"2025-06-06T11:38:32.331341Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Submit","metadata":{}},{"cell_type":"code","source":"# Preprocess the test data\nX_test = test.drop(columns=['id'])\nX_test = pd.DataFrame(preprocessor.transform(X_test), columns=filtered_features.columns)\n\nbest_model = StackingRegressor(\n    estimators=[\n        ('lgb', LGB_Model),\n        ('cat', CatBoost_Model),\n        ('xgb', XGB_Model)\n    ]# Meta-model\n)\n\nbest_model.fit(X_train, y_train)\n\n# Dự đoán trên tập test\ny_test_pred = best_model.predict(X_test)\n\n# Tối ưu hóa threshold để làm tròn\nKappaOptimizer = minimize(\n    evaluate_predictions,\n    x0=[0.5, 1.5, 2.5], args=(y_train, best_model.predict(X_train)),\n    method='Nelder-Mead'\n)\noptimized_thresholds = KappaOptimizer.x\n\n# Làm tròn kết quả về nhãn classification\ny_test_pred_rounded = threshold_Rounder(y_test_pred, optimized_thresholds)\n\n# Tạo file submission\nsubmission = pd.DataFrame({\n    'id': test['id'],\n    'sii': y_test_pred_rounded\n})\nsubmission.to_csv('submission.csv', index=False)\n\n# Load và kiểm tra kết quả\nsubmiss = pd.read_csv('submission.csv')\nprint(submiss)\nprint(\"✅ Submission file created successfully.\")  # In ra thông báo khi hoàn thành","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:32.332621Z","iopub.execute_input":"2025-06-06T11:38:32.333260Z","iopub.status.idle":"2025-06-06T11:50:12.261372Z","shell.execute_reply.started":"2025-06-06T11:38:32.333240Z","shell.execute_reply":"2025-06-06T11:50:12.260682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}