{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30776,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Installing necessary packages","metadata":{}},{"cell_type":"code","source":"pip install lightgbm xgboost scikit-learn","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.metrics import classification_report\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.neighbors import KNeighborsClassifier, KNeighborsRegressor\nfrom sklearn.ensemble import RandomForestRegressor, VotingRegressor\nfrom xgboost import XGBClassifier,XGBRegressor\nimport lightgbm as lgb\nfrom sklearn.datasets import load_iris\nfrom sklearn.model_selection import StratifiedKFold, cross_val_score\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.metrics import precision_score, recall_score, f1_score, accuracy_score\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.metrics import cohen_kappa_score, confusion_matrix, ConfusionMatrixDisplay\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom colorama import Fore, Style\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error\nfrom sklearn.base import clone\nfrom scipy.optimize import minimize\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.model_selection import KFold\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nimport os\nfrom tqdm import tqdm\nfrom concurrent.futures import ThreadPoolExecutor","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> # Data Loading and Preprocessing","metadata":{}},{"cell_type":"code","source":"# Load the CSV data\ntrain_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ndata_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')\ntrain_data.head()\ntarget_labels = ['None', 'Mild', 'Moderate', 'Severe']","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get Statistical details\ntrain_data.describe().transpose()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.info()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Hàm phân bổ sai số","metadata":{}},{"cell_type":"code","source":"def calculate_stats(data, columns):\n    if isinstance(columns, str):\n        columns = [columns]\n\n    stats = []\n    for col in columns:\n        if data[col].dtype in ['object', 'category']:\n            counts = data[col].value_counts(dropna=False, sort=False)\n            percents = data[col].value_counts(normalize=True, dropna=False, sort=False) * 100\n            formatted = counts.astype(str) + ' (' + percents.round(2).astype(str) + '%)'\n            stats_col = pd.DataFrame({'count (%)': formatted})\n            stats.append(stats_col)\n        else:\n            stats_col = data[col].describe().to_frame().transpose()\n            stats_col['missing'] = data[col].isnull().sum()\n            stats_col.index.name = col\n            stats.append(stats_col)\n\n    return pd.concat(stats, axis=0)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Distribution of Label Data\ntrain_data['sii'].value_counts()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Lấy ra các cột dữ liệu không có trong test data","metadata":{}},{"cell_type":"code","source":"train_cols = set(train_data.columns)\ntest_cols = set(test_data.columns)\ncolumns_not_in_test = sorted(list(train_cols - test_cols))\ndata_dict[data_dict['Field'].isin(columns_not_in_test)]\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Kiểm tra dữ liệu thiếu trong PCIAT","metadata":{}},{"cell_type":"code","source":"train_with_sii = train_data[train_data['sii'].notna()][columns_not_in_test]\ntrain_with_sii[train_with_sii.isna().any(axis=1)].head().style.applymap(\n    lambda x: 'background-color: #FFC0CB' if pd.isna(x) else ''\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"như ta có thể thấy ở trên thì với hàng 1 câu hỏi PCIAT thứ 19 bị missing, nếu giả định câu hỏi đó được trả lời thì điểm sẽ rơi vào khoảng từ 1-5 => kq dự đoán sii phải là 1 chứ không phải 0. Ngoài ra còn có một số hàng có kết quả nan rất lớn => gây ra việc khó khăn trong dự đoán => cần loại bỏ cũng như điều chỉnh lại kq sii của train data.","metadata":{}},{"cell_type":"markdown","source":"# Hàm Chỉnh sửa label sii","metadata":{}},{"cell_type":"code","source":"PCIAT_cols = [f'PCIAT-PCIAT_{i+1:02d}' for i in range(20)]\nrecalc_total_score = train_with_sii[PCIAT_cols].sum(\n    axis=1, skipna=True\n)\n\ndef recalculate_sii(row):\n    if pd.isna(row['PCIAT-PCIAT_Total']):\n        return np.nan\n    max_possible = row['PCIAT-PCIAT_Total'] + row[PCIAT_cols].isna().sum() * 5\n    if row['PCIAT-PCIAT_Total'] <= 30 and max_possible <= 30:\n        return 0\n    elif 31 <= row['PCIAT-PCIAT_Total'] <= 49 and max_possible <= 49:\n        return 1\n    elif 50 <= row['PCIAT-PCIAT_Total'] <= 79 and max_possible <= 79:\n        return 2\n    elif row['PCIAT-PCIAT_Total'] >= 80 and max_possible >= 80:\n        return 3\n    return np.nan\ntrain_data['recalc_sii'] = train_data.apply(recalculate_sii, axis=1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Kiểm tra các hàng có kết quả sii khác so với kết quả sau tính toán lại","metadata":{}},{"cell_type":"code","source":"mismatch_rows = train_data[\n    (train_data['recalc_sii'] != train_data['sii']) & train_data['sii'].notna()\n]\n\nmismatch_rows[PCIAT_cols + [\n    'PCIAT-PCIAT_Total', 'sii', 'recalc_sii'\n]].style.applymap(\n    lambda x: 'background-color: #FFC0CB' if pd.isna(x) else ''\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Loại bỏ các hàng data có recal_sii là nan và sửa lại train data","metadata":{}},{"cell_type":"code","source":"train_data['sii'] = train_data['recalc_sii']\ntrain_data = train_data.dropna(subset=['sii'])\ntrain_data.drop(columns='recalc_sii', inplace=True)\ntrain_data['sii'].value_counts()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Kiểm tra phần trăm missing cho từng feature","metadata":{}},{"cell_type":"code","source":"nan_counts = train_data.isna().sum()  # Số lượng NaN\ntotal_rows = len(train_data)          # Tổng số hàng\nnan_percentage = nan_counts / total_rows * 100  # Tỷ lệ NaN\n\n# Tạo bảng tổng hợp\nnan_summary = pd.DataFrame({\n    'Tên cột': nan_counts.index,\n    'Số lượng NaN': nan_counts,\n    'Tổng số hàng': total_rows,\n    'Tỷ lệ NaN (%)': nan_percentage\n})\n\n# Lọc bảng chỉ cho các cột 'FGC-FGC_GSND' và 'FGC-FGC_GSD' nếu chúng có trong train_data\ncolumns_to_filter = ['FGC-FGC_GSND', 'FGC-FGC_GSD']\nfiltered_summary = nan_summary[nan_summary['Tên cột'].isin(columns_to_filter)]\n\nprint(train_data[['FGC-FGC_GSND', 'FGC-FGC_GSD']].isna().sum())\n\n# In kết quả\nprint(filtered_summary)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Sinh ra các cột giá trị mới cho train data","metadata":{}},{"cell_type":"code","source":"BIA_cols = [col for col in train_data.columns if 'BIA-BIA' in col]\nprint(BIA_cols)\nFGC_cols = [col for col in train_data.columns if 'FGC' in col]\nprint(FGC_cols)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n    \n    #Age\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['Physical-Waist_Age'] = df['Basic_Demos-Age'] * df['Physical-Waist_Circumference']\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Physical-Height_Age'] = df['Basic_Demos-Age'] * df['Physical-Height']\n    df['SDS_InternetHours'] = df['SDS-SDS_Total_T'] * df['PreInt_EduHx-computerinternet_hoursday']\n\n    #SDS\n    df['SDS_BMI'] = df['BIA-BIA_BMI'] * df['SDS-SDS_Total_T']\n    df['CGAS_SDS'] = df['CGAS-CGAS_Score'] * df['SDS-SDS_Total_T']\n    df['CGAS_Endurance_Mins'] = df['CGAS-CGAS_Score'] * df['Fitness_Endurance-Time_Mins']\n    df['SDS_Activity'] = df['BIA-BIA_Activity_Level_num'] * df['SDS-SDS_Total_T']\n\n    df['BMI_Systolic_BP'] = df['BIA-BIA_BMI'] * df['Physical-Systolic_BP']\n    df['Age_Systolic_BP'] = df['Basic_Demos-Age'] * df['Physical-Systolic_BP']\n    df['PreInt_Systolic_BP'] = df['Physical-Systolic_BP'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['PAQ_A_Activity'] = df['BIA-BIA_Activity_Level_num'] * df['PAQ_A-PAQ_A_Total']\n    df['Activity_CU_PU'] = df['BIA-BIA_Activity_Level_num'] * df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n\n    #FGC\n    df['FGC_CU_PU'] = df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n    df['FGC_CU_PU_Age'] = df['FGC-FGC_CU'] * df['FGC-FGC_PU'] * df['Basic_Demos-Age']\n    df['FGC_GSND_GSD'] = df['FGC-FGC_GSND'] * df['FGC-FGC_GSD']\n    df['FGC_GSND_GSD_Age'] = df['FGC-FGC_GSND'] * df['FGC-FGC_GSD'] * df['Basic_Demos-Age']\n    df['CGAS_CU_PU'] = df['CGAS-CGAS_Score'] * df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n    df['PreInt_FGC_CU_PU'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n    df['Endurance_CU_PU'] = df['Fitness_Endurance-Time_Mins'] * df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n\n    #Behavioral\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1)\n    \n    return df","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = feature_engineering(train_data)\ntest_data = feature_engineering(test_data)\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.replace([np.inf, -np.inf], np.nan, inplace=True)\ntest_data.replace([np.inf, -np.inf], np.nan, inplace=True)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_data.columns)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(test_data.columns)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Time series","metadata":{}},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df    ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.LeakyReLU(0.2),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.LeakyReLU(0.2),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.LeakyReLU(0.2)\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.LeakyReLU(0.2),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.LeakyReLU(0.2),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n\n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n\n    data_tensor = torch.FloatTensor(df_scaled)\n\n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n\n    criterion = F.smooth_l1_loss\n    optimizer = optim.Adam(autoencoder.parameters())\n\n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n\n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}')\n\n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n\n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n\n    return df_encoded\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ndf_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n\ntrain_ts_encoded = perform_autoencoder(df_train, encoding_dim=60, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test, encoding_dim=60, epochs=100, batch_size=32)\n\ntime_series_cols = train_ts_encoded.columns.tolist()\n\ntrain_ts_encoded[\"id\"] = train_ts[\"id\"]\ntest_ts_encoded[\"id\"] = test_ts[\"id\"]\n\ntrain_data = pd.merge(train_data, train_ts_encoded, how=\"left\", on='id')\ntest_data = pd.merge(test_data, test_ts_encoded, how=\"left\", on='id')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_data[time_series_cols])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(test_data[time_series_cols])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Dự đoán dữ liệu","metadata":{}},{"cell_type":"markdown","source":"# Dự đoán weight, height, bmi ","metadata":{}},{"cell_type":"markdown","source":"# Lọc các data sai thực tế đưa về nan","metadata":{}},{"cell_type":"markdown","source":"# điểm khả năng hoạt động CGAS-CGAS-Score bị vượt mức giời hạn","metadata":{}},{"cell_type":"code","source":"train_data.loc[train_data['CGAS-CGAS_Score'] > 100, 'CGAS-CGAS_Score'] = np.nan","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# sai thực tế dữ liệu Height, weight","metadata":{}},{"cell_type":"markdown","source":"quy đổi đơn vị","metadata":{}},{"cell_type":"code","source":"wh_cols = [\n    'Physical-BMI', 'Physical-Height',\n    'Physical-Weight', 'Physical-Waist_Circumference'\n]\n\nlbs_to_kg = 0.453592\ninches_to_cm = 2.54\n\ntrain_data['Physical-Weight'] = train_data['Physical-Weight'] * lbs_to_kg\ntrain_data['Physical-Height'] = train_data['Physical-Height'] * inches_to_cm\ntrain_data['Physical-Waist_Circumference'] = train_data['Physical-Waist_Circumference'] * inches_to_cm\n\n# Recalculate BMI: BMI = weight (kg) / (height (m)^2)\ntrain_data['Physical-BMI'] = np.where(\n    train_data['Physical-Weight'].notna() & train_data['Physical-Height'].notna(),\n    train_data['Physical-Weight'] / ((train_data['Physical-Height'] / 100) ** 2),\n    np.nan  # If either is NaN, set BMI to NaN\n)\n\ncalculate_stats(train_data, wh_cols)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ntrain_data[wh_cols] = train_data[wh_cols].replace(0, np.nan)\n(train_data[wh_cols] == 0).sum()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Sơ đồ plot tìm các điểm ngoại lệ","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(1, 3, figsize=(15, 5))\n\n# Vẽ biểu đồ Height vs Age\naxs[0].scatter(train_data['Basic_Demos-Age'], train_data['Physical-Height'], color='blue')\naxs[0].set_xlabel('Basic_Demos-Age')\naxs[0].set_ylabel('Physical-Height')\naxs[0].set_title('Height vs Age')\n\n# Vẽ biểu đồ Weight vs Age\naxs[1].scatter(train_data['Basic_Demos-Age'], train_data['Physical-Weight'], color='green')\naxs[1].set_xlabel('Basic_Demos-Age')\naxs[1].set_ylabel('Physical-Weight')\naxs[1].set_title('Weight vs Age')\n\n# Vẽ biểu đồ Waist vs Age\naxs[2].scatter(train_data['Physical-Weight'], train_data['Physical-Waist_Circumference'], color='red')\naxs[2].set_xlabel('Physical-Weight')\naxs[2].set_ylabel('Physical-Waist_Circumference')\naxs[2].set_title('Waist vs Age')\n\n# Hiển thị các biểu đồ\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def find_outliers_iqr(group, column):\n    Q1 = group[column].quantile(0.25)\n    Q3 = group[column].quantile(0.75)\n    IQR = Q3 - Q1\n    lower_bound = Q1 - 1.5 * IQR\n    upper_bound = Q3 + 1.5 * IQR\n    outliers = group[(group[column] < lower_bound) | (group[column] > upper_bound)]\n    return outliers\n\ndef find_outliers_column(column):\n    Q1 = column.quantile(0.25)\n    Q3 = column.quantile(0.75)\n    IQR = Q3 - Q1\n    lower_bound = Q1 - 1.5 * IQR\n    upper_bound = Q3 + 1.5 * IQR\n    outliers = column[(column < lower_bound) | (column > upper_bound)]\n    return outliers","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"outliers_height = train_data.groupby('Basic_Demos-Age').apply(lambda group: find_outliers_iqr(group, 'Physical-Height'))\noutliers_weight = train_data.groupby('Basic_Demos-Age').apply(lambda group: find_outliers_iqr(group, 'Physical-Weight'))\noutliers_waist = train_data.groupby('Physical-Weight').apply(lambda group: find_outliers_iqr(group, 'Physical-Waist_Circumference'))\n\n\nprint(\"Outliers in Height by Age:\")\nprint(outliers_height[['Basic_Demos-Age', 'Physical-Height']])\n\nprint(\"\\nOutliers in Weight by Age:\")\nprint(outliers_weight[['Basic_Demos-Age', 'Physical-Weight']])\n\nprint(\"\\nOutliers in Waist Circumference by Age:\")\nprint(outliers_waist[['Physical-Weight', 'Physical-Waist_Circumference']])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"loại bỏ giá trị height, weight bất thường","metadata":{}},{"cell_type":"code","source":"train_data = train_data[~((train_data['Basic_Demos-Age'] == 7) & (train_data['Physical-Height'] > 160))]\ntrain_data = train_data[~((train_data['Basic_Demos-Age'] == 16) & (train_data['Physical-Weight'] > 120))]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axs = plt.subplots(1, 3, figsize=(15, 5))\n\n# Vẽ biểu đồ Height vs Age\naxs[0].scatter(train_data['Basic_Demos-Age'], train_data['Physical-Height'], color='blue')\naxs[0].set_xlabel('Basic_Demos-Age')\naxs[0].set_ylabel('Physical-Height')\naxs[0].set_title('Height vs Age')\n\n# Vẽ biểu đồ Weight vs Age\naxs[1].scatter(train_data['Basic_Demos-Age'], train_data['Physical-Weight'], color='green')\naxs[1].set_xlabel('Basic_Demos-Age')\naxs[1].set_ylabel('Physical-Weight')\naxs[1].set_title('Weight vs Age')\n\n# Vẽ biểu đồ Waist vs Age\naxs[2].scatter(train_data['Physical-Weight'], train_data['Physical-Waist_Circumference'], color='red')\naxs[2].set_xlabel('Physical-Weight')\naxs[2].set_ylabel('Physical-Waist_Circumference')\naxs[2].set_title('Waist vs Age')\n\n# Hiển thị các biểu đồ\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# sai thực tế dữ liệu chỉ số tim mạch","metadata":{}},{"cell_type":"markdown","source":"có thể loại bỏ các chỉ số <50 vì nó là các chỉ số nguy hiểm","metadata":{}},{"cell_type":"code","source":"bp_hr_cols = [\n    'Physical-Diastolic_BP', 'Physical-Systolic_BP',\n    'Physical-HeartRate'\n]\ntrain_data[bp_hr_cols] = train_data[bp_hr_cols].replace(0, np.nan)\ntrain_data.loc[train_data['Physical-Systolic_BP'] <= train_data['Physical-Diastolic_BP'], bp_hr_cols] = np.nan","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Fitness Edurance thiếu thời gian","metadata":{}},{"cell_type":"markdown","source":"việc loại bỏ các cột có phút hoặc giây về nan chỉ là chủ quan vì giây có thể không quan trọng nên có thể nan và có trường hợp người ta chỉ chạy được mấy giây nên là phút sẽ là nan => có thể điều chỉnh điều kiện lọc để kiểm tra mô hình","metadata":{}},{"cell_type":"code","source":"cols = [\n    'Fitness_Endurance-Max_Stage',\n    'Fitness_Endurance-Time_Mins',\n    'Fitness_Endurance-Time_Sec'\n]\ntrain_data.loc[\n    (train_data['Fitness_Endurance-Max_Stage'].notna()) & \n    (train_data['Fitness_Endurance-Time_Mins'].isna() | \n     train_data['Fitness_Endurance-Time_Sec'].isna()), cols\n] = np.nan\n\ntrain_data.loc[\n    (train_data['Fitness_Endurance-Max_Stage'] == 0), cols\n] = np.nan","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"sau khi lọc thêm cột mới cho train data và test data","metadata":{}},{"cell_type":"code","source":"train_data['Fitness_Endurance-Total_Time_Sec'] = train_data[\n    'Fitness_Endurance-Time_Mins'\n] * 60 + train_data['Fitness_Endurance-Time_Sec']\n\ntest_data['Fitness_Endurance-Total_Time_Sec'] = test_data[\n    'Fitness_Endurance-Time_Mins'\n] * 60 + test_data['Fitness_Endurance-Time_Sec']","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Kiểm tra lại dữ liệu ","metadata":{}},{"cell_type":"code","source":"calculate_stats(train_data, ['Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Total_Time_Sec'])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# BIA","metadata":{}},{"cell_type":"code","source":"# train_data[BIA_cols] = train_data[BIA_cols].applymap(lambda x: np.nan if x < 0 else x)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"outliers_dict = {}\nfor column in BIA_cols:\n    outliers_dict[column] = find_outliers_column(train_data[column])\n\n# In ra các outliers cho từng cột\nfor column, outliers in outliers_dict.items():\n    print(f\"Outliers in {column}:\")\n    print(outliers)\n    print()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Xử lý Missing data","metadata":{}},{"cell_type":"markdown","source":"# Dùng knn để điền missing data","metadata":{}},{"cell_type":"code","source":"#imputer = KNNImputer(n_neighbors=5)\n#numeric_cols = train_data.select_dtypes(include=['float64', 'int64']).columns\n#imputed_data = imputer.fit_transform(train_data[numeric_cols])\n#train_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\n#train_imputed['sii'] = train_imputed['sii'].round().astype(int)\n#for col in train_data.columns:\n    #if col not in numeric_cols:\n      #train_imputed[col] = train_data[col]\n        \n#train_data = train_imputed\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"dùng linear regression để dự đoán vòng eo","metadata":{}},{"cell_type":"code","source":"\ndef cross_validate_model(model, X, y, model_name):\n    \n    if isinstance(X, np.ndarray):\n        X = pd.DataFrame(X)\n    if isinstance(y, np.ndarray):\n        y = pd.Series(y)\n\n    kf = KFold(n_splits=5, shuffle=True, random_state=42)\n    r2_scores = []\n    mae_scores = []\n    rmse_scores = []\n\n    for train_index, test_index in kf.split(X, y):\n        X_train, X_test = X.iloc[train_index], X.iloc[test_index]  \n        y_train, y_test = y.iloc[train_index], y.iloc[test_index]  \n\n        model.fit(X_train, y_train)\n        y_pred = model.predict(X_test)\n\n        \n        r2_scores.append(r2_score(y_test, y_pred))\n        mae_scores.append(mean_absolute_error(y_test, y_pred))\n        rmse_scores.append(mean_squared_error(y_test, y_pred, squared=False))\n\n    \n    print(f\"Model: {model_name}\")\n    print(f\"R²: {np.mean(r2_scores):.4f} ± {np.std(r2_scores):.4f}\")\n    print(f\"MAE: {np.mean(mae_scores):.4f} ± {np.std(mae_scores):.4f}\")\n    print(f\"RMSE: {np.mean(rmse_scores):.4f} ± {np.std(rmse_scores):.4f}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#train_data_Physical_Waist_Circumference = train_data[train_data['Physical-Waist_Circumference'].notna() & train_data['Physical-Weight'].notna()]\n#missing_data_Physical_Waist_Circumference = train_data[train_data['Physical-Waist_Circumference'].isna() & train_data['Physical-Weight'].notna()]\n#X_train_Physical_Waist_Circumference = train_data_Physical_Waist_Circumference['Physical-Weight']\n#y_train_Physical_Waist_Circumference = train_data_Physical_Waist_Circumference['Physical-Waist_Circumference']\n#X_test_Physical_Waist_Circumference = missing_data_Physical_Waist_Circumference['Physical-Weight']\n#X_train_Physical_Waist_Circumference = X_train_Physical_Waist_Circumference.values.reshape(-1, 1)\n#X_test_Physical_Waist_Circumference = X_test_Physical_Waist_Circumference.values.reshape(-1, 1)\n\n# Linear Regression\n#linear_model = LinearRegression()\n#cross_validate_model(linear_model, X_train_Physical_Waist_Circumference, y_train_Physical_Waist_Circumference, \"Linear Regression\")\n\n# Random Forest Regression\n#rf_model = RandomForestRegressor(n_estimators=100, random_state=42)\n#cross_validate_model(rf_model, X_train_Physical_Waist_Circumference, y_train_Physical_Waist_Circumference, \"Random Forest Regression\")\n\n#linear_model.fit(X_train_Physical_Waist_Circumference, y_train_Physical_Waist_Circumference)\n#missing_data_Physical_Waist_Circumference['Physical-Waist_Circumference'] = linear_model.predict(X_test_Physical_Waist_Circumference)\n\n#train_data.loc[missing_data_Physical_Waist_Circumference.index, 'Physical-Waist_Circumference'] = missing_data_Physical_Waist_Circumference['Physical-Waist_Circumference']\n\n#print(train_data['Physical-Waist_Circumference'])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Dùng randomforest để đoán height, weight","metadata":{}},{"cell_type":"code","source":"#train_data_height = train_data[train_data['Physical-Height'].notna() & train_data['Basic_Demos-Age'].notna()]\n#missing_train_data_height = train_data[train_data['Physical-Height'].isna() & train_data['Basic_Demos-Age'].notna()]\n#Xtrain_height = train_data_height['Basic_Demos-Age']\n#y_train_height = train_data_height['Physical-Height']\n#X_test_height = missing_train_data_height['Basic_Demos-Age']\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#X_train_height = X_train_height.values.reshape(-1, 1)\n#X_test_height = X_test_height.values.reshape(-1, 1)\n\n#rf_model.fit(X_train_height, y_train_height)\n#missing_train_data_height['Physical-Height'] = linear_model.predict(X_test_height)\n\n#train_data.loc[missing_train_data_height.index, 'Physical-Height'] = missing_train_data_height['Physical-Height']\n\n#print(train_data[['Basic_Demos-Age', 'Physical-Height']])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#train_data_weight = train_data[train_data['Physical-Weight'].notna() & train_data['Basic_Demos-Age'].notna()]\n#missing_train_data_weight = train_data[train_data['Physical-Weight'].isna() & train_data['Basic_Demos-Age'].notna()]\n#X_train_weight = train_data_weight['Basic_Demos-Age']\n#y_train_weight = train_data_weight['Physical-Weight']\n#X_test_weight = missing_train_data_weight['Basic_Demos-Age']\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#X_train_weight = X_train_weight.values.reshape(-1, 1)\n#X_test_weight = X_test_weight.values.reshape(-1, 1)\n\n#rf_model.fit(X_train_weight, y_train_weight)\n#missing_train_data_weight['Physical-Weight'] = linear_model.predict(X_test_weight)\n\n#train_data.loc[missing_train_data_weight.index, 'Physical-Weight'] = missing_train_data_weight['Physical-Weight']\n\n#print(train_data[['Basic_Demos-Age', 'Physical-Weight']])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Loại bỏ các cột quá nhiều missing data","metadata":{}},{"cell_type":"code","source":"\ntrain_data = train_data.dropna(thresh=10, axis=0)\nthreshold = 0.5 * len(train_data)\ncolumns_with_data = train_data.columns[train_data.isnull().sum() < threshold]\ntrain_data = train_data[columns_with_data]\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# kiểm tra các cột còn lại của train data","metadata":{}},{"cell_type":"code","source":"train_data.info()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Làm sạch missing data","metadata":{}},{"cell_type":"code","source":"\ntrain_data = train_data.fillna(0)\n    \ntrain_data_cleaned = train_data\n\ntrain_data_cleaned.head()\ntrain_data_cleaned.info()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA (Exploratory Data Analysis)","metadata":{}},{"cell_type":"code","source":"groups = data_dict.groupby('Instrument')['Field'].apply(list).to_dict()\n\nfor instrument, features in groups.items():\n    print(f\"{instrument}: {features}\\n\")\n\n\ndata_dict = data_dict[data_dict['Instrument'] != 'Parent-Child Internet Addiction Test']\ncontinuous_cols = data_dict[data_dict['Type'].str.contains(\n    'float|int', case=False\n)]['Field'].tolist()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Phân tích thuộc tính demographics","metadata":{}},{"cell_type":"markdown","source":"# Phân lớp cột age","metadata":{}},{"cell_type":"code","source":"train_data_cleaned['Age Group'] = pd.cut(\n    train_data_cleaned['Basic_Demos-Age'],\n    bins=[4, 12, 18, 22],\n    labels=['Children (5-12)', 'Adolescents (13-18)', 'Adults (19-22)']\n)\n\nprint(train_data_cleaned['Age Group'])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Lập biểu đồ box plot để đánh giá mối liên hệ so với sii","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize=(18, 5))\n\n# Sii by Age\nsns.boxplot(y=train_data_cleaned['Basic_Demos-Age'], x=train_data_cleaned['sii'], ax=axes[0], palette=\"Set3\")\naxes[0].set_title('Sii by Age')\naxes[0].set_ylabel('Age')\naxes[0].set_xlabel('Sii')\n\n# Complete PCIAT Responses by Age Group\nsns.boxplot(\n    x='Age Group', y='PCIAT-PCIAT_Total',\n    data=train_data_cleaned, palette=\"Set3\", ax=axes[1]\n    )\naxes[1].set_title('Revised PCIAT Responses by Age Group')\naxes[1].set_ylabel('PCIAT_Total for Revised Responses')\naxes[1].set_xlabel('Age Group')\n\n# PCIAT_Total by Sex\nsns.histplot(\n    data=train_data_cleaned, x='PCIAT-PCIAT_Total',\n    hue='Basic_Demos-Sex', multiple='stack',\n    palette=\"Set3\", bins=20, ax=axes[2]\n)\naxes[2].set_title('PCIAT_Total Distribution by Sex')\naxes[2].set_xlabel('PCIAT_Total for revised Responses')\naxes[2].set_ylabel('Frequency')\n\n\naxes[1].set_title('Complete PCIAT Responses by Age Group')\naxes[1].set_ylabel('PCIAT_Total for revised Responses')\naxes[1].set_xlabel('Age Group')\n\n# PCIAT_Total by Sex\nsns.histplot(\n    data=train_data_cleaned, x='PCIAT-PCIAT_Total',\n    hue='Basic_Demos-Sex', multiple='stack',\n    palette=\"Set3\", bins=20, ax=axes[2]\n)\naxes[2].set_title('PCIAT_Total Distribution by Sex')\naxes[2].set_xlabel('PCIAT_Total for Complete Responses')\naxes[2].set_ylabel('Frequency')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Bảng Phân bố của sii dựa theo nhóm tuổi ","metadata":{}},{"cell_type":"code","source":"stats = train_data_cleaned.groupby(['Age Group', 'sii']).size().unstack(fill_value=0)\nstats_prop = stats.div(stats.sum(axis=1), axis=0) * 100\n\nstats = stats.astype(str) +' (' + stats_prop.round(1).astype(str) + '%)'\nstats","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Mã hóa cột nhóm tuổi","metadata":{}},{"cell_type":"code","source":"# Tạo LabelEncoder\nlabel_encoder = LabelEncoder()\n\n# Sử dụng fit_transform trực tiếp trên cột\ntrain_data_cleaned['Age Group'] = label_encoder.fit_transform(train_data_cleaned['Age Group'])\n\n# Kiểm tra kết quả\nprint(train_data_cleaned['Age Group'])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#groups.get('Physical Measures', [])\n\n#features_physical = groups.get('Physical Measures', [])\n\n#cols = [col for col in features_physical if col in continuous_cols]\n\n#calculate_stats(train_data_cleaned, cols)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Phân tích chỉ số weight, height","metadata":{}},{"cell_type":"markdown","source":"# Correlation Matrix","metadata":{}},{"cell_type":"code","source":"\nprint(train_data_cleaned.head())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop the 'id' column if present\ntrain_data_no_id = train_data_cleaned.drop(columns=['id'], errors='ignore')\n# Calculate the correlation matrix\ntrain_data_no_id_numeric = train_data_no_id.select_dtypes(include=['number'])\ncorrelation_matrix = train_data_no_id_numeric.corr()\n# Plot the heatmap\nplt.figure(figsize=(30, 30))\nsns.heatmap(correlation_matrix, annot=True, fmt='.1f', cmap='coolwarm', square=True)\nplt.title('Correlation Heatmap')\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Chuẩn bị dữ liệu để train mô hình","metadata":{}},{"cell_type":"code","source":"# Lấy danh sách các cột chung\ncommon_columns = train_data_cleaned.columns.intersection(test_data.columns)\ncommon_columns = list(common_columns)\nprint(common_columns)\n\nX = train_data_cleaned[common_columns].drop(columns=['id'])\ny = train_data_cleaned['sii']\n\ntest_data.fillna(0, inplace=True)\nX_test_data = test_data[common_columns].drop(columns=['id'])\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Contrust Model","metadata":{}},{"cell_type":"markdown","source":"# Tối ưu hóa tham số ","metadata":{}},{"cell_type":"markdown","source":"# Đánh giá hiệu suất các mô hình sẽ sử dụng và in kết quả","metadata":{}},{"cell_type":"code","source":"def round_with_thresholds(raw_preds, thresholds):\n     return np.where(raw_preds < thresholds[0], 0,\n                    np.where(raw_preds < thresholds[1], 1,\n                             np.where(raw_preds < thresholds[2], 2, 3)))\ndef fun(thresholds, y_true, raw_preds):    \n    rounded_preds = round_with_thresholds(raw_preds, thresholds)\n    return - cohen_kappa_score(y_true, rounded_preds, weights='quadratic')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"rf_best = RandomForestRegressor(\n    n_estimators=200,\n    max_depth=6,\n    max_features=0.8,\n    min_samples_split=2,\n    min_samples_leaf=1,\n    bootstrap=True,\n    random_state=42\n)\n\nxgb_best = XGBRegressor(\n    learning_rate=0.01,  \n    max_depth=4,  \n    n_estimators=500,  \n    subsample=0.7,  \n    colsample_bytree=0.7,  \n    reg_alpha=0.1,  \n    reg_lambda=10,  \n    min_child_weight=5, \n    random_state=42\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(len(X))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgboost = xgb_best\nrf = rf_best\n\nvoting_model = VotingRegressor(estimators=[('xgboost', xgboost), ('rf', rf)])\nmodels = {\n    \"XGBoost\": xgboost,\n    \"Random Forest\": rf,\n    \"Voting Regressor\": voting_model\n}\n\n\npca = PCA()\npca.fit(X)\n\nexplained_variance = pca.explained_variance_ratio_\n\nplt.figure(figsize=(8, 6))\nplt.plot(range(1, len(explained_variance) + 1), explained_variance, marker='o', linestyle='--', color='b')\nplt.title('Explained Variance per Principal Component')\nplt.xlabel('Number of Principal Components')\nplt.ylabel('Explained Variance')\nplt.show()\n\ncumulative_variance = np.cumsum(explained_variance)\nprint(\"Cumulative explained variance:\", cumulative_variance)\n\nn_components = np.argmax(cumulative_variance >= 0.95) + 1\nprint(f'Chọn số thành phần PCA: {n_components}')\n\nplt.plot(range(1, len(cumulative_variance) + 1), cumulative_variance, marker='o', color='r')\nplt.title('Cumulative Explained Variance (Elbow Method)')\nplt.xlabel('Number of Principal Components')\nplt.ylabel('Cumulative Explained Variance')\nplt.axvline(x=n_components, color='g', linestyle='--', label=f'Selected {n_components} components')\nplt.legend()\nplt.show()\n\n\n# Tạo StratifiedKFold\nn_splits = 10 #10\n\nfor name, model in models.items():\n    print(f\"\\nEvaluating model: {name}\")\n    kf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=1)\n    oof_raw = np.zeros(len(y), dtype=float)\n    oof = np.zeros(len(y), dtype=int)\n\n    train_scores = []  \n    val_scores = []    \n\n    for fold, (idx_tr, idx_va) in enumerate(kf.split(X, y)):\n        X_tr = X.iloc[idx_tr]\n        X_va = X.iloc[idx_va]\n        y_tr = y.iloc[idx_tr]\n        y_va = y.iloc[idx_va]\n\n        model = clone(model)\n        model.fit(X_tr, y_tr.to_numpy())\n\n        y_tr_pred = model.predict(X_tr)\n        y_tr_pred = y_tr_pred.round(0).astype(int)\n        train_score = cohen_kappa_score(y_tr, y_tr_pred, weights='quadratic')\n        train_scores.append(train_score)\n\n        y_va_pred = model.predict(X_va)\n        oof_raw[idx_va] = y_va_pred\n        y_va_pred = y_va_pred.round(0).astype(int)\n        oof[idx_va] = y_va_pred\n        val_score = cohen_kappa_score(y_va, y_va_pred, weights='quadratic')\n        val_scores.append(val_score)\n\n        print(f\"# Fold {fold}: Train Score = {train_score:.3f}, Validation Score = {val_score:.3f}\")\n\n    res = minimize(fun, x0=[0.5, 1.5, 2.5], args=(y, oof_raw), method='Nelder-Mead')\n    assert res.success\n    oof_tuned = round_with_thresholds(oof_raw, res.x)\n    print(f\"# Optimized thresholds: {res.x.round(2)}\")\n    print(f\"# Score with default rounding:     {cohen_kappa_score(y, oof, weights='quadratic'):.3f}\")\n    print(f\"# Score with optimized thresholds: {cohen_kappa_score(y, oof_tuned, weights='quadratic'):.3f}\")\n\n    ConfusionMatrixDisplay.from_predictions(y, oof)\n    plt.title(f'Confusion matrix without tuned thresholds ({name})')\n    plt.xticks(np.arange(4), target_labels)\n    plt.yticks(np.arange(4), target_labels)\n    plt.show()\n\n    ConfusionMatrixDisplay.from_predictions(y, oof_tuned)\n    plt.title(f'Confusion matrix with tuned thresholds ({name})')\n    plt.xticks(np.arange(4), target_labels)\n    plt.yticks(np.arange(4), target_labels)\n    plt.show()\n\n    # Plot overfitting evaluation\n    plt.figure(figsize=(6, 5))\n    plt.plot(range(len(train_scores)), train_scores, label='Training Score', marker='o', linestyle='-', color='b')\n    plt.plot(range(len(val_scores)), val_scores, label='Validation Score', marker='o', linestyle='--', color='r')\n    plt.xlabel('Fold')\n    plt.ylabel('Cohen Kappa Score')\n    plt.title(f'Overfitting Evaluation - {name}')\n    plt.legend()\n    plt.show()\n\n    # Evaluate overfitting after applying optimized thresholds\n    train_scores_tuned = []\n    val_scores_tuned = []\n\n    for fold, (idx_tr, idx_va) in enumerate(kf.split(X, y)):\n        X_tr = X.iloc[idx_tr]\n        X_va = X.iloc[idx_va]\n        y_tr = y.iloc[idx_tr]\n        y_va = y.iloc[idx_va]\n\n        model = clone(model)\n        model.fit(X_tr, y_tr.to_numpy())\n\n        # Predictions on training set\n        y_tr_raw_pred = model.predict(X_tr)\n        y_tr_tuned_pred = round_with_thresholds(y_tr_raw_pred, res.x)\n        train_score_tuned = cohen_kappa_score(y_tr, y_tr_tuned_pred, weights='quadratic')\n        train_scores_tuned.append(train_score_tuned)\n\n        # Predictions on validation set\n        y_va_raw_pred = model.predict(X_va)\n        y_va_tuned_pred = round_with_thresholds(y_va_raw_pred, res.x)\n        val_score_tuned = cohen_kappa_score(y_va, y_va_tuned_pred, weights='quadratic')\n        val_scores_tuned.append(val_score_tuned)\n\n    # Plot overfitting evaluation (before and after tuning)\n    plt.figure(figsize=(10, 6))\n    plt.plot(range(len(train_scores_tuned)), train_scores_tuned, label='Training Score (tuned)', marker='o', linestyle='-', color='g')\n    plt.plot(range(len(val_scores_tuned)), val_scores_tuned, label='Validation Score (tuned)', marker='o', linestyle='--', color='orange')\n    plt.xlabel('Fold')\n    plt.ylabel('Cohen Kappa Score')\n    plt.title(f'Overfitting Evaluation - {name}')\n    plt.legend()\n    plt.grid(True)\n    plt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Creating Submission file","metadata":{}},{"cell_type":"code","source":"\nrf_model = RandomForestRegressor(\n    n_estimators=200,\n    max_depth=6,\n    max_features=0.8,\n    min_samples_split=2,\n    min_samples_leaf=1,\n    bootstrap=True,\n    random_state=42\n)\n\n\ntest_preds_rf = np.zeros(len(test_data))\nkf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=1)\noof_raw = np.zeros(len(y), dtype=float)\noof = np.zeros(len(y), dtype=int)\n\nfor fold, (idx_tr, idx_va) in enumerate(kf.split(X, y)):\n    X_tr = X.iloc[idx_tr]\n    X_va = X.iloc[idx_va]\n    y_tr = y.iloc[idx_tr]\n    y_va = y.iloc[idx_va]\n\n    model = clone(rf_model)\n    model.fit(X_tr, y_tr.to_numpy())\n\n    y_va_pred = model.predict(X_va)\n    oof_raw[idx_va] += y_va_pred  \n    y_va_pred = y_va_pred.round(0).astype(int)\n    oof[idx_va] += y_va_pred    \n\n    test_preds_rf += model.predict(X_test_data) \n\nres = minimize(fun, x0=[0.5, 1.5, 2.5], args=(y, oof_raw), method='Nelder-Mead')\nassert res.success\n\noof_tuned = round_with_thresholds(oof_raw, res.x)\nprint(f\"# Optimized thresholds: {res.x.round(2)}\")\nprint(f\"# Score with default rounding:     {cohen_kappa_score(y, oof, weights='quadratic'):.3f}\")\nprint(f\"# Score with optimized thresholds: {cohen_kappa_score(y, oof_tuned, weights='quadratic'):.3f}\")\n\n\n\n# Chuẩn bị submission\nsubmission1 = pd.DataFrame({\n    'id': test_data['id'],\n    'sii': round_with_thresholds(test_preds_rf / kf.n_splits, res.x)\n})\n\n\nprint(submission1['sii'])\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nxgb_model = xgb_best\n\ntest_preds_xgb = np.zeros(len(test_data))\nkf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=1)\noof_raw = np.zeros(len(y), dtype=float)\noof = np.zeros(len(y), dtype=int)\n\nfor fold, (idx_tr, idx_va) in enumerate(kf.split(X, y)):\n    X_tr = X.iloc[idx_tr]\n    X_va = X.iloc[idx_va]\n    y_tr = y.iloc[idx_tr]\n    y_va = y.iloc[idx_va]\n\n    model = clone(xgb_model)\n    model.fit(X_tr, y_tr.to_numpy())\n\n    y_va_pred = model.predict(X_va)\n    oof_raw[idx_va] += y_va_pred  \n    y_va_pred = y_va_pred.round(0).astype(int)\n    oof[idx_va] += y_va_pred    \n\n    test_preds_xgb += model.predict(X_test_data) \n\nres = minimize(fun, x0=[0.5, 1.5, 2.5], args=(y, oof_raw), method='Nelder-Mead')\nassert res.success\n\noof_tuned = round_with_thresholds(oof_raw, res.x)\nprint(f\"# Optimized thresholds: {res.x.round(2)}\")\nprint(f\"# Score with default rounding:     {cohen_kappa_score(y, oof, weights='quadratic'):.3f}\")\nprint(f\"# Score with optimized thresholds: {cohen_kappa_score(y, oof_tuned, weights='quadratic'):.3f}\")\n\n\n\n# Chuẩn bị submission\nsubmission2 = pd.DataFrame({\n    'id': test_data['id'],\n    'sii': round_with_thresholds(test_preds_xgb / kf.n_splits, res.x)\n})\n\nprint(submission2['sii'])\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\ntest_preds_vote = np.zeros(len(test_data))\nkf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=1)\noof_raw = np.zeros(len(y), dtype=float)\noof = np.zeros(len(y), dtype=int)\n\nfor fold, (idx_tr, idx_va) in enumerate(kf.split(X, y)):\n    X_tr = X.iloc[idx_tr]\n    X_va = X.iloc[idx_va]\n    y_tr = y.iloc[idx_tr]\n    y_va = y.iloc[idx_va]\n\n    model = clone(voting_model)\n    model.fit(X_tr, y_tr.to_numpy())\n\n    y_va_pred = model.predict(X_va)\n    oof_raw[idx_va] += y_va_pred  \n    y_va_pred = y_va_pred.round(0).astype(int)\n    oof[idx_va] += y_va_pred    \n\n    test_preds_vote += model.predict(X_test_data) \n\nres = minimize(fun, x0=[0.5, 1.5, 2.5], args=(y, oof_raw), method='Nelder-Mead')\nassert res.success\n\noof_tuned = round_with_thresholds(oof_raw, res.x)\nprint(f\"# Optimized thresholds: {res.x.round(2)}\")\nprint(f\"# Score with default rounding:     {cohen_kappa_score(y, oof, weights='quadratic'):.3f}\")\nprint(f\"# Score with optimized thresholds: {cohen_kappa_score(y, oof_tuned, weights='quadratic'):.3f}\")\n\n\n\n# Chuẩn bị submission\nsubmission3 = pd.DataFrame({\n    'id': test_data['id'],\n    'sii': round_with_thresholds(test_preds_vote / kf.n_splits, res.x)\n})\n\nprint(submission3['sii'])\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub1 = submission1\nsub2 = submission2\nsub3 = submission3\n\nsub1 = sub1.sort_values(by='id').reset_index(drop=True)\nsub2 = sub2.sort_values(by='id').reset_index(drop=True)\nsub3 = sub3.sort_values(by='id').reset_index(drop=True)\n\ncombined = pd.DataFrame({\n    'id': sub1['id'],\n    'sii_1': sub1['sii'],\n    'sii_2': sub2['sii'],\n    'sii_3': sub3['sii']\n})\n\ndef majority_vote(row):\n    return row.mode()[0]\n\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\n\nsum_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\nsub3.to_csv('submission.csv', index=False)\nprint(sub3['sii'])","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}