{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nfrom scipy.optimize import minimize\nimport seaborn as sns\nimport optuna\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import make_scorer\nfrom sklearn.metrics import confusion_matrix\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\n\n\nSEED = 42\nn_splits = 7","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:39:47.703953Z","iopub.execute_input":"2025-01-06T15:39:47.704468Z","iopub.status.idle":"2025-01-06T15:39:47.711993Z","shell.execute_reply.started":"2025-01-06T15:39:47.704435Z","shell.execute_reply":"2025-01-06T15:39:47.710718Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1. Loading Data Process","metadata":{}},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:39:47.713560Z","iopub.execute_input":"2025-01-06T15:39:47.713871Z","iopub.status.idle":"2025-01-06T15:39:47.736002Z","shell.execute_reply.started":"2025-01-06T15:39:47.713844Z","shell.execute_reply":"2025-01-06T15:39:47.734763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n         'FGC-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n#Hợp nhất 2 loại dữ liệu\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n#Bỏ cột id\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n#Thêm các cột trong time_series vào\nfeaturesCols += time_series_cols\n#Kiểm lại data train + loại bỏ hàng có sii là null\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:39:47.737879Z","iopub.execute_input":"2025-01-06T15:39:47.738273Z","iopub.status.idle":"2025-01-06T15:41:10.092860Z","shell.execute_reply.started":"2025-01-06T15:39:47.738241Z","shell.execute_reply":"2025-01-06T15:41:10.091715Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Xử lí dữ liệu bị mất mát","metadata":{}},{"cell_type":"code","source":"#Đồ thị biểu thị tỉ lệ null trong mỗi cột\ndata_trainning = train[train['sii'].notnull()]\ndata_trainning_null = data_trainning.isnull().sum() * 100/len(data_trainning)\ndata_trainning_sorted = data_trainning_null.sort_values()\n#data_trainning_sorted = data_trainning_sorted[data_trainning_sorted != 0]\nplt.figure(figsize=(24,5))\nsns.barplot(x=data_trainning_sorted.index, y=data_trainning_sorted.values)\nplt.title(\"Percentage of missing values in each column\")\nplt.xlabel(\"Columns\")\nplt.xticks(rotation=90)\nplt.ylabel(\"Percentage of missing values\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:41:10.094550Z","iopub.execute_input":"2025-01-06T15:41:10.094967Z","iopub.status.idle":"2025-01-06T15:41:11.644563Z","shell.execute_reply.started":"2025-01-06T15:41:10.094927Z","shell.execute_reply":"2025-01-06T15:41:11.643371Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Xoá các đặc trưng bị mất mát dữ liệu lớn hơn 30%\ncolumns_to_drop = data_trainning_sorted[(data_trainning_sorted > 30)].index\ntrain = train.drop(columns=columns_to_drop)\ntest  = test.drop(columns=columns_to_drop)\ntrain.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:41:11.646854Z","iopub.execute_input":"2025-01-06T15:41:11.647223Z","iopub.status.idle":"2025-01-06T15:41:11.664318Z","shell.execute_reply.started":"2025-01-06T15:41:11.647188Z","shell.execute_reply":"2025-01-06T15:41:11.662883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Sau khi loại bỏ các đặc trung có mất mát lớn hơn 30%\ndata_trainning = train[train['sii'].notnull()]\ndata_trainning_null = data_trainning.isnull().sum() * 100/len(data_trainning)\ndata_trainning_sorted = data_trainning_null.sort_values()\n#data_trainning_sorted = data_trainning_sorted[data_trainning_sorted != 0]\nplt.figure(figsize=(24,5))\nsns.barplot(x=data_trainning_sorted.index, y=data_trainning_sorted.values)\nplt.title(\"Percentage of missing values in each column\")\nplt.xlabel(\"Columns\")\nplt.xticks(rotation=90)\nplt.ylabel(\"Percentage of missing values\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:41:11.666113Z","iopub.execute_input":"2025-01-06T15:41:11.666558Z","iopub.status.idle":"2025-01-06T15:41:12.133781Z","shell.execute_reply.started":"2025-01-06T15:41:11.666515Z","shell.execute_reply":"2025-01-06T15:41:12.132561Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Xử lí dữ liệu về BMI","metadata":{}},{"cell_type":"markdown","source":"Có thể thấy dữ liệu về bmi có độ mất mát ít (<10%) vì vậy mà việc làm đầy dữ liệu của cột này là khả thi.","metadata":{}},{"cell_type":"code","source":"def fill_bmi_nulls(df):\n    # Tạo một hàm để lấy giá trị mode cho mỗi nhóm, với điều kiện kiểm tra nhóm có mode hay không\n    def mode_for_group(group):\n        mode_val = group.mode()\n        # Kiểm tra xem có mode không, nếu không có, trả về NaN hoặc giá trị mặc định\n        return mode_val[0] if not mode_val.empty else None\n\n    # Áp dụng mode_for_group cho mỗi nhóm của Basic_Demos-Age và Basic_Demos-Gender\n    bmi_mode = df.groupby(['Basic_Demos-Age', 'Basic_Demos-Sex'])['Physical-BMI'].apply(mode_for_group)\n\n    # Thay thế giá trị null trong 'Physical-BMI' bằng giá trị mode của nhóm tương ứng\n    df['Physical-BMI'] = df.apply(\n        lambda row: bmi_mode.loc[row['Basic_Demos-Age'], row['Basic_Demos-Sex']] \n        if pd.isnull(row['Physical-BMI']) else row['Physical-BMI'], axis=1\n    )\n    \n    return df\n\n# Sử dụng hàm để xử lý dữ liệu train và test\ntrain = fill_bmi_nulls(train)\ntest = fill_bmi_nulls(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:41:12.134861Z","iopub.execute_input":"2025-01-06T15:41:12.135131Z","iopub.status.idle":"2025-01-06T15:41:12.197165Z","shell.execute_reply.started":"2025-01-06T15:41:12.135107Z","shell.execute_reply":"2025-01-06T15:41:12.196139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:41:12.198193Z","iopub.execute_input":"2025-01-06T15:41:12.198539Z","iopub.status.idle":"2025-01-06T15:41:12.212632Z","shell.execute_reply.started":"2025-01-06T15:41:12.198512Z","shell.execute_reply":"2025-01-06T15:41:12.211389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#chuẩn hóa các thang về bmi\ndef process_data(df):\n    # Chuyển chỉ số BMI thành 7 mức đánh số\n    df['BMI_Category'] = pd.cut(\n        df['Physical-BMI'],\n        bins=[float('-inf'), 16, 17, 18.5, 25, 30, 35, 40, float('inf')],\n        labels=[0, 1, 2, 3, 4, 5, 6, 7]  # Sử dụng số thay vì nhãn văn bản\n    )\n    def assign_age_group(age):\n        thresholds = [5, 6, 7, 8, 10, 12, 14, 18, 22]\n        for i, j in enumerate(thresholds):\n            if age <= j:\n                return i\n        return np.nan\n    df[\"age_group\"] = df['Basic_Demos-Age'].apply(assign_age_group)\n    features_to_normalize = [\n        'FGC-FGC_CU', 'FGC-FGC_TL'\n    ]\n    group_stats = df.groupby('age_group')[features_to_normalize].agg(['mean', 'std']).reset_index()\n    group_stats.columns = ['_'.join(col).strip('_') for col in group_stats.columns.values]\n    df = df.merge(group_stats, on='age_group', how='left')\n    for feature in features_to_normalize:\n        df[f\"{feature.split('_')[-1]}_norm\"] = (df[feature] - df[f\"{feature}_mean\"]) / df[f\"{feature}_std\"]\n    \n    # Mở rộng danh mục để bao gồm -1\n    df['BMI_Category'] = df['BMI_Category'].cat.add_categories([-1])\n    # Thay thế NaN bằng -1\n    df['BMI_Category'] = df['BMI_Category'].fillna(-1)\n    # Chuyển đổi sang kiểu int nếu cần\n    df['BMI_Category'] = df['BMI_Category'].astype('int')\n    # Tính toán các cột mới\n    df['BMI_Age'] = (df['BMI_Category'] + 1) * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n\n    drop_feats = ['Physical-BMI','Physical-Height','Physical-Weight','FGC-FGC_CU', 'FGC-FGC_TL', 'age_group','FGC-FGC_CU_mean','FGC-FGC_CU_std','FGC-FGC_TL_mean','FGC-FGC_TL_std']\n    df = df.drop(drop_feats, axis=1)\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:41:12.213837Z","iopub.execute_input":"2025-01-06T15:41:12.214235Z","iopub.status.idle":"2025-01-06T15:41:12.231819Z","shell.execute_reply.started":"2025-01-06T15:41:12.214193Z","shell.execute_reply":"2025-01-06T15:41:12.230556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = process_data(train)\ntest = process_data(test)\ntrain_tmp= train\ntest_tmp = test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:41:12.234876Z","iopub.execute_input":"2025-01-06T15:41:12.235259Z","iopub.status.idle":"2025-01-06T15:41:12.282488Z","shell.execute_reply.started":"2025-01-06T15:41:12.235211Z","shell.execute_reply":"2025-01-06T15:41:12.281218Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:41:12.284458Z","iopub.execute_input":"2025-01-06T15:41:12.284750Z","iopub.status.idle":"2025-01-06T15:41:12.298829Z","shell.execute_reply.started":"2025-01-06T15:41:12.284725Z","shell.execute_reply":"2025-01-06T15:41:12.297405Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Features EDA","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15, 8))\nsns.barplot(x='Basic_Demos-Age', y='sii', data=train, estimator='mean', ci=None)\nplt.title('Giá trị trung bình của sii theo Basic_Demos-Age ')\nplt.xlabel('Basic_Demos-Age ')\nplt.ylabel('Trung bình sii')\nplt.xticks(rotation=90)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:41:12.300034Z","iopub.execute_input":"2025-01-06T15:41:12.300484Z","iopub.status.idle":"2025-01-06T15:41:12.670390Z","shell.execute_reply.started":"2025-01-06T15:41:12.300440Z","shell.execute_reply":"2025-01-06T15:41:12.669196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\ntrain_tmp_copy = train_tmp.copy()\ntrain_tmp_copy['Age_Group'] = pd.cut(train_tmp_copy['Basic_Demos-Age'], bins=[0, 6, 12, 18, 24, 30], \n                                     labels=['<6', '6-12', '12-18', '18-24', '24-30'])\n# Tính trung bình SII theo nhóm\ngrouped_data = train_tmp_copy.groupby(['BMI_Category', 'Age_Group'])['sii'].agg(['mean', 'std']).reset_index()\n\n# Vẽ biểu đồ\nplt.figure(figsize=(12, 6))\nsns.barplot(data=grouped_data, x='BMI_Category', y='mean', hue='Age_Group', ci=None, palette='viridis')\n\n# Thêm chi tiết\nplt.title('SII theo BMI_Category và Age Group', fontsize=16)\nplt.xlabel('BMI_Category', fontsize=14)\nplt.ylabel('SII (Trung bình)', fontsize=14)\nplt.legend(title='Nhóm tuổi', fontsize=12)\nplt.grid(axis='y', linestyle='--', alpha=0.7)\nplt.tight_layout()\nplt.show()\n#biểu đồ Giá trị trung bình của sii theo BMI_Age\nplt.figure(figsize=(12, 6))\nsns.barplot(x='BMI_Age', y='sii', data=train_tmp, estimator='mean', ci=None)\nplt.title('Giá trị trung bình của sii theo BMI_Age')\nplt.xlabel('BMI_Age')\nplt.ylabel('Trung bình sii')\nplt.xticks(rotation=90)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:41:12.671539Z","iopub.execute_input":"2025-01-06T15:41:12.671852Z","iopub.status.idle":"2025-01-06T15:41:13.840393Z","shell.execute_reply.started":"2025-01-06T15:41:12.671828Z","shell.execute_reply":"2025-01-06T15:41:13.839291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(15, 8))\nsns.barplot(x='PreInt_EduHx-computerinternet_hoursday', y='sii', data=train, estimator='mean', ci=None)\nplt.title('Giá trị trung bình của sii theo computerinternet_hoursday')\nplt.xlabel('PreInt_EduHx-computerinternet_hoursday')\nplt.ylabel('Trung bình sii')\nplt.xticks(rotation=90)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:41:13.841583Z","iopub.execute_input":"2025-01-06T15:41:13.841940Z","iopub.status.idle":"2025-01-06T15:41:14.116766Z","shell.execute_reply.started":"2025-01-06T15:41:13.841913Z","shell.execute_reply":"2025-01-06T15:41:14.115674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(15, 8))\nsns.barplot(x='Internet_Hours_Age', y='sii', data=train, estimator='mean', ci=None)\nplt.title('Giá trị trung bình của sii theo Internet_Hours_Age')\nplt.xlabel('Internet_Hours_Age')\nplt.ylabel('Trung bình sii')\nplt.xticks(rotation=90)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:41:14.117911Z","iopub.execute_input":"2025-01-06T15:41:14.118201Z","iopub.status.idle":"2025-01-06T15:41:14.568749Z","shell.execute_reply.started":"2025-01-06T15:41:14.118177Z","shell.execute_reply":"2025-01-06T15:41:14.567534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Phân loại các cột\nnumerical_features = train.select_dtypes(include=['int64', 'float64']).columns.drop('sii')\n\n# Vẽ biểu đồ đường cho các feature dạng số\nfor feature in numerical_features:\n    plt.figure(figsize=(10, 5))\n    grouped_data = train.groupby(feature)['sii'].mean().reset_index()\n    sns.lineplot(data=grouped_data, x=feature, y='sii', marker='o', color='b')\n    plt.title(f'Biểu đồ Đường giữa {feature} và sii', fontsize=14)\n    plt.xlabel(feature, fontsize=12)\n    plt.ylabel('Trung bình sii', fontsize=12)\n    plt.grid(True, linestyle='--', alpha=0.5)\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:41:14.569889Z","iopub.execute_input":"2025-01-06T15:41:14.570259Z","iopub.status.idle":"2025-01-06T15:41:19.664883Z","shell.execute_reply.started":"2025-01-06T15:41:14.570220Z","shell.execute_reply":"2025-01-06T15:41:19.663815Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Mapping data","metadata":{}},{"cell_type":"code","source":"def update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\ntrain = update(train)\ntest = update(test)\n\n#Hàm đưa ra giá trị duy nhất\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n#Xử lí các cột Season và đánh số thay cho các mùa.\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:41:19.666148Z","iopub.execute_input":"2025-01-06T15:41:19.666530Z","iopub.status.idle":"2025-01-06T15:41:19.710778Z","shell.execute_reply.started":"2025-01-06T15:41:19.666490Z","shell.execute_reply":"2025-01-06T15:41:19.709870Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Trainning","metadata":{}},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:41:19.711842Z","iopub.execute_input":"2025-01-06T15:41:19.712210Z","iopub.status.idle":"2025-01-06T15:41:19.717972Z","shell.execute_reply.started":"2025-01-06T15:41:19.712173Z","shell.execute_reply":"2025-01-06T15:41:19.716822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_col = train.drop(['sii'], axis=1).columns\nall_importances = pd.DataFrame({'feature': feature_col})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T16:22:42.852241Z","iopub.execute_input":"2025-01-06T16:22:42.852654Z","iopub.status.idle":"2025-01-06T16:22:42.859886Z","shell.execute_reply.started":"2025-01-06T16:22:42.852622Z","shell.execute_reply":"2025-01-06T16:22:42.858619Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    lgb_importances_list = []\n    xgb_importances_list = []\n    cat_importances_list = []\n    \n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, KappaOPtimizer.x)\n    named_estimators = model.named_estimators_\n    feature_names = X.columns\n    # LightGBM\n    if 'lightgbm' in named_estimators and hasattr(named_estimators['lightgbm'], 'feature_importances_'):\n        lgb_importances_list.append(named_estimators['lightgbm'].feature_importances_)\n\n        # XGBoost\n    if 'xgboost' in named_estimators and hasattr(named_estimators['xgboost'], 'feature_importances_'):\n        xgb_importances_list.append(named_estimators['xgboost'].feature_importances_)\n\n        # CatBoost\n    if 'catboost' in named_estimators and hasattr(named_estimators['catboost'], 'get_feature_importance'):\n        cat_importances_list.append(named_estimators['catboost'].get_feature_importance())\n\n    def mean_importances(importances_list):\n        if len(importances_list) > 0:\n            return np.mean(importances_list, axis=0)\n        else:\n            return None\n\n    lgb_mean = mean_importances(lgb_importances_list)\n    xgb_mean = mean_importances(xgb_importances_list)\n    cat_mean = mean_importances(cat_importances_list)\n\n    def normalize_importances(importance_array):\n        if importance_array is not None:\n            return importance_array / importance_array.sum()\n        else:\n            return None\n\n    lgb_mean_normalized = normalize_importances(lgb_mean)\n    xgb_mean_normalized = normalize_importances(xgb_mean)\n    cat_mean_normalized = normalize_importances(cat_mean)\n\n    if lgb_mean_normalized is not None:\n        lgb_df = pd.DataFrame({'feature': feature_names, 'importance': lgb_mean_normalized}).sort_values('importance', ascending=False)\n        plt.figure(figsize=(10,20))\n        plt.barh(lgb_df['feature'], lgb_df['importance'])\n        plt.gca().invert_yaxis()\n        plt.title(\"LightGBM Feature Importance\")\n        plt.show()\n        all_importances['LightGBM'] = lgb_mean_normalized\n\n    if xgb_mean_normalized is not None:\n        xgb_df = pd.DataFrame({'feature': feature_names, 'importance': xgb_mean_normalized}).sort_values('importance', ascending=False)\n        plt.figure(figsize=(10,20))\n        plt.barh(xgb_df['feature'], xgb_df['importance'])\n        plt.gca().invert_yaxis()\n        plt.title(\"XGBoost Feature Importance\")\n        plt.show()\n        all_importances['XGBoost'] = xgb_mean_normalized\n\n    if cat_mean_normalized is not None:\n        cat_df = pd.DataFrame({'feature': feature_names, 'importance': cat_mean_normalized}).sort_values('importance', ascending=False)\n        plt.figure(figsize=(10,20))\n        plt.barh(cat_df['feature'], cat_df['importance'])\n        plt.gca().invert_yaxis()\n        plt.title(\"CatBoost Feature Importance\")\n        plt.show()\n        all_importances['CatBoost'] = cat_mean_normalized\n    return tp_rounded","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T16:22:46.901776Z","iopub.execute_input":"2025-01-06T16:22:46.902167Z","iopub.status.idle":"2025-01-06T16:22:46.921552Z","shell.execute_reply.started":"2025-01-06T16:22:46.902133Z","shell.execute_reply":"2025-01-06T16:22:46.920388Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Trainning best param","metadata":{}},{"cell_type":"code","source":"\ndef optimize_model(trial, model_name, X, y):\n    if model_name == 'LightGBM':\n        params = {\n            'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3),\n            'max_depth': trial.suggest_int('max_depth', 3, 15),\n            'num_leaves': trial.suggest_int('num_leaves', 20, 500),\n            'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 5, 50),\n            'feature_fraction': trial.suggest_float('feature_fraction', 0.5, 1.0),\n            'bagging_fraction': trial.suggest_float('bagging_fraction', 0.5, 1.0),\n            'bagging_freq': trial.suggest_int('bagging_freq', 1, 10),\n            'lambda_l1': trial.suggest_float('lambda_l1', 0, 10),\n            'lambda_l2': trial.suggest_float('lambda_l2', 0, 10),\n            'n_estimators': trial.suggest_int('n_estimators', 100, 500)\n        }\n        model = LGBMRegressor(**params, random_state=42, verbose=-1)\n\n    elif model_name == 'XGBoost':\n        params = {\n            'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3),\n            'max_depth': trial.suggest_int('max_depth', 3, 15),\n            'n_estimators': trial.suggest_int('n_estimators', 100, 500),\n            'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n            'reg_alpha': trial.suggest_float('reg_alpha', 0, 10),\n            'reg_lambda': trial.suggest_float('reg_lambda', 0, 10),\n            'tree_method': 'exact'\n        }\n        model = XGBRegressor(**params, random_state=42)\n\n    elif model_name == 'CatBoost':\n        params = {\n            'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3),\n            'depth': trial.suggest_int('depth', 3, 12),\n            'iterations': trial.suggest_int('iterations', 100, 500),\n            'l2_leaf_reg': trial.suggest_float('l2_leaf_reg', 1, 10),\n            'verbose': 0\n        }\n        model = CatBoostRegressor(**params, random_seed=42)\n\n    else:\n        raise ValueError(f\"Unsupported model_name: {model_name}\")\n\n    # Cross-validation to evaluate the model's performance\n    skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n    scores = []\n\n    for train_idx, val_idx in skf.split(X, y):\n        X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n\n        model.fit(X_train, y_train)\n        preds = model.predict(X_val)\n        preds_rounded = np.round(preds).astype(int)\n        score = quadratic_weighted_kappa(y_val, preds_rounded)\n        scores.append(score)\n\n    # Return the mean QWK score as the objective value for optimization\n    return np.mean(scores)\n\n# Hàm chạy Optuna để tối ưu hyperparameters cho từng mô hình\ndef run_optimization(model_name, X, y, n_trials=50):\n    study = optuna.create_study(direction='maximize')\n    study.optimize(lambda trial: optimize_model(trial, model_name, X, y), n_trials=n_trials)\n    \n    print(f\"Best parameters for {model_name}: {study.best_params}\")\n    print(f\"Best score: {study.best_value}\")\n    return study.best_params\n\n# Chuẩn bị dữ liệu\nX = train.drop(['sii'], axis=1)\ny = train['sii']\n\n# Ví dụ sử dụng Optuna cho các mô hình\nbest_params_lgb = run_optimization('LightGBM', X, y)\nbest_params_xgb = run_optimization('XGBoost', X, y)\nbest_params_cb = run_optimization('CatBoost', X, y)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:41:19.744670Z","iopub.execute_input":"2025-01-06T15:41:19.745039Z","iopub.status.idle":"2025-01-06T15:52:23.688556Z","shell.execute_reply.started":"2025-01-06T15:41:19.745005Z","shell.execute_reply":"2025-01-06T15:52:23.684872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()\ntest.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:52:23.691900Z","iopub.execute_input":"2025-01-06T15:52:23.692853Z","iopub.status.idle":"2025-01-06T15:52:23.755230Z","shell.execute_reply.started":"2025-01-06T15:52:23.692754Z","shell.execute_reply":"2025-01-06T15:52:23.751879Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3 Model","metadata":{}},{"cell_type":"code","source":"Params = {\n    'learning_rate': best_params_lgb['learning_rate'],\n    'max_depth': best_params_lgb['max_depth'],\n    'num_leaves': best_params_lgb['num_leaves'],\n    'min_data_in_leaf': best_params_lgb['min_data_in_leaf'],\n    'feature_fraction': best_params_lgb['feature_fraction'],\n    'bagging_fraction': best_params_lgb['bagging_fraction'],\n    'bagging_freq': best_params_lgb['bagging_freq'],\n    'lambda_l1': best_params_lgb['lambda_l1'],  \n    'lambda_l2': best_params_lgb['lambda_l2']\n}\n\n\nXGB_Params = {\n    'learning_rate': best_params_xgb['learning_rate'],\n    'max_depth': best_params_xgb['max_depth'],\n    'n_estimators': best_params_xgb['n_estimators'],\n    'subsample': best_params_xgb['subsample'],\n    'colsample_bytree': best_params_xgb['colsample_bytree'],\n    'reg_alpha': best_params_xgb['reg_alpha'],\n    'reg_lambda': best_params_xgb['reg_lambda'],\n    'random_state': SEED,\n    'tree_method': 'exact'\n}\n\n\nCatBoost_Params = {\n    'learning_rate': best_params_cb['learning_rate'],\n    'depth': best_params_cb['depth'],\n    'iterations': best_params_cb['iterations'],\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': best_params_cb['l2_leaf_reg']  # Increase this value\n}\n\n#Tạo model\ndef optimize_model(trial, model_name, X, y):\n    if model_name == 'LightGBM':\n        params = {\n            'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3),\n            'max_depth': trial.suggest_int('max_depth', 3, 15),\n            'num_leaves': trial.suggest_int('num_leaves', 20, 500),\n            'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 5, 50),\n            'feature_fraction': trial.suggest_float('feature_fraction', 0.5, 1.0),\n            'bagging_fraction': trial.suggest_float('bagging_fraction', 0.5, 1.0),\n            'bagging_freq': trial.suggest_int('bagging_freq', 1, 10),\n            'lambda_l1': trial.suggest_float('lambda_l1', 0, 10),\n            'lambda_l2': trial.suggest_float('lambda_l2', 0, 10),\n            'n_estimators': trial.suggest_int('n_estimators', 100, 500)\n        }\n        model = LGBMRegressor(**params, random_state=42, verbose=-1)\n\n    elif model_name == 'XGBoost':\n        params = {\n            'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3),\n            'max_depth': trial.suggest_int('max_depth', 3, 15),\n            'n_estimators': trial.suggest_int('n_estimators', 100, 500),\n            'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n            'reg_alpha': trial.suggest_float('reg_alpha', 0, 10),\n            'reg_lambda': trial.suggest_float('reg_lambda', 0, 10),\n            'tree_method': 'exact'\n        }\n        model = XGBRegressor(**params, random_state=42)\n\n    elif model_name == 'CatBoost':\n        params = {\n            'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3),\n            'depth': trial.suggest_int('depth', 3, 12),\n            'iterations': trial.suggest_int('iterations', 100, 500),\n            'l2_leaf_reg': trial.suggest_float('l2_leaf_reg', 1, 10),\n            'verbose': 0\n        }\n        model = CatBoostRegressor(**params, random_seed=42)\n\n    else:\n        raise ValueError(f\"Unsupported model_name: {model_name}\")\n\n    # Cross-validation to evaluate the model's performance\n    skf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)\n    scores = []\n\n    for train_idx, val_idx in skf.split(X, y):\n        X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n\n        model.fit(X_train, y_train)\n        preds = model.predict(X_val)\n        preds_rounded = np.round(preds).astype(int)\n        score = quadratic_weighted_kappa(y_val, preds_rounded)\n        scores.append(score)\n\n    # Return the mean QWK score as the objective value for optimization\n    return np.mean(scores)\n\n#Tạo model\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n#Lựa chọn model\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])\n\nSubmission = TrainML(voting_model, test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T16:22:52.614124Z","iopub.execute_input":"2025-01-06T16:22:52.614511Z","iopub.status.idle":"2025-01-06T16:23:07.801944Z","shell.execute_reply.started":"2025-01-06T16:22:52.614480Z","shell.execute_reply":"2025-01-06T16:23:07.800776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:52:23.852212Z","iopub.status.idle":"2025-01-06T15:52:23.853069Z","shell.execute_reply":"2025-01-06T15:52:23.852738Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission4 = pd.DataFrame({\n    'id': sample['id'],\n    'sii': Submission\n})\nSubmission4.to_csv('submission.csv', index=False)\nSubmission4","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T15:52:23.855429Z","iopub.status.idle":"2025-01-06T15:52:23.856237Z","shell.execute_reply":"2025-01-06T15:52:23.855941Z"}},"outputs":[],"execution_count":null}]}