{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30805,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# from pytorch_tabnet.tab_model import TabNetRegressor\nimport torch\nimport numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone, BaseEstimator, RegressorMixin\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score, mean_squared_error, mean_absolute_error, mean_absolute_percentage_error\nfrom sklearn.model_selection import StratifiedKFold, train_test_split, KFold\nfrom scipy.optimize import minimize\nfrom sklearn.decomposition import PCA\nfrom concurrent.futures import ThreadPoolExecutor\nimport random\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\nfrom torch.utils.data import Dataset,DataLoader\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, RobustScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nfrom matplotlib.ticker import MaxNLocator\nimport seaborn as sns\nwarnings.filterwarnings('ignore')\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n# pd.options.display.max_columns = None\n\nseed = 42\nn_splits = 4","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:23:23.322725Z","iopub.execute_input":"2024-12-04T14:23:23.325751Z","iopub.status.idle":"2024-12-04T14:23:42.906675Z","shell.execute_reply.started":"2024-12-04T14:23:23.325681Z","shell.execute_reply":"2024-12-04T14:23:42.905718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\ndef seed_everything(seed): #(đảm bảo dữ liệu ko đổi sau các vòng lặp)\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\nseed_everything(2024)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:23:42.908301Z","iopub.execute_input":"2024-12-04T14:23:42.908996Z","iopub.status.idle":"2024-12-04T14:23:42.919923Z","shell.execute_reply.started":"2024-12-04T14:23:42.908957Z","shell.execute_reply":"2024-12-04T14:23:42.919005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_data():\n    train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n    test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n    sample =  pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n    return train, test, sample","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:24:17.319905Z","iopub.execute_input":"2024-12-04T14:24:17.320611Z","iopub.status.idle":"2024-12-04T14:24:17.324771Z","shell.execute_reply.started":"2024-12-04T14:24:17.320577Z","shell.execute_reply":"2024-12-04T14:24:17.324002Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train, test, sample = load_data()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:24:17.991048Z","iopub.execute_input":"2024-12-04T14:24:17.991901Z","iopub.status.idle":"2024-12-04T14:24:18.062639Z","shell.execute_reply.started":"2024-12-04T14:24:17.991866Z","shell.execute_reply":"2024-12-04T14:24:18.061976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"imputer = KNNImputer(n_neighbors=5) #(xử lý dữ liệu thiếu)\nnumeric_cols = train.select_dtypes(include = ['int32', 'int64', 'float32', 'float64']).columns\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\n\n# add non-numeric columns\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\ntrain = train_imputed","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:24:18.460458Z","iopub.execute_input":"2024-12-04T14:24:18.461168Z","iopub.status.idle":"2024-12-04T14:24:25.429287Z","shell.execute_reply.started":"2024-12-04T14:24:18.461135Z","shell.execute_reply":"2024-12-04T14:24:25.428533Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df: pd.DataFrame) -> pd.DataFrame: #(phân tích, loại bỏ các cột ko cần thiết)\n    season_cols = [col for col in df.columns if 'season' in col.lower()]\n    df = df.drop(columns=season_cols, axis = 1)\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    df['BMI_PHR'] = df['Physical-BMI'] * df['Physical-HeartRate']\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:24:25.430739Z","iopub.execute_input":"2024-12-04T14:24:25.43105Z","iopub.status.idle":"2024-12-04T14:24:25.43809Z","shell.execute_reply.started":"2024-12-04T14:24:25.431007Z","shell.execute_reply":"2024-12-04T14:24:25.437204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess_tabular_data(df: pd.DataFrame, test: bool = False) -> pd.DataFrame: #(chạy hàm trên)\n    df = feature_engineering(df)\n    df = df.dropna(thresh = 10) # Drop rows with less than 10 non-NA values\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:24:25.439105Z","iopub.execute_input":"2024-12-04T14:24:25.439399Z","iopub.status.idle":"2024-12-04T14:24:25.452829Z","shell.execute_reply.started":"2024-12-04T14:24:25.439374Z","shell.execute_reply":"2024-12-04T14:24:25.452059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = preprocess_tabular_data(train)\ntest = feature_engineering(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:24:25.454561Z","iopub.execute_input":"2024-12-04T14:24:25.454874Z","iopub.status.idle":"2024-12-04T14:24:25.483557Z","shell.execute_reply.started":"2024-12-04T14:24:25.454848Z","shell.execute_reply":"2024-12-04T14:24:25.482924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.drop('id', axis = 1)\ntest.drop('id', axis = 1)\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:24:25.484443Z","iopub.execute_input":"2024-12-04T14:24:25.484691Z","iopub.status.idle":"2024-12-04T14:24:25.51163Z","shell.execute_reply.started":"2024-12-04T14:24:25.484667Z","shell.execute_reply":"2024-12-04T14:24:25.510768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"featuresCols = ['Basic_Demos-Age', 'Basic_Demos-Sex', #(xử lý các đặc trưng theo cột trong tập train)\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW', 'BMI_PHR']\n\nlen(featuresCols)\ntrain = train[featuresCols]\ntrain = train.dropna(subset = 'sii')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:24:25.512754Z","iopub.execute_input":"2024-12-04T14:24:25.513053Z","iopub.status.idle":"2024-12-04T14:24:25.521578Z","shell.execute_reply.started":"2024-12-04T14:24:25.513026Z","shell.execute_reply":"2024-12-04T14:24:25.520593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(train) #(xuất tập train ra kiểm tra thử)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:24:25.522702Z","iopub.execute_input":"2024-12-04T14:24:25.522974Z","iopub.status.idle":"2024-12-04T14:24:25.535588Z","shell.execute_reply.started":"2024-12-04T14:24:25.522924Z","shell.execute_reply":"2024-12-04T14:24:25.534878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head() #(xuất tập train ra kiểm tra thử)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:24:25.536466Z","iopub.execute_input":"2024-12-04T14:24:25.536716Z","iopub.status.idle":"2024-12-04T14:24:25.562992Z","shell.execute_reply.started":"2024-12-04T14:24:25.536692Z","shell.execute_reply":"2024-12-04T14:24:25.561885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if np.any(np.isinf(train)): #(kiểm tra có giá trị inf ko, nếu có -> NaN)\n    train = train.replace([np.inf, -np.inf], np.nan)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:24:25.56416Z","iopub.execute_input":"2024-12-04T14:24:25.564465Z","iopub.status.idle":"2024-12-04T14:24:25.577341Z","shell.execute_reply.started":"2024-12-04T14:24:25.56442Z","shell.execute_reply":"2024-12-04T14:24:25.576517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def QWK(y_true, y_pred): #(hàm tính độ tương đồng giữa 2 tập dữ liệu)\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds): #(hàm chuyển đổi giá trị liên tục -> các lớp phân loại)\n    oof_non_rounded = np.array(oof_non_rounded)\n    \n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n    \ndef evaluate_predictions(thresholds, y_true, oof_non_rounded): #(Tính giá trị QWK của dự đoán sau khi làm tròn dựa trên các ngưỡng)\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -QWK(y_true, rounded_p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:24:25.579842Z","iopub.execute_input":"2024-12-04T14:24:25.58038Z","iopub.status.idle":"2024-12-04T14:24:25.586067Z","shell.execute_reply.started":"2024-12-04T14:24:25.580342Z","shell.execute_reply":"2024-12-04T14:24:25.585276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = test.drop('id', axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:24:25.587011Z","iopub.execute_input":"2024-12-04T14:24:25.587254Z","iopub.status.idle":"2024-12-04T14:24:25.602922Z","shell.execute_reply.started":"2024-12-04T14:24:25.58723Z","shell.execute_reply":"2024-12-04T14:24:25.601995Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def TrainML(model_class, test_data): #(Hàm huấn luyện mô hình)\n    X = train.drop('sii', axis = 1)\n    y = train['sii']\n    \n    SKF = StratifiedKFold(n_splits = n_splits, shuffle = True, random_state = seed)\n    \n    train_Score = []\n    test_Score = []\n    oof_non_rounded = np.zeros(len(y), dtype = float)\n    oof_rounded = np.zeros(len(y), dtype = int)\n    test_preds = np.zeros((len(test_data), n_splits))\n    \n    for fold, (train_idx, valid_idx) in enumerate(tqdm(SKF.split(X, y), desc='Training Folds', total = n_splits)):\n        X_train, X_valid = X.iloc[train_idx], X.iloc[valid_idx]\n        y_train, y_valid = y.iloc[train_idx], y.iloc[valid_idx]\n        \n        model = clone(model_class)\n        model.fit(X_train, y_train)\n        \n        y_train_pred = model.predict(X_train)\n        y_valid_pred = model.predict(X_valid)\n        \n        oof_non_rounded[valid_idx] = y_valid_pred\n        y_valid_pred_rounded = y_valid_pred.round(0).astype(int)\n        oof_rounded[valid_idx] = y_valid_pred_rounded\n        \n        train_kappa = QWK(y_train, y_train_pred.round(0).astype(int))\n        valid_kappa = QWK(y_valid, y_valid_pred_rounded)\n        \n        train_Score.append(train_kappa)\n        test_Score.append(valid_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        print(f\"Fold {fold + 1} - Train Kappa: {train_kappa:.6f}, Valid Kappa: {valid_kappa:.6f}\")\n        clear_output(wait = True)\n    \n    print(f'Mean Train Kappa: {np.mean(train_Score):.6f}, Mean Valid Kappa: {np.mean(test_Score):.6f}')\n    print(f'Min Train Kappa: {np.min(train_Score):.6f}, Min Valid Kappa: {np.min(test_Score):.6f}')\n    print(f'Max Train Kappa: {np.max(train_Score):.6f}, Max Valid Kappa: {np.max(test_Score):.6f}')\n    \n    KappaOptimizer = minimize(evaluate_predictions, x0 = [0.5, 1.5, 2.5], args = (y, oof_non_rounded), method = 'Nelder-Mead')\n    assert KappaOptimizer.success, KappaOptimizer.message\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOptimizer.x)\n    tKappa = QWK(y, oof_tuned)\n    print(f'Tuned Thresholds: {KappaOptimizer.x}, Optimized QWK Score: {tKappa:.6f}')\n    \n    tpm = test_preds.mean(axis = 1)\n    tp_rounded = threshold_Rounder(tpm, KappaOptimizer.x)\n    \n    submission = pd.DataFrame({'id': sample['id'], 'sii': tp_rounded})\n    \n    return submission\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:24:25.604245Z","iopub.execute_input":"2024-12-04T14:24:25.60472Z","iopub.status.idle":"2024-12-04T14:24:25.615454Z","shell.execute_reply.started":"2024-12-04T14:24:25.604675Z","shell.execute_reply":"2024-12-04T14:24:25.614636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LGBParams = { #(Params mô hình LGB)\n    'learning_rate': 0.046,\n    'max_depth': 6,\n    'num_leaves': 70,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.1,  # Increased from 2.68e-06\n    'device': 'cpu',\n}\n\nXGB_Params = { #(Params mô hình XGBoost)\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 70,\n    'subsample': 0.85,\n    'colsample_bytree': 0.84,\n    'reg_alpha': 0.1,  # Increased from 0.1\n    'reg_lambda': 1,  # Increased from 1\n    'random_state': seed,\n    'objective': 'reg:squarederror',\n    'tree_method': 'gpu_hist',\n    'eval_metric': 'rmse',\n}\n\n\nCatBoost_Params = { #(Params mô hình CatBoost)\n    'learning_rate': 0.035,\n    'depth': 6,\n    'iterations': 80,\n    'random_seed': seed,\n    'verbose': 0,\n    'l2_leaf_reg': 10,  # Increase this value\n    'task_type': 'GPU',\n}\n\nLight = LGBMRegressor(**LGBParams, random_state=seed, verbose=-1, n_estimators=75)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\nvoting_model = VotingRegressor(estimators=[ #(Hàm kết hợp 3 mô hình hồi quy trên để cải thiện độ chính xác)\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    #('odt', ODT_Model),\n], weights=[4.0,8.0,4.0])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:24:25.616573Z","iopub.execute_input":"2024-12-04T14:24:25.616865Z","iopub.status.idle":"2024-12-04T14:24:25.631522Z","shell.execute_reply.started":"2024-12-04T14:24:25.616819Z","shell.execute_reply":"2024-12-04T14:24:25.630457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Huấn luyện mô hình\nSubmission1 = TrainML(voting_model, test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:24:25.63272Z","iopub.execute_input":"2024-12-04T14:24:25.633081Z","iopub.status.idle":"2024-12-04T14:24:25.64076Z","shell.execute_reply.started":"2024-12-04T14:24:25.633051Z","shell.execute_reply":"2024-12-04T14:24:25.639994Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Đưa kết quả lần huấn luyện 1 vào file csv\nSubmission1.to_csv('submission.csv', index = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:24:25.642097Z","iopub.execute_input":"2024-12-04T14:24:25.643079Z","iopub.status.idle":"2024-12-04T14:24:25.653565Z","shell.execute_reply.started":"2024-12-04T14:24:25.643032Z","shell.execute_reply":"2024-12-04T14:24:25.652531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train, test, sample = load_data() #(Cell này xử lí dữ liệu lần 2 trước khi cho huấn luyện mô hình)\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:24:25.654631Z","iopub.execute_input":"2024-12-04T14:24:25.654891Z","iopub.status.idle":"2024-12-04T14:24:25.770006Z","shell.execute_reply.started":"2024-12-04T14:24:25.654865Z","shell.execute_reply":"2024-12-04T14:24:25.769114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Thiết lập các params tương tự như bước ở trên\n\nLGBParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01  # Increased from 2.68e-06\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': seed\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': seed,\n    'cat_features': cat_c,\n    'verbose': 0,\n    'l2_leaf_reg': 10  # Increase this value\n}\n\n# Create model instances\nLight = LGBMRegressor(**LGBParams, random_state=seed, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n# TabNet_Model = TabNetWrapper(**TabNet_Params)\n#ODT_Model = ObliqueDecisionTreeRegressor(**ODT_Params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n])\n\n# Huấn luyện mô hình lần 2\nSubmission2 = TrainML(voting_model, test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:24:25.771299Z","iopub.execute_input":"2024-12-04T14:24:25.771589Z","iopub.status.idle":"2024-12-04T14:24:49.730955Z","shell.execute_reply.started":"2024-12-04T14:24:25.771561Z","shell.execute_reply":"2024-12-04T14:24:49.730088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Đưa kết quả lần huấn luyện 2 vào file csv\nSubmission2.to_csv('submission.csv', index = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:24:49.732085Z","iopub.execute_input":"2024-12-04T14:24:49.732341Z","iopub.status.idle":"2024-12-04T14:24:49.739083Z","shell.execute_reply.started":"2024-12-04T14:24:49.732316Z","shell.execute_reply":"2024-12-04T14:24:49.738315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Xử lí dữ liệu lần 3\ntrain, test, sample = load_data()\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T15:06:44.576865Z","iopub.execute_input":"2024-12-04T15:06:44.577726Z","iopub.status.idle":"2024-12-04T15:06:44.678709Z","shell.execute_reply.started":"2024-12-04T15:06:44.577695Z","shell.execute_reply":"2024-12-04T15:06:44.677999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Thiết lập các params\n\nimputer = SimpleImputer(strategy='median')\n\nLGBParams = {\n    'learning_rate': 0.046,\n    'max_depth': 6,\n    'num_leaves': 70,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.1,  # Increased from 2.68e-06\n    'device': 'cpu',\n}\n\nXGB_Params = {\n    'learning_rate': 0.04,\n    'max_depth': 6,\n    'n_estimators': 70,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 14,  # Increased from 1\n    'random_state': seed,\n    'tree_method': 'gpu_hist'\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.035,\n    'depth': 6,\n    'iterations': 80,\n    'random_seed': seed,\n    'verbose': 0,\n    'l2_leaf_reg': 10,  # Increase this value\n    'task_type': 'GPU',\n}\n\nLight = LGBMRegressor(**LGBParams, random_state=seed, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\nensemble = VotingRegressor(estimators=[\n    ('lgb',    Pipeline(steps=[('imputer', imputer), ('regressor', Light)])),\n    ('xgb',    Pipeline(steps=[('imputer', imputer), ('regressor', XGB_Model)])),\n    ('cat',    Pipeline(steps=[('imputer', imputer), ('regressor', CatBoost_Model)])),\n    ('rf',     Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=seed))])),\n    ('gb',     Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=seed))])),\n])\n\n# Huấn luyện mô hình lần 3\nSubmission3 = TrainML(ensemble, test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T15:17:45.779459Z","iopub.execute_input":"2024-12-04T15:17:45.779813Z","iopub.status.idle":"2024-12-04T15:18:07.491355Z","shell.execute_reply.started":"2024-12-04T15:17:45.779786Z","shell.execute_reply":"2024-12-04T15:18:07.490439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Đưa kết quả lần huấn luyện 3 vào file csv\nSubmission3.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T15:08:38.443104Z","iopub.execute_input":"2024-12-04T15:08:38.443465Z","iopub.status.idle":"2024-12-04T15:08:38.448986Z","shell.execute_reply.started":"2024-12-04T15:08:38.443434Z","shell.execute_reply":"2024-12-04T15:08:38.448149Z"}},"outputs":[],"execution_count":null}]}