{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30805,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# import model và thư viện\nimport torch\nimport numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone, BaseEstimator, RegressorMixin\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score, mean_squared_error, mean_absolute_error, mean_absolute_percentage_error\nfrom sklearn.model_selection import StratifiedKFold, train_test_split, KFold\nfrom scipy.optimize import minimize\nfrom sklearn.decomposition import PCA\nfrom concurrent.futures import ThreadPoolExecutor\nimport random\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\nfrom torch.utils.data import Dataset,DataLoader\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, RobustScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nfrom matplotlib.ticker import MaxNLocator\nimport seaborn as sns\n\n\nwarnings.filterwarnings('ignore')\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n# pd.options.display.max_columns = None\n\n#seed: sau mỗi loop giá trị ko đổi\n#n_splits: phân chia data ra 4 phần\nseed = 42\nn_splits = 4","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:38:10.895067Z","iopub.status.busy":"2024-12-05T04:38:10.894828Z","iopub.status.idle":"2024-12-05T04:38:29.680996Z","shell.execute_reply":"2024-12-05T04:38:29.680227Z","shell.execute_reply.started":"2024-12-05T04:38:10.895041Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\n# Đảm bảo dữ liệu ko đổi sau các vòng lặp\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\nseed_everything(2024)","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:38:29.683126Z","iopub.status.busy":"2024-12-05T04:38:29.682463Z","iopub.status.idle":"2024-12-05T04:38:29.692224Z","shell.execute_reply":"2024-12-05T04:38:29.691593Z","shell.execute_reply.started":"2024-12-05T04:38:29.683098Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_data():\n    train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n    test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n    sample =  pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n    return train, test, sample","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:38:29.693564Z","iopub.status.busy":"2024-12-05T04:38:29.693299Z","iopub.status.idle":"2024-12-05T04:38:29.709245Z","shell.execute_reply":"2024-12-05T04:38:29.708647Z","shell.execute_reply.started":"2024-12-05T04:38:29.693538Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train, test, sample = load_data()","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:38:29.711850Z","iopub.status.busy":"2024-12-05T04:38:29.711525Z","iopub.status.idle":"2024-12-05T04:38:29.789210Z","shell.execute_reply":"2024-12-05T04:38:29.788322Z","shell.execute_reply.started":"2024-12-05T04:38:29.711812Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":" # Xử lý dữ liệu thiếu\nimputer = KNNImputer(n_neighbors=5)\n\n# Numeric columns\nnumeric_cols = train.select_dtypes(include = ['int32', 'int64', 'float32', 'float64']).columns\n\n# Lấy vài numeric columns từ tập train, bắt đầu lấp đầy các dữ liệu thiếu = giá trị trung bình \n# hoặc trung vị hoặc giá trị xuất hiện nhiều nhất\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\n\n# Thêm non-numeric columns vào tệp train\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\ntrain = train_imputed","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:38:29.790697Z","iopub.status.busy":"2024-12-05T04:38:29.790347Z","iopub.status.idle":"2024-12-05T04:38:36.490131Z","shell.execute_reply":"2024-12-05T04:38:36.489462Z","shell.execute_reply.started":"2024-12-05T04:38:29.790642Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Phân tích, loại bỏ các cột ko cần thiết\ndef feature_engineering(df: pd.DataFrame) -> pd.DataFrame: \n    season_cols = [col for col in df.columns if 'season' in col.lower()]\n    df = df.drop(columns=season_cols, axis = 1)\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    df['BMI_PHR'] = df['Physical-BMI'] * df['Physical-HeartRate']\n    \n    return df","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:38:36.491531Z","iopub.status.busy":"2024-12-05T04:38:36.491170Z","iopub.status.idle":"2024-12-05T04:38:36.498663Z","shell.execute_reply":"2024-12-05T04:38:36.497804Z","shell.execute_reply.started":"2024-12-05T04:38:36.491494Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Chạy hàm loại bỏ các cột ko cần thiết\ndef preprocess_tabular_data(df: pd.DataFrame, test: bool = False) -> pd.DataFrame:\n    df = feature_engineering(df)\n    df = df.dropna(thresh = 10) # Drop rows with less than 10 giá trị ko rỗng\n    return df","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:38:36.499861Z","iopub.status.busy":"2024-12-05T04:38:36.499623Z","iopub.status.idle":"2024-12-05T04:38:36.515004Z","shell.execute_reply":"2024-12-05T04:38:36.514270Z","shell.execute_reply.started":"2024-12-05T04:38:36.499838Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cập nhật 2 tệp train và test\ntrain = preprocess_tabular_data(train)\ntest = feature_engineering(test)","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:38:36.516264Z","iopub.status.busy":"2024-12-05T04:38:36.516020Z","iopub.status.idle":"2024-12-05T04:38:36.548862Z","shell.execute_reply":"2024-12-05T04:38:36.548253Z","shell.execute_reply.started":"2024-12-05T04:38:36.516240Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.drop('id', axis = 1)\ntest.drop('id', axis = 1)\ntrain.head()","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:38:36.549946Z","iopub.status.busy":"2024-12-05T04:38:36.549704Z","iopub.status.idle":"2024-12-05T04:38:36.574910Z","shell.execute_reply":"2024-12-05T04:38:36.574201Z","shell.execute_reply.started":"2024-12-05T04:38:36.549922Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Xử lý các đặc trưng theo cột trong tập train\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW', 'BMI_PHR']\n\nlen(featuresCols)\ntrain = train[featuresCols]\n# Bỏ dòng có giá trị bị thiếu (sii)\ntrain = train.dropna(subset = 'sii')","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:38:36.577648Z","iopub.status.busy":"2024-12-05T04:38:36.577405Z","iopub.status.idle":"2024-12-05T04:38:36.586003Z","shell.execute_reply":"2024-12-05T04:38:36.585192Z","shell.execute_reply.started":"2024-12-05T04:38:36.577624Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# xuất tập train ra kiểm tra thử\nlen(train)","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:38:36.587258Z","iopub.status.busy":"2024-12-05T04:38:36.587034Z","iopub.status.idle":"2024-12-05T04:38:36.603762Z","shell.execute_reply":"2024-12-05T04:38:36.602911Z","shell.execute_reply.started":"2024-12-05T04:38:36.587235Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:38:36.605190Z","iopub.status.busy":"2024-12-05T04:38:36.604873Z","iopub.status.idle":"2024-12-05T04:38:36.631170Z","shell.execute_reply":"2024-12-05T04:38:36.630444Z","shell.execute_reply.started":"2024-12-05T04:38:36.605152Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Kiểm tra có giá trị vượt ngưỡng biểu diện đc (aka vô hạn) ko, nếu có -> NaN\nif np.any(np.isinf(train)):\n    train = train.replace([np.inf, -np.inf], np.nan)","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:38:36.632439Z","iopub.status.busy":"2024-12-05T04:38:36.632169Z","iopub.status.idle":"2024-12-05T04:38:36.642738Z","shell.execute_reply":"2024-12-05T04:38:36.641843Z","shell.execute_reply.started":"2024-12-05T04:38:36.632399Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Hàm tính độ tương đồng giữa 2 tập dữ liệu\ndef QWK(y_true, y_pred): \n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n# Hàm chuyển đổi giá trị liên tục -> các lớp phân loại\ndef threshold_Rounder(oof_non_rounded, thresholds): \n    oof_non_rounded = np.array(oof_non_rounded)\n    \n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\n# Tính giá trị QWK của dự đoán sau khi làm tròn dựa trên các ngưỡng\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -QWK(y_true, rounded_p)","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:38:36.644162Z","iopub.status.busy":"2024-12-05T04:38:36.643826Z","iopub.status.idle":"2024-12-05T04:38:36.652210Z","shell.execute_reply":"2024-12-05T04:38:36.651447Z","shell.execute_reply.started":"2024-12-05T04:38:36.644126Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = test.drop('id', axis = 1)","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:38:36.653410Z","iopub.status.busy":"2024-12-05T04:38:36.653130Z","iopub.status.idle":"2024-12-05T04:38:36.666940Z","shell.execute_reply":"2024-12-05T04:38:36.666096Z","shell.execute_reply.started":"2024-12-05T04:38:36.653386Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Hàm huấn luyện mô hình\ndef TrainML(model_class, test_data): #(Hàm huấn luyện mô hình)\n    X = train.drop('sii', axis = 1)\n    y = train['sii']\n    \n    SKF = StratifiedKFold(n_splits = n_splits, shuffle = True, random_state = seed)\n    \n    train_Score = []\n    test_Score = []\n    oof_non_rounded = np.zeros(len(y), dtype = float)\n    oof_rounded = np.zeros(len(y), dtype = int)\n    test_preds = np.zeros((len(test_data), n_splits))\n    \n    for fold, (train_idx, valid_idx) in enumerate(tqdm(SKF.split(X, y), desc='Training Folds', total = n_splits)):\n        X_train, X_valid = X.iloc[train_idx], X.iloc[valid_idx]\n        y_train, y_valid = y.iloc[train_idx], y.iloc[valid_idx]\n        \n        model = clone(model_class)\n        model.fit(X_train, y_train)\n        \n        y_train_pred = model.predict(X_train)\n        y_valid_pred = model.predict(X_valid)\n        \n        oof_non_rounded[valid_idx] = y_valid_pred\n        y_valid_pred_rounded = y_valid_pred.round(0).astype(int)\n        oof_rounded[valid_idx] = y_valid_pred_rounded\n        \n        train_kappa = QWK(y_train, y_train_pred.round(0).astype(int))\n        valid_kappa = QWK(y_valid, y_valid_pred_rounded)\n        \n        train_Score.append(train_kappa)\n        test_Score.append(valid_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        print(f\"Fold {fold + 1} - Train Kappa: {train_kappa:.6f}, Valid Kappa: {valid_kappa:.6f}\")\n        clear_output(wait = True)\n    \n    print(f'Mean Train Kappa: {np.mean(train_Score):.6f}, Mean Valid Kappa: {np.mean(test_Score):.6f}')\n    print(f'Min Train Kappa: {np.min(train_Score):.6f}, Min Valid Kappa: {np.min(test_Score):.6f}')\n    print(f'Max Train Kappa: {np.max(train_Score):.6f}, Max Valid Kappa: {np.max(test_Score):.6f}')\n    \n    KappaOptimizer = minimize(evaluate_predictions, x0 = [0.5, 1.5, 2.5], args = (y, oof_non_rounded), method = 'Nelder-Mead')\n    assert KappaOptimizer.success, KappaOptimizer.message\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOptimizer.x)\n    tKappa = QWK(y, oof_tuned)\n    print(f'Tuned Thresholds: {KappaOptimizer.x}, Optimized QWK Score: {tKappa:.6f}')\n    \n    tpm = test_preds.mean(axis = 1)\n    tp_rounded = threshold_Rounder(tpm, KappaOptimizer.x)\n    \n    submission = pd.DataFrame({'id': sample['id'], 'sii': tp_rounded})\n    \n    return submission\n    ","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:38:36.668189Z","iopub.status.busy":"2024-12-05T04:38:36.667938Z","iopub.status.idle":"2024-12-05T04:38:36.680233Z","shell.execute_reply":"2024-12-05T04:38:36.679533Z","shell.execute_reply.started":"2024-12-05T04:38:36.668165Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Chủ yếu là thay đổi Params của 3 models so với bản gốc (#|value| là giá trị ban đầu)**","metadata":{}},{"cell_type":"code","source":"# Params mô hình LGB\nLGBParams = {\n    'learning_rate': 0.06,          #0.046\n    'max_depth': 6,\n    'num_leaves': 70,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.693,      #0.893\n    'bagging_fraction': 0.784,      #0.784\n    'bagging_freq': 4,\n    'lambda_l1': 10,  \n    'lambda_l2': 0.5,               #0.1\n    'device': 'cpu',\n}\n\n# Params mô hình XGBoost\nXGB_Params = {\n    'learning_rate': 0.05,          #0.04\n    'max_depth': 6,\n    'n_estimators': 70,\n    'subsample': 0.6,               #0.8\n    'colsample_bytree': 0.7,        #0.8\n    'reg_alpha': 1, \n    'reg_lambda': 14,\n    'random_state': seed,\n    'tree_method': 'gpu_hist'\n}\n\n# Params mô hình CatBoost\nCatBoost_Params = { \n    'learning_rate': 0.055,         #0.035\n    'depth': 6,\n    'iterations': 80,\n    'random_seed': seed,\n    'verbose': 0,\n    'l2_leaf_reg': 10,  \n    'task_type': 'GPU',\n}\n\nLight = LGBMRegressor(**LGBParams, random_state=seed, verbose=-1, n_estimators=75)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n# Hàm kết hợp 3 mô hình hồi quy trên để cải thiện độ chính xác\nvoting_model = VotingRegressor(estimators=[ \n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    #('odt', ODT_Model),\n], weights=[4.0,8.0,4.0])\n","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:38:36.681553Z","iopub.status.busy":"2024-12-05T04:38:36.681236Z","iopub.status.idle":"2024-12-05T04:38:36.699440Z","shell.execute_reply":"2024-12-05T04:38:36.698708Z","shell.execute_reply.started":"2024-12-05T04:38:36.681529Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Huấn luyện mô hình\nSubmission1 = TrainML(voting_model, test)","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:38:36.700523Z","iopub.status.busy":"2024-12-05T04:38:36.700245Z","iopub.status.idle":"2024-12-05T04:38:58.461297Z","shell.execute_reply":"2024-12-05T04:38:58.459821Z","shell.execute_reply.started":"2024-12-05T04:38:36.700498Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Đưa kết quả lần huấn luyện 1 vào file csv\nSubmission1.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:38:58.462985Z","iopub.status.busy":"2024-12-05T04:38:58.462462Z","iopub.status.idle":"2024-12-05T04:38:58.468928Z","shell.execute_reply":"2024-12-05T04:38:58.468243Z","shell.execute_reply.started":"2024-12-05T04:38:58.462949Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":" # Xử lí dữ liệu lần 2 trước khi cho huấn luyện mô hình\ntrain, test, sample = load_data()\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:38:58.469970Z","iopub.status.busy":"2024-12-05T04:38:58.469709Z","iopub.status.idle":"2024-12-05T04:38:58.610519Z","shell.execute_reply":"2024-12-05T04:38:58.609603Z","shell.execute_reply.started":"2024-12-05T04:38:58.469939Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Thiết lập các params tương tự như bước ở trên\n\nLGBParams = {\n    'learning_rate': 0.066,          #0.046\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.666,       #0.893\n    'bagging_fraction': 0.666,       #0.784\n    'bagging_freq': 4,\n    'lambda_l1': 10,  \n    'lambda_l2': 0.06                #0.01\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.0666,           #0.05\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.6,                #0.8\n    'colsample_bytree': 0.6,         #0.8\n    'reg_alpha': 1,  \n    'reg_lambda': 5, \n    'random_state': seed\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.06,            #0.05\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': seed,\n    'cat_features': cat_c,\n    'verbose': 0,\n    'l2_leaf_reg': 10 \n}\n\nLight = LGBMRegressor(**LGBParams, random_state=seed, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n# TabNet_Model = TabNetWrapper(**TabNet_Params)\n#ODT_Model = ObliqueDecisionTreeRegressor(**ODT_Params)\n\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n])\n\n# Huấn luyện mô hình lần 2\nSubmission2 = TrainML(voting_model, test)","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:38:58.611955Z","iopub.status.busy":"2024-12-05T04:38:58.611643Z","iopub.status.idle":"2024-12-05T04:39:09.610112Z","shell.execute_reply":"2024-12-05T04:39:09.609316Z","shell.execute_reply.started":"2024-12-05T04:38:58.611920Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Đưa kết quả lần huấn luyện 2 vào file csv\nSubmission2.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:39:09.611269Z","iopub.status.busy":"2024-12-05T04:39:09.611014Z","iopub.status.idle":"2024-12-05T04:39:09.616499Z","shell.execute_reply":"2024-12-05T04:39:09.615631Z","shell.execute_reply.started":"2024-12-05T04:39:09.611243Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Xử lí dữ liệu lần 3\ntrain, test, sample = load_data()\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:39:09.618147Z","iopub.status.busy":"2024-12-05T04:39:09.617815Z","iopub.status.idle":"2024-12-05T04:39:09.717256Z","shell.execute_reply":"2024-12-05T04:39:09.716589Z","shell.execute_reply.started":"2024-12-05T04:39:09.618110Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Thiết lập các params\n\nimputer = SimpleImputer(strategy='median')\n\nLGBParams = {\n    'learning_rate': 0.095,         #0.046\n    'max_depth': 6,\n    'num_leaves': 70,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.543,      #0.893\n    'bagging_fraction': 0.347,      #0.784\n    'bagging_freq': 4,\n    'lambda_l1': 10,\n    'lambda_l2': 0.1,\n    'device': 'cpu',\n}\n\nXGB_Params = {\n    'learning_rate': 0.03,           #0.04\n    'max_depth': 6,\n    'n_estimators': 70,\n    'subsample': 0.4,                #0.8\n    'colsample_bytree': 0.7,         #0.8\n    'reg_alpha': 1,\n    'reg_lambda': 14,\n    'random_state': seed,\n    'tree_method': 'gpu_hist'\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.069,           #0.035\n    'depth': 6,\n    'iterations': 80,\n    'random_seed': seed,\n    'verbose': 0,\n    'l2_leaf_reg': 10,\n    'task_type': 'GPU',\n}\n\nLight = LGBMRegressor(**LGBParams, random_state=seed, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\nensemble = VotingRegressor(estimators=[\n    ('lgb',    Pipeline(steps=[('imputer', imputer), ('regressor', Light)])),\n    ('xgb',    Pipeline(steps=[('imputer', imputer), ('regressor', XGB_Model)])),\n    ('cat',    Pipeline(steps=[('imputer', imputer), ('regressor', CatBoost_Model)])),\n    ('rf',     Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=seed))])),\n    ('gb',     Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=seed))])),\n])\n\n# Huấn luyện mô hình lần 3\nSubmission3 = TrainML(ensemble, test)","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:39:09.719207Z","iopub.status.busy":"2024-12-05T04:39:09.718444Z","iopub.status.idle":"2024-12-05T04:39:30.536655Z","shell.execute_reply":"2024-12-05T04:39:30.535821Z","shell.execute_reply.started":"2024-12-05T04:39:09.719166Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Đưa kết quả lần huấn luyện 3 vào file csv\nSubmission3.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.execute_input":"2024-12-05T04:39:30.538186Z","iopub.status.busy":"2024-12-05T04:39:30.537837Z","iopub.status.idle":"2024-12-05T04:39:30.543430Z","shell.execute_reply":"2024-12-05T04:39:30.542650Z","shell.execute_reply.started":"2024-12-05T04:39:30.538149Z"},"trusted":true},"outputs":[],"execution_count":null}]}