{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nimport pandas as pd\nfrom sklearn.base import clone\nfrom copy import deepcopy\nimport optuna\nfrom scipy.optimize import minimize\nimport matplotlib.pyplot as plt\nimport missingno as msno\nimport re\nfrom colorama import Fore, Style\n\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nimport lightgbm as lgb\nfrom catboost import CatBoostRegressor, CatBoostClassifier\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nimport xgboost as xgb\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.model_selection import *\nfrom sklearn.metrics import *\n\nSEED = 42\nn_splits = 5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T17:53:32.593752Z","iopub.execute_input":"2024-12-30T17:53:32.594037Z","iopub.status.idle":"2024-12-30T17:53:37.506234Z","shell.execute_reply.started":"2024-12-30T17:53:32.594015Z","shell.execute_reply":"2024-12-30T17:53:37.505350Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# Đọc dữ liệu file CSV\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n# Bỏ cột id\ntrain = train.drop('id',axis=1)\ntest = test.drop('id',axis=1)\n\n# Lấy features - x - all columns (cả categorical + numerical)\nfeatures = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n       'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n       'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n       'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n       'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n       'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n       'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n       'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n       'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n       'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n       'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n       'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n       'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n       'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n       'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n       'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n       'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n       'PreInt_EduHx-computerinternet_hoursday','sii']\n\ntrain = train[features]\ntrain = train.dropna(subset='sii') # bỏ sii - nhãn chính\n\n# features categorical\ncat_c = ['Basic_Demos-Enroll_Season','CGAS-Season','Physical-Season','Fitness_Endurance-Season','FGC-Season',\n 'BIA-Season','PAQ_A-Season','PAQ_C-Season','SDS-Season','PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c : \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n        \n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\n    \nfor col in cat_c:\n    all_values = pd.concat([train[col], test[col]]).unique()\n    mapping = {value: idx for idx, value in enumerate(all_values)}\n\n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mapping).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T17:53:37.507503Z","iopub.execute_input":"2024-12-30T17:53:37.508085Z","iopub.status.idle":"2024-12-30T17:53:37.646750Z","shell.execute_reply.started":"2024-12-30T17:53:37.508059Z","shell.execute_reply":"2024-12-30T17:53:37.645912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ntrain","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T17:53:37.648834Z","iopub.execute_input":"2024-12-30T17:53:37.649179Z","iopub.status.idle":"2024-12-30T17:53:37.720718Z","shell.execute_reply.started":"2024-12-30T17:53:37.649155Z","shell.execute_reply":"2024-12-30T17:53:37.719854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ntrain.hist(figsize=(15, 10), bins=20, xlabelsize=8, ylabelsize=8)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T17:53:37.722001Z","iopub.execute_input":"2024-12-30T17:53:37.722324Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[\"sii\"].hist()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_percent_train = train.isnull().mean() * 100\nmissing_percent_train\n\nmissing_percent_test = test.isnull().mean() * 100\nmissing_percent_test","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def nhân 2 feature có tính tương tác \ndef create_interaction_features(df, feature_pairs):\n    for feature1, feature2 in feature_pairs:\n        new_feature_name = f\"{feature1}_x_{feature2}\"\n        df[new_feature_name] = df[feature1] * df[feature2]\n    return df\n\n# Các cặp feature tương tác với nhau\nfeature_pairs = [\n    ('PreInt_EduHx-computerinternet_hoursday', 'Basic_Demos-Age'),\n    ('Basic_Demos-Age', 'SDS-SDS_Total_T'),\n    ('FGC-FGC_SRR_Zone', 'SDS-SDS_Total_T'),\n    ('BIA-BIA_BMC', 'Physical-HeartRate'),\n    ('Fitness_Endurance-Season', 'Physical-Waist_Circumference'),\n    ('BIA-BIA_Fat', 'Physical-BMI'),\n    ('PreInt_EduHx-Season', 'Fitness_Endurance-Season'),\n    ('SDS-SDS_Total_T', 'Physical-Systolic_BP'),\n    ('Basic_Demos-Sex', 'FGC-FGC_PU_Zone')\n]\n\ntrain = create_interaction_features(train, feature_pairs)\ntest = create_interaction_features(test, feature_pairs)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# *HISTOGRAMS*","metadata":{}},{"cell_type":"code","source":"# Important features \n\nX = train.drop(['sii'], axis=1)\ny = train['sii']\n\n# Dùng model XGBoost để phân tích độ quan trọng các feature \nXGBoost = xgb.XGBRegressor(random_state=SEED)\nXGBoost.fit(X, y)\n\nimportance = XGBoost.feature_importances_\n\nfeatures = X.columns\n\nimportance_df = pd.DataFrame({'Feature': features, 'Importance': importance})\nimportance_df = importance_df.sort_values(by='Importance', ascending=False)\n\nplt.figure(figsize=(10, 20))\nplt.barh(importance_df['Feature'], importance_df['Importance'])\nplt.xlabel('Feature Importance')\nplt.ylabel('Features')\nplt.title('Feature Importance in XGBoost')\nplt.gca().invert_yaxis() \nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# *MODEL*","metadata":{}},{"cell_type":"code","source":"%%time\n# def tính Quadratic Weighted Kappa (QWK)\ndef qwk(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n# Phân đoạn nhãn - làm tròn \ndef round_preds(preds, thresholds):\n    return np.where(preds < thresholds[0], 0,\n                    np.where(preds < thresholds[1], 1,\n                             np.where(preds < thresholds[2], 2, 3)))\n    \n# def tính giá trị QWK âm để tối ưu hóa\ndef evaluate_preds(thresholds, y_true, oof_preds):\n    rounded_p = round_preds(oof_preds, thresholds)\n    return -qwk(y_true, rounded_p)\n\n\ndef TrainModel(model, test_data):\n    \n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n    \n    # Cross-Validation (5-Folds)\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_preds = np.zeros(len(y), dtype=float) \n    oof_labels = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    # Train model từng fold\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_preds[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_labels[test_idx] = y_val_pred_rounded\n\n        train_kappa = qwk(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = qwk(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n\n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    # Tối ưu hóa QWK\n    optim = minimize(evaluate_preds,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_preds), \n                              method='Nelder-Mead') \n    assert optim.success, \"Optimization did not converge.\"\n    \n    oof_tuned = round_preds(oof_preds, optim.x)\n    qwk_train = qwk(y, oof_tuned)\n    \n    # QWK train theo fold\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {qwk_train:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = round_preds(tpm, optim.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n    \n    # Trả về file submission và QWK tối ưu trên train\n    return submission, qwk_train","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# Train model lấy best param cho từng model\ndef objective(trial, model_name):\n    if model_name == 'xgb':\n        param = {\n            'learning_rate': trial.suggest_loguniform('learning_rate', 1e-5, 1e-1),\n            'max_depth': trial.suggest_int('max_depth', 3, 10),\n            'n_estimators': trial.suggest_int('n_estimators', 100, 1000),\n            'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n        }\n        model = XGBRegressor(**param, random_state=SEED)\n\n    elif model_name == 'lgbm':\n        param = {\n            'num_leaves': trial.suggest_int('num_leaves', 31, 256),\n            'learning_rate': trial.suggest_loguniform('learning_rate', 1e-5, 1e-1),\n            'n_estimators': trial.suggest_int('n_estimators', 100, 1000),\n            'max_depth': trial.suggest_int('max_depth', -1, 15),\n            'min_child_weight': trial.suggest_loguniform('min_child_weight', 1e-3, 10.0),\n            'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n            'reg_alpha': trial.suggest_loguniform('reg_alpha', 1e-3, 10.0),\n            'reg_lambda': trial.suggest_loguniform('reg_lambda', 1e-3, 10.0),\n        }\n        model = lgb.LGBMRegressor(**param,random_state=SEED, verbose=-1)\n\n    elif model_name == 'catboost':\n        param = {\n            'iterations': trial.suggest_int('iterations', 100, 1000),\n            'learning_rate': trial.suggest_loguniform('learning_rate', 1e-5, 1e-1),\n            'depth': trial.suggest_int('depth', 4, 10),\n            'l2_leaf_reg': trial.suggest_loguniform('l2_leaf_reg', 1e-3, 10.0),\n            'subsample': trial.suggest_uniform('subsample', 0.5, 1.0),\n            'random_strength': trial.suggest_uniform('random_strength', 1e-9, 10),\n            'bagging_temperature': trial.suggest_loguniform('bagging_temperature', 0.01, 10.0),\n            'border_count': trial.suggest_int('border_count', 1, 255)\n        }\n        model = CatBoostRegressor(**param, verbose=0, random_seed=SEED)\n\n    submission, score = TrainModel(model, test)\n\n    return score  \n\ndef optimize_with_optuna(model_name, n_trials=100):\n    study = optuna.create_study(direction=\"maximize\")  \n    study.optimize(lambda trial: objective(trial, model_name), n_trials=n_trials)\n\n    print(f\"Best params for {model_name}: {study.best_params}\")\n    print(f\"Best QWK score for {model_name}: {study.best_value}\")\n\n    return study","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nstudy_xgb = optimize_with_optuna(model_name='xgb', n_trials=200)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nstudy_lgbm = optimize_with_optuna(model_name='lgbm', n_trials=200)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nstudy_catboost = optimize_with_optuna(model_name='catboost', n_trials=200)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_params_lgbm = {'num_leaves': 60, 'learning_rate': 0.011747572224219955, 'n_estimators': 993, 'max_depth': 1, 'min_child_weight': 0.0025036281384857462, 'subsample': 0.8252622287203014, 'colsample_bytree': 0.6648896193058901, 'reg_alpha': 0.7153672744430527, 'reg_lambda': 0.12158717311465662}\nbest_params_xgb = {'learning_rate': 0.007356059931165658, 'max_depth': 3, 'n_estimators': 957, 'subsample': 0.6555266544650088, 'colsample_bytree': 0.7712019245727745}\nbest_params_catboost = {'iterations': 804, 'learning_rate': 0.007849710402582562, 'depth': 6, 'l2_leaf_reg': 7.31183636902306, 'subsample': 0.5630297785016092, 'random_strength': 1.7097065892440113, 'bagging_temperature': 0.026593521316435192, 'border_count': 12}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# LightGBM\nLight = lgb.LGBMRegressor(**best_params_lgbm, random_state=SEED, verbose=-1)\nSubmission_LGBM, k_lgbm = TrainModel(Light, test)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# XGBoost\nXGBoost = xgb.XGBRegressor(**best_params_xgb, random_state=SEED)\nSubmission_XGB, k_xgb = TrainModel(XGBoost, test)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CatBoost\nCatBoost = CatBoostRegressor(**best_params_catboost, random_state=SEED, verbose=0)\nSubmission_CatBoost , k_cat= TrainModel(CatBoost, test)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# *SUBMIT TO CSV*","metadata":{}},{"cell_type":"code","source":"# Kết hợp 3 model lấy best score\n\ntotal_k = k_cat + k_xgb + k_lgbm\n\nweight_cat = k_cat / total_k    # CatBoost \nweight_xgb = k_xgb / total_k    # XGBoost \nweight_lgbm = k_lgbm / total_k  # LightGBM \n\nensemble_df = pd.DataFrame({\n    'id': Submission_LGBM['id'],\n    'cat': Submission_CatBoost[\"sii\"],\n    'xgb': Submission_XGB[\"sii\"],\n    'lgbm': Submission_LGBM[\"sii\"]\n})\n\nmelted = ensemble_df.melt(id_vars='id', value_vars=['cat', 'xgb', 'lgbm'], \n                          var_name='model', value_name='sii')\n\n\nmelted['weight'] = melted['model'].map({\n    'cat': weight_cat,\n    'xgb': weight_xgb,\n    'lgbm': weight_lgbm\n})\n\ngrouped = melted.groupby(['id', 'sii'])['weight'].sum().reset_index()\n\nbest_submission = grouped.loc[grouped.groupby('id')['weight'].idxmax()][['id', 'sii']]\n\n\ncomparison_df = best_submission.merge(ensemble_df, on='id', how='left')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"comparison_df","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}