{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport re\nimport numpy as np\nimport polars as pl\nimport pandas as pd\nfrom copy import deepcopy\nfrom scipy.optimize import minimize\nfrom scipy.stats import mode\nfrom colorama import Fore, Style\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\nfrom concurrent.futures import ThreadPoolExecutor\nimport seaborn as sns\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.model_selection import RandomizedSearchCV, StratifiedKFold\nfrom sklearn.metrics import make_scorer, cohen_kappa_score\nfrom sklearn.base import clone\nimport matplotlib.pyplot as plt\n\nSEED = 2025\nn_splits = 7","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-07T05:37:43.111679Z","iopub.execute_input":"2025-01-07T05:37:43.112039Z","iopub.status.idle":"2025-01-07T05:37:48.444747Z","shell.execute_reply.started":"2025-01-07T05:37:43.112009Z","shell.execute_reply":"2025-01-07T05:37:48.443576Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1. Process train & test dataset","metadata":{}},{"cell_type":"code","source":"# Processing parquet files is not really matter cause we don't use them anyway..\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"Stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df\n\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\n# merge HBN instruments & actigraphy file\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T05:37:49.410649Z","iopub.execute_input":"2025-01-07T05:37:49.411162Z","iopub.status.idle":"2025-01-07T05:39:20.648593Z","shell.execute_reply.started":"2025-01-07T05:37:49.411114Z","shell.execute_reply":"2025-01-07T05:39:20.647348Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Feature engineering\n","metadata":{}},{"cell_type":"code","source":"def feature_engineering(df):\n    # convert season columns to integer\n    season_cols = [col for col in df.columns if 'Season' in col]\n    cat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n          'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n    \n    def update(df):\n        for c in cat_c: \n            df[c] = df[c].fillna('Missing')\n            df[c] = df[c].astype('category')\n        return df\n            \n    df = update(df)\n    \n    def create_mapping(column, dataset):\n        unique_values = dataset[column].unique()\n        return {value: idx for idx, value in enumerate(unique_values)}\n    \n    for col in cat_c:\n        mapping_df = create_mapping(col, df)\n        \n        df[col] = train[col].replace(mapping_df).astype(int)\n        \n    # assign age group for each participants for normalization\n    def assign_age_group(age):\n        thresholds = [5, 6, 7, 8, 10, 12, 14, 18, 22]\n        for i, j in enumerate(thresholds):\n            if age <= j:\n                return i\n        return np.nan\n    \n    # age groups\n    df[\"age_group\"] = df['Basic_Demos-Age'].apply(assign_age_group)\n\n    df['BMI'] = df[['Physical-BMI', 'BIA-BIA_BMI']].max(axis=1)\n    df['FGC_GS'] =  df[['FGC-FGC_GSND', 'FGC-FGC_GSD']].max(axis=1)\n    df['FGC_SR'] = df[['FGC-FGC_SRL', 'FGC-FGC_SRR']].max(axis=1)\n    \n    features_to_normalize = [\n        'BMI',\n        'FGC_GS',\n        'FGC-FGC_CU', 'FGC-FGC_PU', 'FGC-FGC_TL',\n        'FGC_SR',\n        'BIA-BIA_BMR', 'BIA-BIA_DEE', \n        'BIA-BIA_FFM'\n    ]\n    # Normalized Value= Standard Deviation / Original Value−Mean\n    group_stats = df.groupby('age_group')[features_to_normalize].agg(['mean', 'std']).reset_index()\n\n    group_stats.columns = ['_'.join(col).strip('_') for col in group_stats.columns.values]\n    \n    df = df.merge(group_stats, on='age_group', how='left')\n    \n    for feature in features_to_normalize:\n        df[f\"{feature.split('_')[-1]}_norm\"] = (df[feature] - df[f\"{feature}_mean\"]) / df[f\"{feature}_std\"]\n\n    # Intracellular Water / Extracellular Water rate\n    df[\"ICW_ECW\"] = df[\"BIA-BIA_ECW\"] / df[\"BIA-BIA_ICW\"]\n    # FGC_Zones_mean = sum of FGC_Zone\n    zones = ['FGC-FGC_CU_Zone', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone',\n         'FGC-FGC_PU_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone',\n         'FGC-FGC_TL_Zone']\n    df['FGC_Zones_mean'] = df[zones].sum(axis=1)\n    # Internet_Hours_Age\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    # columns to drops\n    drop_feats = ['FGC-FGC_GSND', 'FGC-FGC_GSD', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone',\n                  'FGC-FGC_PU_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL_Zone',\n                  'Physical-BMI', 'BIA-BIA_BMI', 'FGC-FGC_CU', 'FGC-FGC_PU', 'FGC-FGC_TL', 'FGC-FGC_SRL', 'FGC-FGC_SRR',\n                 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_Frame_num', \"BIA-BIA_FFM\",'age_group']\n    df = df.drop(drop_feats, axis=1) \n\n    return df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T05:37:48.446305Z","iopub.execute_input":"2025-01-07T05:37:48.446993Z","iopub.status.idle":"2025-01-07T05:37:48.462086Z","shell.execute_reply.started":"2025-01-07T05:37:48.446955Z","shell.execute_reply":"2025-01-07T05:37:48.460659Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = feature_engineering(train)\ntest = feature_engineering(test)\n# all the important \nfeaturesCols = [\n                'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC','BIA-BIA_ECW',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW',  'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw','SDS-SDS_Total_T',\n                'FGC_Zones_mean',\n                'GS_norm',\"CU_norm\",\"PU_norm\",\"TL_norm\",\"SR_norm\",\"BMR_norm\",\"DEE_norm\",\"FFM_norm\",\n                \"BMI_norm\",\n                'Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n                'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season','Internet_Hours_Age'\n]\nfeaturesCols += time_series_cols\n\ntrainFeatures = featuresCols + ['sii']\ntrain = train[trainFeatures]\ntrain = train.dropna(subset='sii')\n\ntest = test[featuresCols]\n# drops rows with null sii\ntrain = train[train['sii'].notnull()]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T05:39:20.649727Z","iopub.execute_input":"2025-01-07T05:39:20.650019Z","iopub.status.idle":"2025-01-07T05:39:20.788330Z","shell.execute_reply.started":"2025-01-07T05:39:20.649992Z","shell.execute_reply":"2025-01-07T05:39:20.786983Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Remove shortage features","metadata":{}},{"cell_type":"code","source":"# Show null percentage of features\ndata_trainning = train[train['sii'].notnull()]\ndata_trainning_null = data_trainning.isnull().sum() * 100/len(data_trainning)\ndata_trainning_sorted = data_trainning_null.sort_values()\n\nplt.figure(figsize=(24,5))\nsns.barplot(x=data_trainning_sorted.index, y=data_trainning_sorted.values, color='skyblue')\nplt.title(\"Percentage of missing values in each column\")\nplt.xlabel(\"Columns\")\nplt.xticks(rotation=90)\nplt.ylabel(\"Percentage of missing values\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T05:41:01.700483Z","iopub.execute_input":"2025-01-07T05:41:01.700949Z","iopub.status.idle":"2025-01-07T05:41:03.140642Z","shell.execute_reply.started":"2025-01-07T05:41:01.700916Z","shell.execute_reply":"2025-01-07T05:41:03.139395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# drops columns with greater than 60% NaN values\ncolumns_to_drop = data_trainning_sorted[(data_trainning_sorted > 60)].index\ntrain = train.drop(columns=columns_to_drop)\ntest  = test.drop(columns=columns_to_drop)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T05:42:16.194407Z","iopub.execute_input":"2025-01-07T05:42:16.194813Z","iopub.status.idle":"2025-01-07T05:42:16.406301Z","shell.execute_reply.started":"2025-01-07T05:42:16.194781Z","shell.execute_reply":"2025-01-07T05:42:16.404686Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Scoring & Train functions","metadata":{}},{"cell_type":"code","source":"feature_col = train.drop(['sii'], axis=1).columns\nall_importances = pd.DataFrame({'feature': feature_col})\n# scoring function\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n# round thresholds before scoring\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\n# train function\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    feature_names = X.columns\n\n    lgb_importances_list = []\n    xgb_importances_list = []\n    cat_importances_list = []\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n        named_estimators = model.named_estimators_\n        \n        # LightGBM\n        if 'lightgbm' in named_estimators and hasattr(named_estimators['lightgbm'], 'feature_importances_'):\n            lgb_importances_list.append(named_estimators['lightgbm'].feature_importances_)\n\n        # XGBoost\n        if 'xgboost' in named_estimators and hasattr(named_estimators['xgboost'], 'feature_importances_'):\n            xgb_importances_list.append(named_estimators['xgboost'].feature_importances_)\n\n        # CatBoost\n        if 'catboost' in named_estimators and hasattr(named_estimators['catboost'], 'get_feature_importance'):\n            cat_importances_list.append(named_estimators['catboost'].get_feature_importance())\n\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    # optimize sii thresholds\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    def mean_importances(importances_list):\n        if len(importances_list) > 0:\n            return np.mean(importances_list, axis=0)\n        else:\n            return None\n\n    lgb_mean = mean_importances(lgb_importances_list)\n    xgb_mean = mean_importances(xgb_importances_list)\n    cat_mean = mean_importances(cat_importances_list)\n\n    # plot features importance\n    def normalize_importances(importance_array):\n        if importance_array is not None:\n            return importance_array / importance_array.sum()\n        else:\n            return None\n\n    lgb_mean_normalized = normalize_importances(lgb_mean)\n    xgb_mean_normalized = normalize_importances(xgb_mean)\n    cat_mean_normalized = normalize_importances(cat_mean)\n    \n    if lgb_mean_normalized is not None:\n        lgb_df = pd.DataFrame({'feature': feature_names, 'importance': lgb_mean_normalized}).sort_values('importance', ascending=False)\n        plt.figure(figsize=(10,20))\n        plt.barh(lgb_df['feature'], lgb_df['importance'])\n        plt.gca().invert_yaxis()\n        plt.title(\"LightGBM Feature Importance\")\n        plt.show()\n        all_importances['LightGBM'] = lgb_mean_normalized\n\n    if xgb_mean_normalized is not None:\n        xgb_df = pd.DataFrame({'feature': feature_names, 'importance': xgb_mean_normalized}).sort_values('importance', ascending=False)\n        plt.figure(figsize=(10,20))\n        plt.barh(xgb_df['feature'], xgb_df['importance'])\n        plt.gca().invert_yaxis()\n        plt.title(\"XGBoost Feature Importance\")\n        plt.show()\n        all_importances['XGBoost'] = xgb_mean_normalized\n\n    if cat_mean_normalized is not None:\n        cat_df = pd.DataFrame({'feature': feature_names, 'importance': cat_mean_normalized}).sort_values('importance', ascending=False)\n        plt.figure(figsize=(10,20))\n        plt.barh(cat_df['feature'], cat_df['importance'])\n        plt.gca().invert_yaxis()\n        plt.title(\"CatBoost Feature Importance\")\n        plt.show()\n        all_importances['CatBoost'] = cat_mean_normalized\n\n    return submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T08:36:30.271418Z","iopub.execute_input":"2025-01-06T08:36:30.271846Z","iopub.status.idle":"2025-01-06T08:36:30.294280Z","shell.execute_reply.started":"2025-01-06T08:36:30.271810Z","shell.execute_reply":"2025-01-06T08:36:30.293000Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. Optimize with RandomizedSearchCV","metadata":{}},{"cell_type":"code","source":"# define custom scorer for quadratic weighted kappa\ndef quadratic_weighted_kappa_scorer(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\nkappa_scorer = make_scorer(quadratic_weighted_kappa_scorer, greater_is_better=True)\n\n# parameter grid for LightGBM\nlightgbm_param_grid = {\n    'learning_rate': [0.01, 0.05, 0.1, 0.2],\n    'max_depth': [6, 8, 10, 12,14],\n    'num_leaves': [31, 63, 127, 255],\n    'min_data_in_leaf': [20, 40, 60, 80, 100, 120, 140, 160],\n    'feature_fraction': [0.2, 0.4, 0.8, 1.0],\n    'bagging_fraction': [0.2, 0.4, 0.8, 1.0],\n    'bagging_freq': [1, 3, 5, 7, 9, 11],\n    'lambda_l1': [0, 0.1, 1, 10],\n    'lambda_l2': [0, 0.1, 1, 10],\n    'n_estimators': [100, 200, 300, 400, 500]\n}\n\n# parameter grid for XGBoost\nxgb_param_grid = {\n    'learning_rate': [0.01, 0.05, 0.1, 0.2],\n    'max_depth': [3, 5, 7, 9, 11],\n    'n_estimators': [100, 200, 300, 400, 500],\n    'subsample': [0.6, 0.8, 1.0],\n    'colsample_bytree': [0.6, 0.8, 1.0],\n    'reg_alpha': [0, 0.1, 1, 5, 10],\n    'reg_lambda': [0, 0.1, 1, 5, 10],\n    'gamma': [0, 0.1, 0.5, 1, 2, 5, 10]\n}\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def optimize_xgboost(X, y, param_grid, cv, scoring, n_iter=20, random_state=SEED):\n    xgb = XGBRegressor(random_state=random_state, verbosity=0, tree_method='auto')\n    random_search = RandomizedSearchCV(\n        estimator=xgb,\n        param_distributions=param_grid,\n        n_iter=n_iter,\n        scoring=scoring,\n        cv=cv,\n        verbose=1,\n        random_state=random_state,\n        n_jobs=-1\n    )\n    random_search.fit(X, y)\n    print(f\"Best XGBoost Params: {random_search.best_params_}\")\n    print(f\"Best XGBoost Score: {random_search.best_score_}\")\n    return random_search.best_estimator_\n\ndef optimize_lightgbm(X, y, param_grid, cv, scoring, n_iter=20, random_state=SEED):\n    lgbm = LGBMRegressor(random_state=random_state, verbose=-1)\n    random_search = RandomizedSearchCV(\n        estimator=lgbm,\n        param_distributions=param_grid,\n        n_iter=n_iter,\n        scoring=scoring,\n        cv=cv,\n        verbose=4,\n        random_state=random_state,\n        n_jobs=-1,\n        refit = True\n    )\n    random_search.fit(X, y)\n    print(f\"Best LightGBM Params: {random_search.best_params_}\")\n    print(f\"Best LightGBM Score: {random_search.best_score_}\")\n    return random_search.best_estimator_\n\nX = train.drop(['sii'], axis=1)\ny = train['sii']\n\n# cross-validation strategy\ncv = StratifiedKFold(n_splits=5, shuffle=True, random_state=2025)\n\nbest_lgbm = optimize_lightgbm(X, y, lightgbm_param_grid, cv, kappa_scorer)\nbest_xgb = optimize_xgboost(X, y, xgb_param_grid, cv, kappa_scorer)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 6. Train & Submit prediction","metadata":{}},{"cell_type":"code","source":"# optimized parameters\nLight_Params = {\n'num_leaves': 31, 'n_estimators': 100, 'min_data_in_leaf': 40, 'max_depth': 14, 'learning_rate': 0.05, 'lambda_l2': 1, 'lambda_l1': 0, 'feature_fraction': 0.8, 'bagging_freq': 7, 'bagging_fraction': 1.0,\n'random_state' :SEED,\n               }\nXGB_Params = {\n'subsample': 0.8, 'reg_lambda': 1, 'reg_alpha': 0, 'n_estimators': 200, 'max_depth': 11, 'learning_rate': 0.4, 'gamma': 0.5, 'colsample_bytree': 1.0,    'random_state': SEED,\n    'tree_method': 'auto',\n}\nLight = LGBMRegressor(**Light_Params, verbose=-1)\nXGB_Model = XGBRegressor(**XGB_Params)\nvoting_model = VotingRegressor(estimators=[\n   ('lightgbm', Light),\n   ('xgboost', XGB_Model)\n])\n\nsubmission = TrainML(voting_model, test)\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T11:04:45.117606Z","iopub.execute_input":"2025-01-06T11:04:45.118061Z","iopub.status.idle":"2025-01-06T11:04:59.462544Z","shell.execute_reply.started":"2025-01-06T11:04:45.118027Z","shell.execute_reply":"2025-01-06T11:04:59.461518Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}