{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:19:38.198646Z","iopub.execute_input":"2024-12-19T22:19:38.199104Z","iopub.status.idle":"2024-12-19T22:19:38.226849Z","shell.execute_reply.started":"2024-12-19T22:19:38.199058Z","shell.execute_reply":"2024-12-19T22:19:38.225734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nfrom sklearn.impute import KNNImputer\nfrom sklearn.model_selection import train_test_split\nimport pandas as pd\n\nimport numpy as np\nimport pandas as pd\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:19:38.228962Z","iopub.execute_input":"2024-12-19T22:19:38.229323Z","iopub.status.idle":"2024-12-19T22:19:41.457715Z","shell.execute_reply.started":"2024-12-19T22:19:38.229282Z","shell.execute_reply":"2024-12-19T22:19:41.456644Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    df['BMI_PHR'] = df['Physical-BMI'] * df['Physical-HeartRate']\n    return df\n\ndef quadratic_weighted_kappa(actual, predicted):\n    \"\"\"\n    Calculate the quadratic weighted kappa (QWK).\n\n    Parameters:\n        actual (array-like): True labels (ground truth)\n        predicted (array-like): Predicted labels\n        num_labels (int): Number of distinct labels (N)\n\n    Returns:\n        float: Quadratic weighted kappa score\n    \"\"\"\n    actual = actual.astype(int)\n    predicted = predicted.astype(int)\n    num_labels = 4\n    # Initialize O (N x N histogram matrix)\n    O = np.zeros((num_labels, num_labels), dtype=np.float64)\n    for a, p in zip(actual, predicted):\n        O[a, p] += 1\n\n    # Initialize W (N x N weight matrix)\n    W = np.zeros((num_labels, num_labels), dtype=np.float64)\n    for i in range(num_labels):\n        for j in range(num_labels):\n            W[i, j] = ((i - j) ** 2) / ((num_labels - 1) ** 2)\n\n    # Calculate the actual and predicted histograms\n    actual_hist = np.sum(O, axis=1)\n    predicted_hist = np.sum(O, axis=0)\n\n    # Compute E (N x N expected matrix)\n    E = np.outer(actual_hist, predicted_hist) / np.sum(O)\n\n    # Calculate QWK\n    numerator = np.sum(W * O)\n    denominator = np.sum(W * E)\n    kappa = 1 - (numerator / denominator)\n\n    return kappa\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:19:41.459006Z","iopub.execute_input":"2024-12-19T22:19:41.459747Z","iopub.status.idle":"2024-12-19T22:19:41.472501Z","shell.execute_reply.started":"2024-12-19T22:19:41.459701Z","shell.execute_reply":"2024-12-19T22:19:41.471401Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def prep_df(dataframe, index):\n        ds = []\n        for x in ['enmo', 'light']:\n                keys = dataframe[x].describe().keys().tolist()\n                keys = [*keys, 'kurt', 'skew']\n                values = dataframe[x].describe().values\n                kurt = dataframe[x].kurt()\n                skew = dataframe[x].skew()\n                values = [*values, kurt, skew]\n                d = dict(zip([x+'_'+el.replace('%', '') for el in keys if 'min' not in el], values))\n                ds.append(d)\n        ds[0].update(ds[1])\n\n        dataframe = pd.DataFrame.from_dict({index: ds[0]}, orient='index')\n        #dataframe = dataframe.drop(columns=['enmo_min', 'light_min'])\n        return dataframe\n\npath = '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/'\ndfs = []\nfor el in tqdm(os.listdir(path)):\n    pq = pd.read_parquet(path+el)\n    index = el[3:]\n    df = prep_df(pq, index)\n    dfs.append(df)\ndf_acc_train = pd.concat(dfs)\ndf_acc_train = df_acc_train.reset_index().rename(columns={'index':'id'})\n\npath = '/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet/'\ndfs = []\nfor el in tqdm(os.listdir(path)):\n    pq = pd.read_parquet(path+el)\n    index = el[3:]\n    df = prep_df(pq, index)\n    dfs.append(df)\ndf_acc_test = pd.concat(dfs)\ndf_acc_test = df_acc_test.reset_index().rename(columns={'index':'id'})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:19:41.475318Z","iopub.execute_input":"2024-12-19T22:19:41.475830Z","iopub.status.idle":"2024-12-19T22:22:00.379392Z","shell.execute_reply.started":"2024-12-19T22:19:41.475779Z","shell.execute_reply":"2024-12-19T22:22:00.378027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ndf_test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:22:00.381386Z","iopub.execute_input":"2024-12-19T22:22:00.381881Z","iopub.status.idle":"2024-12-19T22:22:00.489709Z","shell.execute_reply.started":"2024-12-19T22:22:00.381820Z","shell.execute_reply":"2024-12-19T22:22:00.488319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test_raw = pd.merge(df_test, df_acc_test, on='id', how='outer')\ndf_train_raw = pd.merge(df_train, df_acc_train, on='id', how='outer')\ndf_train_raw['Physical-Weight'] = df_train_raw['Physical-Weight'].replace(0, np.nan)\ndf_test_raw['Physical-Weight'] = df_test_raw['Physical-Weight'].replace(0, np.nan)\ndf_train_raw = df_train_raw.drop(columns=['Physical-Waist_Circumference', 'PAQ_A-PAQ_A_Total'])\ndf_test_raw = df_test_raw.drop(columns=['Physical-Waist_Circumference', 'PAQ_A-PAQ_A_Total'])\nprint(df_train_raw.shape, df_test_raw.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:22:00.491420Z","iopub.execute_input":"2024-12-19T22:22:00.491805Z","iopub.status.idle":"2024-12-19T22:22:00.539800Z","shell.execute_reply.started":"2024-12-19T22:22:00.491769Z","shell.execute_reply":"2024-12-19T22:22:00.538699Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"matching_columns = [el for el in df_test_raw.columns if el in df_train_raw.columns]\ndf_train_raw = df_train_raw[[*matching_columns, 'sii']]\ndf_test_raw = df_test_raw[matching_columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:22:00.541479Z","iopub.execute_input":"2024-12-19T22:22:00.542881Z","iopub.status.idle":"2024-12-19T22:22:00.559999Z","shell.execute_reply.started":"2024-12-19T22:22:00.542826Z","shell.execute_reply":"2024-12-19T22:22:00.558613Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"seasons = [el for el in df_train_raw.columns if 'Season' in el]\nfor col in seasons:\n    df_train_raw[col] = df_train_raw[col].map({\n        'Spring':1,\n        'Summer':2,\n        'Fall':3,\n        'Winter':4,\n    })\n\nseasons = [el for el in df_test_raw.columns if 'Season' in el]\nfor col in seasons:\n    df_test_raw[col] = df_test_raw[col].map({\n        'Spring':1,\n        'Summer':2,\n        'Fall':3,\n        'Winter':4,\n    })","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:22:00.561515Z","iopub.execute_input":"2024-12-19T22:22:00.561956Z","iopub.status.idle":"2024-12-19T22:22:00.592029Z","shell.execute_reply.started":"2024-12-19T22:22:00.561907Z","shell.execute_reply":"2024-12-19T22:22:00.590762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\ndef weighted_euclidean(x, y, missing_values=np.nan):\n    \"\"\"Custom distance metric: weighted Euclidean distance.\"\"\"\n    diff = x - y\n    return np.sqrt(np.sum(diff ** 2))\n# TRAIN\nimputer = KNNImputer(\n    n_neighbors=50,\n    weights='distance',\n    metric=weighted_euclidean,\n    )\ntrain_imputed = imputer.fit_transform(df_train_raw.drop(columns=['id']))\n\ndf_train_imputed = pd.DataFrame(train_imputed, columns=df_train_raw.drop(columns=['id']).columns)\ndf_train_imputed = df_train_imputed.reset_index()\ndf_train_imputed['sii'] = df_train_imputed['sii'].astype(int)\n\n# TEST\nimputer = KNNImputer(n_neighbors=5, weights='distance')\ntest_imputed = imputer.fit_transform(df_test_raw.drop(columns=['id']))\n\ndf_test_imputed = pd.DataFrame(test_imputed, columns=df_test_raw.drop(columns=['id']).columns)\ndf_test_imputed = df_test_imputed.reset_index()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:22:00.597104Z","iopub.execute_input":"2024-12-19T22:22:00.597885Z","iopub.status.idle":"2024-12-19T22:23:58.379763Z","shell.execute_reply.started":"2024-12-19T22:22:00.597829Z","shell.execute_reply":"2024-12-19T22:23:58.378162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Add Features\ndf_train_imputed = feature_engineering(df_train_imputed)\ndf_test_imputed = feature_engineering(df_test_imputed)\n\nto_be_integer_cols = ['Basic_Demos-Sex', 'FGC-FGC_CU_Zone',\n'FGC-FGC_PU_Zone', 'FGC-FGC_PU', 'FGC-FGC_SRL_Zone',\n'FGC-FGC_SRR_Zone', 'FGC-FGC_TL_Zone', 'BIA-BIA_Frame_num',\n'PreInt_EduHx-computerinternet_hoursday']\n\ndf_train_imputed[to_be_integer_cols] = df_train_imputed[to_be_integer_cols].astype(int)\ndf_test_imputed[to_be_integer_cols] = df_test_imputed[to_be_integer_cols].astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:23:58.381752Z","iopub.execute_input":"2024-12-19T22:23:58.382214Z","iopub.status.idle":"2024-12-19T22:23:58.419585Z","shell.execute_reply.started":"2024-12-19T22:23:58.382172Z","shell.execute_reply":"2024-12-19T22:23:58.418426Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def common_elements(lists):\n    if not lists:\n        return []  # Return an empty list if the input is empty\n    common_set = set(lists[0])\n    for lst in lists[1:]:\n        common_set &= set(lst)\n    return sorted(common_set)\n\n\nindices_train = []\nfor col in df_train_imputed.columns:\n    mean = df_train_imputed[col].describe()['mean']\n    std = df_train_imputed[col].describe()['std']\n    f = df_train_imputed[col].between(mean-6*std, mean+6*std)\n    indices_train.append(f[f].index.tolist())\n\nindices_train = common_elements(indices_train)\nprint(len(indices_train))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:23:58.421175Z","iopub.execute_input":"2024-12-19T22:23:58.421543Z","iopub.status.idle":"2024-12-19T22:23:58.779585Z","shell.execute_reply.started":"2024-12-19T22:23:58.421508Z","shell.execute_reply":"2024-12-19T22:23:58.778441Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"important_features = [\n    'Internet_Hours_Age',\n    'Physical-Height',\n    'SDS-SDS_Total_T',\n    'BMI_Age',\n    'Basic_Demos-Age']\nnot_important_features = [\n    'enmo_count', 'enmo_25', 'enmo_50', 'enmo_75',\n    'enmo_max', 'enmo_kurt', 'enmo_skew', 'light_count',\n    'light_25', 'light_50', 'light_75', 'light_max',\n    'light_kurt', 'light_skew', 'BIA-BIA_FFM',\n]\ns = [list(df_train_imputed.columns).index(el) for el in important_features] \nt = [list(df_train_imputed.columns).index(el) for el in not_important_features]\nweights = np.ones(df_train_imputed.shape[1])\nweights[s] = 2\nweights[t] = .5\nweights\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:23:58.780872Z","iopub.execute_input":"2024-12-19T22:23:58.781278Z","iopub.status.idle":"2024-12-19T22:23:58.793864Z","shell.execute_reply.started":"2024-12-19T22:23:58.781238Z","shell.execute_reply":"2024-12-19T22:23:58.792627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\ndf_train_imputed[cat_c] = df_train_imputed[cat_c].astype('category')\ndf_test_imputed[cat_c] = df_test_imputed[cat_c].astype('category')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:23:58.795304Z","iopub.execute_input":"2024-12-19T22:23:58.795627Z","iopub.status.idle":"2024-12-19T22:23:58.830433Z","shell.execute_reply.started":"2024-12-19T22:23:58.795598Z","shell.execute_reply":"2024-12-19T22:23:58.829290Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Separate target variable and features\ny = df_train_imputed.loc[indices_train, 'sii'].values\nX = df_train_imputed.drop(columns=['sii']).loc[indices_train]\nX_t = df_test_imputed\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, train_size=0.4, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:23:58.831895Z","iopub.execute_input":"2024-12-19T22:23:58.832216Z","iopub.status.idle":"2024-12-19T22:23:58.857917Z","shell.execute_reply.started":"2024-12-19T22:23:58.832186Z","shell.execute_reply":"2024-12-19T22:23:58.856711Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = X.drop(columns=['index'])\nX_t = X_t.drop(columns=['index'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:23:58.859308Z","iopub.execute_input":"2024-12-19T22:23:58.859654Z","iopub.status.idle":"2024-12-19T22:23:58.873473Z","shell.execute_reply.started":"2024-12-19T22:23:58.859620Z","shell.execute_reply":"2024-12-19T22:23:58.872299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, X)\n    mappingTe = create_mapping(col, X_t)\n    X[col] = X[col].map(mapping).astype(int)\n    X_t[col] = X_t[col].map(mappingTe).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:23:58.875584Z","iopub.execute_input":"2024-12-19T22:23:58.876120Z","iopub.status.idle":"2024-12-19T22:23:58.910076Z","shell.execute_reply.started":"2024-12-19T22:23:58.876008Z","shell.execute_reply":"2024-12-19T22:23:58.908737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Constants\nSEED = 42\nN_SPLITS = 5\nSAMPLE_SUBMISSION_PATH = '/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv'\n\n# Load sample submission\nsample = pd.read_csv(SAMPLE_SUBMISSION_PATH)\n\n# Define utility functions\ndef quadratic_weighted_kappa(y_true, y_pred):\n    \"\"\"Compute the quadratic weighted kappa.\"\"\"\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_rounder(oof_non_rounded, thresholds):\n    \"\"\"Apply threshold rounding based on defined thresholds.\"\"\"\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    \"\"\"Evaluate predictions using the quadratic weighted kappa.\"\"\"\n    rounded_preds = threshold_rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_preds)\n\ndef optimize_thresholds(y_true, oof_non_rounded):\n    \"\"\"Optimize thresholds using Nelder-Mead method.\"\"\"\n    optimizer = minimize(evaluate_predictions, x0=[0.5, 1.49, 2.5], \n                         args=(y_true, oof_non_rounded), method='Nelder-Mead')\n    if not optimizer.success:\n        raise ValueError(\"Threshold optimization did not converge.\")\n    return optimizer.x\n\ndef train_model(model_class, X, y, test_data):\n    \"\"\"Train machine learning model using Stratified K-Fold.\"\"\"\n    skf = StratifiedKFold(n_splits=N_SPLITS, shuffle=True, random_state=SEED)\n\n    oof_non_rounded = np.zeros(len(y))\n    test_preds = np.zeros((len(test_data), N_SPLITS))\n    train_scores, val_scores = [], []\n\n    for fold, (train_idx, val_idx) in enumerate(tqdm(skf.split(X, y), desc=\"Training Folds\", total=N_SPLITS)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n        y_train, y_val = y[train_idx], y[val_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_val_pred = model.predict(X_val)\n        oof_non_rounded[val_idx] = y_val_pred\n\n        train_score = quadratic_weighted_kappa(y_train, model.predict(X_train).round().astype(int))\n        val_score = quadratic_weighted_kappa(y_val, y_val_pred.round().astype(int))\n\n        train_scores.append(train_score)\n        val_scores.append(val_score)\n\n        test_preds[:, fold] = model.predict(test_data)\n\n        print(f\"Fold {fold + 1} - Train QWK: {train_score:.4f}, Validation QWK: {val_score:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK: {np.mean(train_scores):.4f}\")\n    print(f\"Mean Validation QWK: {np.mean(val_scores):.4f}\")\n\n    thresholds = optimize_thresholds(y, oof_non_rounded)\n    final_oof_preds = threshold_rounder(oof_non_rounded, thresholds)\n    final_qwk = quadratic_weighted_kappa(y, final_oof_preds)\n    print(f\"Optimized QWK: {final_qwk:.3f}\")\n\n    fold_weights = np.ones(N_SPLITS)\n    weighted_preds = test_preds.dot(fold_weights) / fold_weights.sum()\n    final_test_preds = threshold_rounder(weighted_preds, thresholds)\n\n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': final_test_preds\n    })\n\n    return submission\n\n\n# Model parameters\nlgb_params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,\n    'lambda_l2': 0.01,\n    'random_state': SEED,\n    'n_estimators': 300,\n    'verbose': -1,\n    'feature_weights':weights\n}\n\nxgb_params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,\n    'reg_lambda': 5,\n    'random_state': SEED,\n    'feature_weights':weights\n}\n\ncatboost_params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10,\n    'feature_weights':weights\n}\n\n# Create model instances\nlgb_model = LGBMRegressor(**lgb_params)\nxgb_model = XGBRegressor(**xgb_params)\ncatboost_model = CatBoostRegressor(**catboost_params)\n\n# Ensemble model\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', lgb_model),\n    ('xgboost', xgb_model),\n    ('catboost', catboost_model)\n])\n\n# Train the ensemble model and generate submission\nsubmission = train_model(voting_model, X, y, X_t)\n\nsubmission.to_csv('submission.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:23:58.911652Z","iopub.execute_input":"2024-12-19T22:23:58.912101Z","iopub.status.idle":"2024-12-19T22:24:25.244724Z","shell.execute_reply.started":"2024-12-19T22:23:58.912056Z","shell.execute_reply":"2024-12-19T22:24:25.243631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:24:25.246082Z","iopub.execute_input":"2024-12-19T22:24:25.246410Z","iopub.status.idle":"2024-12-19T22:24:25.251454Z","shell.execute_reply.started":"2024-12-19T22:24:25.246377Z","shell.execute_reply":"2024-12-19T22:24:25.250197Z"}},"outputs":[],"execution_count":null}]}