{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-12T04:36:38.342978Z","iopub.execute_input":"2024-12-12T04:36:38.343688Z","iopub.status.idle":"2024-12-12T04:36:38.348324Z","shell.execute_reply.started":"2024-12-12T04:36:38.343653Z","shell.execute_reply":"2024-12-12T04:36:38.347461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nfrom sklearn.utils.class_weight import compute_sample_weight","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T04:36:38.349788Z","iopub.execute_input":"2024-12-12T04:36:38.350030Z","iopub.status.idle":"2024-12-12T04:36:38.366218Z","shell.execute_reply.started":"2024-12-12T04:36:38.350006Z","shell.execute_reply":"2024-12-12T04:36:38.365569Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 42\nn_splits = 5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T04:36:38.367557Z","iopub.execute_input":"2024-12-12T04:36:38.367848Z","iopub.status.idle":"2024-12-12T04:36:38.379904Z","shell.execute_reply.started":"2024-12-12T04:36:38.367811Z","shell.execute_reply":"2024-12-12T04:36:38.379255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T04:36:38.380760Z","iopub.execute_input":"2024-12-12T04:36:38.380992Z","iopub.status.idle":"2024-12-12T04:36:38.427034Z","shell.execute_reply.started":"2024-12-12T04:36:38.380969Z","shell.execute_reply":"2024-12-12T04:36:38.426355Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\n\nclass AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.ReLU()\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.ReLU(),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.ReLU(),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n        \n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\n\n\ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    \n    data_tensor = torch.FloatTensor(df_scaled)\n    \n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    \n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n    \n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n            \n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n                 \n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n        \n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    \n    return df_encoded\n\ndef feature_engineering(df):\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1) \n    #jerry's features\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Height_Weight'] = df['Physical-Height'] * df['Physical-Weight']\n    df['Age_Weight'] = df['Basic_Demos-Age'] * df['Physical-Weight']\n    df['Waist_Weight'] = df['Physical-Waist_Circumference'] * df['Physical-Weight']\n    df['BMI_Weight'] = df['Physical-BMI'] * df['Physical-Weight']\n    df['BIA_BMI_Weight'] = df['BIA-BIA_BMI'] * df['Physical-Weight']\n    df['BIA_Fat_Waist'] = df['BIA-BIA_Fat'] * df['Physical-Waist_Circumference']\n    df['BIA_BMI_Waist'] = df['BIA-BIA_BMI'] * df['Physical-Waist_Circumference']\n    df['BMI_Waist'] = df['Physical-BMI'] * df['Physical-Waist_Circumference']\n    df['BIA_FMI_Waist'] = df['BIA-BIA_FMI'] * df['Physical-Waist_Circumference']\n    df['BIA_LST_Waist'] = df['BIA-BIA_LST'] * df['Physical-Waist_Circumference']\n    df['BIA_FFM_Waist'] = df['BIA-BIA_FFM'] * df['Physical-Waist_Circumference']\n    df['BIA_BMR_Waist'] = df['BIA-BIA_BMR'] * df['Physical-Waist_Circumference']\n    df['BIA_FFMI_Waist'] = df['BIA-BIA_FFMI'] * df['Physical-Waist_Circumference']\n    df['BIA_TBW_Waist'] = df['BIA-BIA_TBW'] * df['Physical-Waist_Circumference']\n    df['BIA_ECW_Waist'] = df['BIA-BIA_ECW'] * df['Physical-Waist_Circumference']\n    \n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T04:36:38.429766Z","iopub.execute_input":"2024-12-12T04:36:38.430243Z","iopub.status.idle":"2024-12-12T04:36:38.445788Z","shell.execute_reply.started":"2024-12-12T04:36:38.430214Z","shell.execute_reply":"2024-12-12T04:36:38.444850Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ndf_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n\ntrain_ts_encoded = perform_autoencoder(df_train, encoding_dim=60, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test, encoding_dim=60, epochs=100, batch_size=32)\n\ntime_series_cols = train_ts_encoded.columns.tolist()\ntrain_ts_encoded[\"id\"]=train_ts[\"id\"]\ntest_ts_encoded['id']=test_ts[\"id\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T04:36:38.447023Z","iopub.execute_input":"2024-12-12T04:36:38.447382Z","iopub.status.idle":"2024-12-12T04:37:57.560186Z","shell.execute_reply.started":"2024-12-12T04:36:38.447343Z","shell.execute_reply":"2024-12-12T04:37:57.559278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')\n\nimputer = KNNImputer(n_neighbors=7)\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\nnumeric_cols = [col for col in numeric_cols if 'PCIAT' not in col and col != 'sii']\n\nimputed_train_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_train_data, columns=numeric_cols)\n# train_imputed['sii'] = train_imputed['sii'].round().astype(int)\n\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n        \ntrain = train_imputed\nprint(f\"Size of train before: {len(train)}\")\ntrain = train_imputed[~train['sii'].isna()] # Drop datapoints where SII was imputed\nprint(f\"Size of train after: {len(train)}\")\n\n# For the test set, do the same filtering\nnumeric_cols_test = test.select_dtypes(include=['float64', 'int64']).columns\nnumeric_cols_test = [col for col in numeric_cols_test if 'PCIAT' not in col and col != 'sii']\n\nimputed_train_data = imputer.fit_transform(train[numeric_cols_test])\nimputed_test_data = imputer.transform(test[numeric_cols_test])\ntest_imputed = pd.DataFrame(imputed_test_data, columns=numeric_cols_test)\n\n# Reattach non-numeric and excluded columns\nfor col in test.columns:\n    if col not in numeric_cols_test:\n        test_imputed[col] = test[col]\n        \ntest = test_imputed\n\n# Impute Test Data\nnumeric_cols_test = test.select_dtypes(include=['float64', 'int64']).columns\nimputed_train_data = imputer.fit_transform(train[numeric_cols_test])\nimputed_test_data = imputer.transform(test[numeric_cols_test])\ntest_imputed = pd.DataFrame(imputed_test_data, columns=numeric_cols_test)\nfor col in test.columns:\n    if col not in numeric_cols_test:\n        test_imputed[col] = test[col]\ntest = test_imputed\n\ntrain = feature_engineering(train)\ntrain = train.dropna(thresh=10, axis=0)\ntest = feature_engineering(test)\n\ntrain = train.drop('id', axis=1)\ntest  = test .drop('id', axis=1)   \n\n\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday','sii',\n    \n    # Interaction terms based on correlations:\n    'BMI_Age', 'Height_Weight', 'Age_Weight', 'Waist_Weight', \n    'BMI_Weight', 'BIA_BMI_Weight', 'BIA_Fat_Waist', 'BIA_BMI_Waist', \n    'BMI_Waist', 'BIA_FMI_Waist', 'BIA_LST_Waist', 'BIA_FFM_Waist', \n    'BIA_BMR_Waist', 'BIA_FFMI_Waist', 'BIA_TBW_Waist', 'BIA_ECW_Waist'\n]\n# featuresCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n#                 'CGAS-CGAS_Score', 'Physical-BMI',\n#                 'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n#                 'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n#                 'Fitness_Endurance-Max_Stage',\n#                 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n#                 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n#                 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n#                 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n#                 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n#                 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n#                 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n#                 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n#                 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n#                 'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n#                 'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n#                 'SDS-SDS_Total_T',\n#                 'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\n#train = train.dropna(subset='sii')\n\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday',\n    \n    # Interaction terms based on correlations:\n    'BMI_Age', 'Height_Weight', 'Age_Weight', 'Waist_Weight', \n    'BMI_Weight', 'BIA_BMI_Weight', 'BIA_Fat_Waist', 'BIA_BMI_Waist', \n    'BMI_Waist', 'BIA_FMI_Waist', 'BIA_LST_Waist', 'BIA_FFM_Waist', \n    'BIA_BMR_Waist', 'BIA_FFMI_Waist', 'BIA_TBW_Waist', 'BIA_ECW_Waist'\n]\n# featuresCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n#                 'CGAS-CGAS_Score', 'Physical-BMI',\n#                 'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n#                 'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n#                 'Fitness_Endurance-Max_Stage',\n#                 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n#                 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n#                 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n#                 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n#                 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n#                 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n#                 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n#                 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n#                 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n#                 'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n#                 'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n#                 'SDS-SDS_Total_T',\n#                 'PreInt_EduHx-computerinternet_hoursday']\n\nfeaturesCols += time_series_cols\ntest = test[featuresCols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T04:37:57.561385Z","iopub.execute_input":"2024-12-12T04:37:57.561692Z","iopub.status.idle":"2024-12-12T04:38:02.305470Z","shell.execute_reply.started":"2024-12-12T04:37:57.561665Z","shell.execute_reply":"2024-12-12T04:38:02.304579Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if np.any(np.isinf(train)):\n    train = train.replace([np.inf, -np.inf], np.nan)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T04:38:02.306679Z","iopub.execute_input":"2024-12-12T04:38:02.307323Z","iopub.status.idle":"2024-12-12T04:38:02.314260Z","shell.execute_reply.started":"2024-12-12T04:38:02.307283Z","shell.execute_reply":"2024-12-12T04:38:02.313482Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from imblearn.under_sampling import RandomUnderSampler\nfrom imblearn.over_sampling import RandomOverSampler\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n    \n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    oversampler = RandomOverSampler(random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n        X_train, y_train = oversampler.fit_resample(X_train, y_train)\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T04:38:02.315593Z","iopub.execute_input":"2024-12-12T04:38:02.315857Z","iopub.status.idle":"2024-12-12T04:38:02.335031Z","shell.execute_reply.started":"2024-12-12T04:38:02.315832Z","shell.execute_reply":"2024-12-12T04:38:02.334390Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Params7 = {'learning_rate': 0.03884249148676395, 'max_depth': 12, 'num_leaves': 413, 'min_data_in_leaf': 14,\n           'feature_fraction': 0.7987976913702801, 'bagging_fraction': 0.7602261703576205, 'bagging_freq': 2, \n           'lambda_l1': 4.735462555910575, 'lambda_l2': 4.735028557007343e-06} # CV : 0.4094 | LB : 0.471\n\nLight = LGBMRegressor(**Params7,random_state=SEED, verbose=-1,n_estimators=200)\n\nSubmission1 = TrainML(Light, test)\n\n# Submission1.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T04:38:02.335884Z","iopub.execute_input":"2024-12-12T04:38:02.336164Z","iopub.status.idle":"2024-12-12T04:38:11.882305Z","shell.execute_reply.started":"2024-12-12T04:38:02.336139Z","shell.execute_reply":"2024-12-12T04:38:11.881332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import optuna\nfrom lightgbm import LGBMRegressor\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.metrics import make_scorer\n\nX_train = train.drop('sii', axis=1)\ny_train = train['sii']\n\ndef qwk_scorer(y_true, y_pred):\n    # Round predictions to nearest integer before calculating QWK\n    y_pred_rounded = np.round(y_pred).astype(int)\n    return quadratic_weighted_kappa(y_true, y_pred_rounded)\n\nqwk_score = make_scorer(qwk_scorer, greater_is_better=True)\n\nbest_model = None\nbest_train_kappa = None\nbest_val_kappa = None\n\ndef objective(trial):\n    global best_model, best_train_kappa, best_val_kappa\n\n    # Define hyperparameters for LightGBM\n    param = {\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.1),\n        'max_depth': trial.suggest_int('max_depth', 4, 16),\n        'num_leaves': trial.suggest_int('num_leaves', 31, 511),\n        'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 5, 50),\n        'feature_fraction': trial.suggest_float('feature_fraction', 0.5, 1.0),\n        'bagging_fraction': trial.suggest_float('bagging_fraction', 0.5, 1.0),\n        'bagging_freq': trial.suggest_int('bagging_freq', 1, 5),\n        'lambda_l1': trial.suggest_float('lambda_l1', 0, 5),\n        'lambda_l2': trial.suggest_float('lambda_l2', 1e-6, 5)\n    }\n\n    # Initialize model with trial parameters\n    model = LGBMRegressor(**param, random_state=SEED, n_estimators=200, verbose=-1)\n\n    # Use TrainML to evaluate model\n    train_kappas = []\n    val_kappas = []\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    oversampler = RandomOverSampler(random_state=SEED)\n\n    for train_idx, val_idx in SKF.split(X_train, y_train):\n        X_fold_train, X_fold_val = X_train.iloc[train_idx], X_train.iloc[val_idx]\n        y_fold_train, y_fold_val = y_train.iloc[train_idx], y_train.iloc[val_idx]\n\n        # Oversample training data\n        X_fold_train, y_fold_train = oversampler.fit_resample(X_fold_train, y_fold_train)\n\n        # Train model\n        model.fit(X_fold_train, y_fold_train)\n\n        # Predictions\n        y_train_pred = model.predict(X_fold_train).round(0).astype(int)\n        y_val_pred = model.predict(X_fold_val).round(0).astype(int)\n\n        # Compute kappa scores\n        train_kappa = quadratic_weighted_kappa(y_fold_train, y_train_pred)\n        val_kappa = quadratic_weighted_kappa(y_fold_val, y_val_pred)\n\n        train_kappas.append(train_kappa)\n        val_kappas.append(val_kappa)\n\n    mean_train_kappa = np.mean(train_kappas)\n    mean_val_kappa = np.mean(val_kappas)\n\n    # Update best model and scores if current trial is the best\n    if best_val_kappa is None or mean_val_kappa > best_val_kappa:\n        best_model = clone(model)\n        best_train_kappa = mean_train_kappa\n        best_val_kappa = mean_val_kappa\n\n    return mean_val_kappa  # Maximize QWK\n\n\nstudy = optuna.create_study(direction='maximize')\nstudy.optimize(objective, n_trials=50, show_progress_bar=True)\n\nprint(\"Best trial:\")\ntrial = study.best_trial\nprint(\"  Value: \", trial.value)\nprint(\"  Params: \", trial.params)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T04:38:11.886585Z","iopub.execute_input":"2024-12-12T04:38:11.887027Z","iopub.status.idle":"2024-12-12T04:42:51.363812Z","shell.execute_reply.started":"2024-12-12T04:38:11.886981Z","shell.execute_reply":"2024-12-12T04:42:51.362805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"  Best Training Kappa: {best_train_kappa:.4f}\")\nprint(f\"  Best Validation Kappa: {best_val_kappa:.4f}\")\n\n# Use the best parameters\nbest_params = trial.params\nbest_model = LGBMRegressor(**best_params, random_state=SEED, n_estimators=200, verbose=-1)\nbest_model.fit(X_train, y_train)\n\nSubmission1 = TrainML(best_model, test)\nSubmission1.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T04:42:51.365095Z","iopub.execute_input":"2024-12-12T04:42:51.365470Z","iopub.status.idle":"2024-12-12T04:42:55.886231Z","shell.execute_reply.started":"2024-12-12T04:42:51.365411Z","shell.execute_reply":"2024-12-12T04:42:55.885371Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# # Model parameters for LightGBM\n# Params = {\n#     'learning_rate': 0.046,\n#     'max_depth': 12,\n#     'num_leaves': 478,\n#     'min_data_in_leaf': 13,\n#     'feature_fraction': 0.893,\n#     'bagging_fraction': 0.784,\n#     'bagging_freq': 4,\n#     'lambda_l1': 10,  # Increased from 6.59\n#     'lambda_l2': 0.01,  # Increased from 2.68e-06\n#     'device': 'gpu'\n\n# }\n\n\n# # XGBoost parameters\n# XGB_Params = {\n#     'learning_rate': 0.05,\n#     'max_depth': 6,\n#     'n_estimators': 200,\n#     'subsample': 0.8,\n#     'colsample_bytree': 0.8,\n#     'reg_alpha': 1,  # Increased from 0.1\n#     'reg_lambda': 5,  # Increased from 1\n#     'random_state': SEED,\n#     'tree_method': 'gpu_hist',\n\n# }\n\n\n# CatBoost_Params = {\n#     'learning_rate': 0.05,\n#     'depth': 6,\n#     'iterations': 200,\n#     'random_seed': SEED,\n#     'verbose': 0,\n#     'l2_leaf_reg': 10,  # Increase this value\n#     'task_type': 'GPU'\n\n# }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T04:42:55.887254Z","iopub.execute_input":"2024-12-12T04:42:55.887554Z","iopub.status.idle":"2024-12-12T04:42:55.891744Z","shell.execute_reply.started":"2024-12-12T04:42:55.887520Z","shell.execute_reply":"2024-12-12T04:42:55.890884Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Create model instances\n# Light = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\n# XGB_Model = XGBRegressor(**XGB_Params)\n# CatBoost_Model = CatBoostRegressor(**CatBoost_Params)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T04:42:55.892746Z","iopub.execute_input":"2024-12-12T04:42:55.893083Z","iopub.status.idle":"2024-12-12T04:42:55.905803Z","shell.execute_reply.started":"2024-12-12T04:42:55.893037Z","shell.execute_reply":"2024-12-12T04:42:55.905040Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Combine models using Voting Regressor\n\n# voting_model = VotingRegressor(estimators=[\n#     ('lightgbm', Light),\n#     ('xgboost', XGB_Model),\n#     ('catboost', CatBoost_Model)\n# ])\n\n# # Train the ensemble model\n# Submission1 = TrainML(voting_model, test)\n\n# Submission1.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T04:42:55.906737Z","iopub.execute_input":"2024-12-12T04:42:55.906971Z","iopub.status.idle":"2024-12-12T04:42:55.916941Z","shell.execute_reply.started":"2024-12-12T04:42:55.906947Z","shell.execute_reply":"2024-12-12T04:42:55.916078Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from catboost import CatBoostRegressor\n# from sklearn.model_selection import GridSearchCV\n\n# # Hyperparameter grid\n# param_grid = {\n#     'learning_rate': [0.01, 0.05, 0.1],\n#     'depth': [4, 6, 10],\n#     'iterations': [100, 200, 300],\n#     'l2_leaf_reg': [1, 3, 10],\n#     'border_count': [32, 64, 128]\n# }\n\n# # Initialize the model\n# catboost_model = CatBoostRegressor(task_type='GPU', random_seed=SEED, verbose=0)\n\n# # Set up GridSearchCV\n# grid_search_cat = GridSearchCV(estimator=catboost_model, param_grid=param_grid, cv=5, verbose=1)\n\n# # Fit the model\n# grid_search_cat.fit(X,y)\n\n# # Best parameters\n# print(f\"Best parameters found: {grid_search_cat.best_params_}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T04:42:55.918065Z","iopub.execute_input":"2024-12-12T04:42:55.918716Z","iopub.status.idle":"2024-12-12T04:42:55.926775Z","shell.execute_reply.started":"2024-12-12T04:42:55.918676Z","shell.execute_reply":"2024-12-12T04:42:55.925987Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from sklearn.model_selection import train_test_split\n# from sklearn.metrics import mean_squared_error\n# import optuna\n\n# # Define the objective function for Optuna\n# def objective(trial):\n#     # Define hyperparameters to tune\n#     learning_rate = trial.suggest_float('learning_rate', 0.001, 0.1)\n#     depth = trial.suggest_int('depth', 4, 12)\n#     iterations = trial.suggest_int('iterations', 100, 1000)\n#     l2_leaf_reg = trial.suggest_float('l2_leaf_reg', 1, 100)\n\n#     # Create a CatBoost model with these hyperparameters\n#     model = CatBoostRegressor(\n#         learning_rate=learning_rate,\n#         depth=depth,\n#         iterations=iterations,\n#         l2_leaf_reg=l2_leaf_reg,\n#         verbose=0  # Suppress output during training\n#     )\n    \n#     # Train/test split (you need to define X and y)\n#     X = train.drop(['sii'], axis=1)\n#     y = train['sii']\n#     X_train, X_val, y_train, y_val = train_test_split(X,y, test_size=0.2, random_state=42)\n    \n#     # Train the model\n#     model.fit(X_train, y_train)\n    \n#     # Make predictions\n#     y_pred = model.predict(X_val)\n    \n#     # Calculate and return the evaluation metric (e.g., Mean Squared Error)\n#     return mean_squared_error(y_val, y_pred)\n\n# # Create a study for optimization\n# study = optuna.create_study(direction='minimize')  # Minimize MSE (lower is better)\n\n# # Optimize the objective function\n# study.optimize(objective, n_trials=100)  # Run 100 trials\n\n# # Print the best parameters and best score\n# print(f\"Best hyperparameters: {study.best_params}\")\n# print(f\"Best score (MSE): {study.best_value}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T04:42:55.927747Z","iopub.execute_input":"2024-12-12T04:42:55.928001Z","iopub.status.idle":"2024-12-12T04:42:55.939296Z","shell.execute_reply.started":"2024-12-12T04:42:55.927976Z","shell.execute_reply":"2024-12-12T04:42:55.938551Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# X = train.drop(['sii'], axis=1)\n# y = train['sii']\n# X_train, X_val, y_train, y_val = train_test_split(X,y, test_size=0.2, random_state=42)\n# def objective_lgb(trial):\n#     # Suggest hyperparameters for LightGBM\n#     param = {\n#         'objective': 'regression',\n#         'metric': 'l2',\n#         'boosting_type': 'gbdt',\n#         'num_leaves': trial.suggest_int('num_leaves', 31, 500),\n#         'max_depth': trial.suggest_int('max_depth', 3, 12),\n#         'learning_rate': trial.suggest_loguniform('learning_rate', 1e-4, 0.1),\n#         'n_estimators': trial.suggest_int('n_estimators', 50, 500),\n#         'feature_fraction': trial.suggest_uniform('feature_fraction', 0.6, 1.0),\n#         'bagging_fraction': trial.suggest_uniform('bagging_fraction', 0.6, 1.0),\n#         'bagging_freq': trial.suggest_int('bagging_freq', 1, 7),\n#         'lambda_l1': trial.suggest_loguniform('lambda_l1', 1e-5, 20),\n#         'lambda_l2': trial.suggest_loguniform('lambda_l2', 1e-5, 0.1)\n#     }\n\n#      # Train LightGBM model\n#     model = LGBMRegressor(**param)\n#     model.fit(X_train, y_train)\n\n#     # Predict and evaluate the model\n#     y_pred = model.predict(X_val)\n#     score = mean_squared_error(y_val, y_pred)\n\n#     return score  # Return the MSE for minimization\n\n\n# # Create the Optuna study for LightGBM\n# study_lgb = optuna.create_study(direction='minimize')\n# study_lgb.optimize(objective_lgb, n_trials=100)\n\n# # Print the best trial for LightGBM\n# print(f\"Best trial for LightGBM: {study_lgb.best_trial.params}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T04:42:55.940264Z","iopub.execute_input":"2024-12-12T04:42:55.940525Z","iopub.status.idle":"2024-12-12T04:42:55.953377Z","shell.execute_reply.started":"2024-12-12T04:42:55.940501Z","shell.execute_reply":"2024-12-12T04:42:55.952498Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Objective function for XGBoost optimization\n# def objective_xgb(trial):\n#     # Suggest hyperparameters for XGBoost\n#     param = {\n#         'objective': 'reg:squarederror',\n#         'eval_metric': 'rmse',\n#         'booster': 'gbtree',\n#         'learning_rate': trial.suggest_loguniform('learning_rate', 1e-4, 0.1),\n#         'max_depth': trial.suggest_int('max_depth', 3, 12),\n#         'n_estimators': trial.suggest_int('n_estimators', 50, 400),\n#         'subsample': trial.suggest_uniform('subsample', 0.6, 1.0),\n#         'colsample_bytree': trial.suggest_uniform('colsample_bytree', 0.6, 1.0),\n#         'reg_alpha': trial.suggest_loguniform('reg_alpha', 1e-3, 100),\n#         'reg_lambda': trial.suggest_loguniform('reg_lambda', 1e-3, 100)\n#     }\n\n#    # Train XGBoost model\n#     model = XGBRegressor(**param)\n#     model.fit(X_train, y_train)\n\n#     # Predict and evaluate the model\n#     y_pred = model.predict(X_val)\n#     score = mean_squared_error(y_val, y_pred)\n\n#     return score  # Return the MSE for minimization\n# # Create the Optuna study for XGBoost\n# study_xgb = optuna.create_study(direction='minimize')\n# study_xgb.optimize(objective_xgb, n_trials=100)\n\n# # Print the best trial for XGBoost\n# print(f\"Best trial for XGBoost: {study_xgb.best_trial.params}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T04:42:55.954539Z","iopub.execute_input":"2024-12-12T04:42:55.954816Z","iopub.status.idle":"2024-12-12T04:42:55.965884Z","shell.execute_reply.started":"2024-12-12T04:42:55.954792Z","shell.execute_reply":"2024-12-12T04:42:55.965243Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from sklearn.datasets import make_regression\n# from sklearn.model_selection import train_test_split\n# from sklearn.metrics import mean_squared_error\n# import matplotlib.pyplot as plt\n\n# X = train.drop(['sii'], axis=1)\n# y = train['sii']\n# X_train, X_val, y_train, y_val = train_test_split(X,y, test_size=0.2, random_state=42)\n\n# # Store MSE values for each trial for plotting\n# mse_values_lgb = []\n# mse_values_xgb = []\n# mse_values_cat = []\n\n# # Objective function for LightGBM optimization\n# def objective_lgb(trial):\n#     param = {\n#         'objective': 'regression',\n#         'metric': 'l2',\n#         'boosting_type': 'gbdt',\n#         'num_leaves': trial.suggest_int('num_leaves', 31, 256),\n#         'max_depth': trial.suggest_int('max_depth', 3, 12),\n#         'learning_rate': trial.suggest_loguniform('learning_rate', 1e-5, 0.1),\n#         'n_estimators': trial.suggest_int('n_estimators', 50, 500),\n#         'feature_fraction': trial.suggest_uniform('feature_fraction', 0.6, 1.0),\n#         'bagging_fraction': trial.suggest_uniform('bagging_fraction', 0.6, 1.0),\n#         'bagging_freq': trial.suggest_int('bagging_freq', 1, 7),\n#         'lambda_l1': trial.suggest_loguniform('lambda_l1', 1e-5, 100),\n#         'lambda_l2': trial.suggest_loguniform('lambda_l2', 1e-5, 100)\n#     }\n    \n#     model = LGBMRegressor(**param)\n#     model.fit(X_train, y_train)\n#     y_pred = model.predict(X_val)\n#     score = mean_squared_error(y_val, y_pred)\n    \n#     mse_values_lgb.append(score)\n#     return score\n\n# # Objective function for XGBoost optimization\n# def objective_xgb(trial):\n#     param = {\n#         'objective': 'reg:squarederror',\n#         'eval_metric': 'rmse',\n#         'max_depth': trial.suggest_int('max_depth', 3, 12),\n#         'learning_rate': trial.suggest_loguniform('learning_rate', 1e-5, 0.1),\n#         'n_estimators': trial.suggest_int('n_estimators', 50, 500),\n#         'subsample': trial.suggest_uniform('subsample', 0.6, 1.0),\n#         'colsample_bytree': trial.suggest_uniform('colsample_bytree', 0.6, 1.0),\n#         'lambda': trial.suggest_loguniform('lambda', 1e-5, 100),\n#         'alpha': trial.suggest_loguniform('alpha', 1e-5, 100)\n#     }\n\n#     model = XGBRegressor(**param)\n#     model.fit(X_train, y_train)\n#     y_pred = model.predict(X_val)\n#     score = mean_squared_error(y_val, y_pred)\n    \n#     mse_values_xgb.append(score)\n#     return score\n\n# # Objective function for CatBoost optimization\n# def objective_cat(trial):\n#     param = {\n#         'iterations': trial.suggest_int('iterations', 50, 500),\n#         'depth': trial.suggest_int('depth', 3, 12),\n#         'learning_rate': trial.suggest_loguniform('learning_rate', 1e-5, 0.1),\n#         'l2_leaf_reg': trial.suggest_loguniform('l2_leaf_reg', 1e-5, 100),\n#         'border_count': trial.suggest_int('border_count', 32, 255),\n#         'bagging_temperature': trial.suggest_uniform('bagging_temperature', 0.0, 1.0)\n#     }\n    \n#     model = CatBoostRegressor(**param, verbose=0)\n#     model.fit(X_train, y_train)\n#     y_pred = model.predict(X_val)\n#     score = mean_squared_error(y_val, y_pred)\n    \n#     mse_values_cat.append(score)\n#     return score\n\n# # Create Optuna studies for each model\n# study_lgb = optuna.create_study(direction='minimize')\n# study_xgb = optuna.create_study(direction='minimize')\n# study_cat = optuna.create_study(direction='minimize')\n\n# # Optimize each model\n# study_lgb.optimize(objective_lgb, n_trials=100)\n# study_xgb.optimize(objective_xgb, n_trials=20)\n# study_cat.optimize(objective_cat, n_trials=20)\n\n# # Extract MSE values for plotting\n# mse_per_trial_lgb = mse_values_lgb\n# mse_per_trial_xgb = mse_values_xgb\n# mse_per_trial_cat = mse_values_cat\n\n# # Plot MSE for all models\n# plt.figure(figsize=(12, 6))\n# plt.plot(range(len(mse_per_trial_lgb)), mse_per_trial_lgb, marker='o', linestyle='-', label='LightGBM', color='b')\n# plt.plot(range(len(mse_per_trial_xgb)), mse_per_trial_xgb, marker='^', linestyle='-', label='XGBoost', color='g')\n# plt.plot(range(len(mse_per_trial_cat)), mse_per_trial_cat, marker='*', linestyle='-', label='CatBoost', color='r')\n\n# plt.xlabel('Trial Number')\n# plt.ylabel('Mean Squared Error (MSE)')\n# plt.title('MSE vs Trial Number during Optuna Optimization for LightGBM, XGBoost, and CatBoost')\n# plt.legend()\n# plt.grid(True)\n# plt.show()\n\n# # Print best trial parameters for each model\n# print(f\"Best trial for LightGBM: {study_lgb.best_trial.params}\")\n# print(f\"Best trial for XGBoost: {study_xgb.best_trial.params}\")\n# print(f\"Best trial for CatBoost: {study_cat.best_trial.params}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T04:42:55.967250Z","iopub.execute_input":"2024-12-12T04:42:55.967596Z","iopub.status.idle":"2024-12-12T04:42:55.980779Z","shell.execute_reply.started":"2024-12-12T04:42:55.967560Z","shell.execute_reply":"2024-12-12T04:42:55.980143Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Model parameters for LightGBM\n# Params = {\n#     'learning_rate': 0.016587585708381213, #from 0.046,\n#     'max_depth': 8, #from 12\n#     'num_leaves': 53, # from 478\n#     'min_data_in_leaf': 13,\n#     'feature_fraction': 0.771, #from 0.893,\n#     'bagging_fraction': 0.720, # from 0.784,\n#     'bagging_freq': 5, #from 4,\n#     'lambda_l1': 3.32, #from 10,  # Increased from 6.59\n#     'lambda_l2': 0.0017, #from 0.0017031847844866162,  # Increased from 2.68e-06\n#     'n_estimators': 374,\n#     'device': 'gpu'\n# }\n\n\n# # XGBoost parameters\n# XGB_Params = {\n#     'learning_rate': 0.026, #from 0.05,\n#     'max_depth': 8, #from 6,\n#     'n_estimators': 200,\n#     'subsample': 0.95, #from 0.8,\n#     'colsample_bytree': 0.805, #from 0.8,\n#     'reg_alpha': 2.69, #from 1,  # Increased from 0.1\n#     'reg_lambda': 1.15, #from  5,  # Increased from 1\n#     'random_state': SEED,\n#     'n_estimators': 371,\n#     'tree_method': 'gpu_hist',\n\n# }\n\n# CatBoost_Params = {\n#     'learning_rate': 0.077, #from 0.05,\n#     'depth': 5, #from 6,\n#     'iterations': 134, #from 200,\n#     'random_seed': SEED,\n#     'verbose': 0,\n#     'l2_leaf_reg': 0.019, #from 10,  # Increase this value\n#     'border_count': 203,\n#     'bagging_temperature': 0.643,\n#     'task_type': 'GPU'\n\n# }\n               \n\n# # Create model instances\n# Light = LGBMRegressor(**Params, random_state=SEED, verbose=-1)\n# XGB_Model = XGBRegressor(**XGB_Params)\n# CatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n\n# # Combine models using Voting Regressor\n\n# voting_model = VotingRegressor(estimators=[\n#     ('lightgbm', Light),\n#     ('xgboost', XGB_Model),\n#     ('catboost', CatBoost_Model)\n# ])\n\n\n\n# # Train the ensemble model\n# Submission1 = TrainML(voting_model, test)\n\n# Submission1.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T04:42:55.981575Z","iopub.execute_input":"2024-12-12T04:42:55.981837Z","iopub.status.idle":"2024-12-12T04:42:55.995948Z","shell.execute_reply.started":"2024-12-12T04:42:55.981812Z","shell.execute_reply":"2024-12-12T04:42:55.995343Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null}]}