{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch\n\nimport numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.decomposition import PCA\nfrom sklearn.metrics import make_scorer, mean_squared_error\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.model_selection import GridSearchCV, KFold\n\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-03T18:03:40.473450Z","iopub.execute_input":"2025-01-03T18:03:40.473861Z","iopub.status.idle":"2025-01-03T18:04:02.254771Z","shell.execute_reply.started":"2025-01-03T18:03:40.473824Z","shell.execute_reply":"2025-01-03T18:04:02.253577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_parquet_file_activity_day_night(filename, dirname, daytime_start='06:00:00', daytime_end='20:00:00',nighttime_start='20:00:00', nighttime_end='06:00:00'):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    daytime_start_dt  = pd.to_datetime(daytime_start, format='%H:%M:%S').time()\n    daytime_start_sec = (daytime_start_dt.hour * 3600 + daytime_start_dt.minute * 60 + daytime_start_dt.second) * 10 ** 9\n    daytime_end_dt  = pd.to_datetime(daytime_end, format='%H:%M:%S').time()\n    daytime_end_sec = (daytime_end_dt.hour * 3600 + daytime_end_dt.minute * 60 + daytime_end_dt.second) * 10 ** 9\n    \n    nighttime_start_dt  = pd.to_datetime(nighttime_start, format='%H:%M:%S').time()\n    nighttime_start_sec = (nighttime_start_dt.hour * 3600 + nighttime_start_dt.minute * 60 + nighttime_start_dt.second) * 10 ** 9\n    nighttime_end_dt  = pd.to_datetime(nighttime_end, format='%H:%M:%S').time()\n    nighttime_end_sec = (nighttime_end_dt.hour * 3600 + nighttime_end_dt.minute * 60 + nighttime_end_dt.second) * 10 ** 9\n    \n    df['is_daytime'] = (df['time_of_day'] >= daytime_start_sec) & (df['time_of_day'] < daytime_end_sec)\n    df['is_nighttime'] = (df['time_of_day'] >= nighttime_start_sec) | (df['time_of_day'] < nighttime_end_sec)\n    \n    df_worn = df[df['non-wear_flag'] == 0.0]\n    target_columns = [\"enmo\", \"light\", \"battery_voltage\"]\n    time_categories = [\"is_daytime\", \"is_nighttime\"]\n    def extract_stats(data):\n        return [\n            data.mean(), \n            data.std(), \n            data.max(), \n            data.min(), \n            data.diff().mean(), \n            data.diff().std()\n        ]\n    time_features = {}\n    for col in target_columns:\n        for time_cat in time_categories:\n            if time_cat == 'is_daytime':\n                time_label = 'day'\n            elif time_cat == 'is_nighttime':\n                time_label = 'night'\n            else:\n                time_label = time_cat\n            \n            filtered_data = df.loc[df[time_cat], col]\n            \n            stats = extract_stats(filtered_data)\n            \n            stat_names = ['mean', 'std', 'max', 'min', 'diff_mean', 'diff_std']\n            for stat, stat_name in zip(stats, stat_names):\n                feature_key = f\"{col}_{time_label}_{stat_name}\"\n                time_features[feature_key] = stat\n    \n    time_features['id'] = filename.split('=')[1]\n    time_series_df = pd.DataFrame([time_features])\n    return time_series_df\ndef load_time_series_activity_day_night(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_parquet_file_activity_day_night(fname, dirname), ids), total=len(ids)))\n    \n    return pd.concat(results, ignore_index=True)\n\ndef feature_engineering(df):\n    # season_cols = [col for col in df.columns if 'Season' in col]\n    # df = df.drop(season_cols, axis=1) \n    \n    # From here on own features\n    def assign_group(age):\n        thresholds = [5, 6, 7, 8, 10, 12, 14, 17, 22]\n        for i, j in enumerate(thresholds):\n            if age <= j:\n                return i\n        return np.nan\n    \n    # Age groups\n    df[\"group\"] = df['Basic_Demos-Age'].apply(assign_group)\n    \n    # BMI \n    BMI_map = {0: 16.3,1: 15.9,2: 16.1,3: 16.8,4: 17.3,5: 19.2,6: 20.2,7: 22.3, 8: 23.6}\n    df['BMI_mean_norm'] = df[['Physical-BMI', 'BIA-BIA_BMI']].mean(axis=1) / df[\"group\"].map(BMI_map)\n    # df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    # FGC zone aggregate\n    zones = ['FGC-FGC_CU_Zone', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone',\n             'FGC-FGC_PU_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone',\n             'FGC-FGC_TL_Zone']\n    \n    df['FGC_Zones_mean'] = df[zones].mean(axis=1)\n    df['FGC_Zones_min'] = df[zones].min(axis=1)\n    df['FGC_Zones_max'] = df[zones].max(axis=1)\n    \n    # Grip\n    GSD_max_map = {0: 9, 1: 9, 2: 9, 3: 9, 4: 16.2, 5: 19.9, 6: 26.1, 7: 31.3, 8: 35.4}\n    GSD_min_map = {0: 9, 1: 9, 2: 9, 3: 9, 4: 14.4, 5: 17.8, 6: 23.4, 7: 27.8, 8: 31.1}\n    \n    df['GS_max'] = df[['FGC-FGC_GSND', 'FGC-FGC_GSD']].max(axis=1) / df[\"group\"].map(GSD_max_map)\n    df['GS_min'] = df[['FGC-FGC_GSND', 'FGC-FGC_GSD']].min(axis=1) / df[\"group\"].map(GSD_min_map)\n    \n    # Curl-ups, push-ups, trunk-lifts... normalized based on age-group\n    cu_map = {0: 1.0, 1: 3.0, 2: 5.0, 3: 7.0, 4: 10.0, 5: 14.0, 6: 20.0, 7: 20.0, 8: 20.0}\n    pu_map = {0: 1.0, 1: 2.0, 2: 3.0, 3: 4.0, 4: 5.0, 5: 7.0, 6: 8.0, 7: 10.0, 8: 14.0}\n    tl_map = {0: 8.0, 1: 8.0, 2: 8.0, 3: 9.0, 4: 9.0, 5: 10.0, 6: 10.0, 7: 10.0, 8: 10.0}\n    \n    df[\"CU_norm\"] = df['FGC-FGC_CU'] / df['group'].map(cu_map)\n    df[\"PU_norm\"] = df['FGC-FGC_PU'] / df['group'].map(pu_map)\n    df[\"TL_norm\"] = df['FGC-FGC_TL'] / df['group'].map(tl_map)\n    \n    # Reach \n    df[\"SR_min\"] = df[['FGC-FGC_SRL', 'FGC-FGC_SRR']].min(axis=1)\n    df[\"SR_max\"] = df[['FGC-FGC_SRL', 'FGC-FGC_SRR']].max(axis=1)\n\n    # BIA Features\n    # Energy Expenditure\n    bmr_map = {0: 934.0, 1: 941.0, 2: 999.0, 3: 1048.0, 4: 1283.0, 5: 1255.0, 6: 1481.0, 7: 1519.0, 8: 1650.0}\n    dee_map = {0: 1471.0, 1: 1508.0, 2: 1640.0, 3: 1735.0, 4: 2132.0, 5: 2121.0, 6: 2528.0, 7: 2566.0, 8: 2793.0}\n    df[\"BMR_norm\"] = df[\"BIA-BIA_BMR\"] / df[\"group\"].map(bmr_map)\n    df[\"DEE_norm\"] = df[\"BIA-BIA_DEE\"] / df[\"group\"].map(dee_map)\n    df[\"DEE_BMR\"] = df[\"BIA-BIA_DEE\"] - df[\"BIA-BIA_BMR\"]\n\n    # FMM\n    ffm_map = {0: 42.0, 1: 43.0, 2: 49.0, 3: 54.0, 4: 60.0, 5: 76.0, 6: 94.0, 7: 104.0, 8: 111.0}\n    df[\"FFM_norm\"] = df[\"BIA-BIA_FFM\"] / df[\"group\"].map(ffm_map)\n\n    # ECW ICW\n    df[\"ICW_ECW\"] = df[\"BIA-BIA_ECW\"] / df[\"BIA-BIA_ICW\"]\n    \n    drop_feats = ['FGC-FGC_GSND', 'FGC-FGC_GSD', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone',\n                  'FGC-FGC_PU_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL_Zone',\n                  'Physical-BMI', 'BIA-BIA_BMI', 'FGC-FGC_CU', 'FGC-FGC_PU', 'FGC-FGC_TL', 'FGC-FGC_SRL', 'FGC-FGC_SRR',\n                 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_Frame_num', \"BIA-BIA_FFM\"]\n    df = df.drop(drop_feats, axis=1) \n    return df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T18:04:02.256040Z","iopub.execute_input":"2025-01-03T18:04:02.256850Z","iopub.status.idle":"2025-01-03T18:04:02.281531Z","shell.execute_reply.started":"2025-01-03T18:04:02.256815Z","shell.execute_reply":"2025-01-03T18:04:02.280018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts = load_time_series_activity_day_night(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series_activity_day_night(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T18:04:02.284016Z","iopub.execute_input":"2025-01-03T18:04:02.284481Z","iopub.status.idle":"2025-01-03T18:05:11.090304Z","shell.execute_reply.started":"2025-01-03T18:04:02.284436Z","shell.execute_reply":"2025-01-03T18:05:11.088989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nseason_cols = [col for col in train.columns if 'Season' in col]\ntrain = train.drop(season_cols, axis=1) \nseason_cols = [col for col in test.columns if 'Season' in col]\ntest = test.drop(season_cols, axis=1) \nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\nPCIAT_drops = ['PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', \n               'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13',\n               'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20']\ntrain = train.drop(PCIAT_drops, axis=1) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T18:05:11.091827Z","iopub.execute_input":"2025-01-03T18:05:11.092152Z","iopub.status.idle":"2025-01-03T18:05:11.207755Z","shell.execute_reply.started":"2025-01-03T18:05:11.092125Z","shell.execute_reply":"2025-01-03T18:05:11.206650Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"imputer = KNNImputer(n_neighbors=10)\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\n# infer NaN values of numerical cols by using KNNImputer\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\n# keep the non-numerical columns\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n\n\ntrain = train_imputed\ntrain = train.dropna(thresh=20, axis=0)\n\npciat_feature_cols = [col for col in train.columns \n                      if col not in ['id','sii','PCIAT-PCIAT_Total']]\nX_pciat = train[pciat_feature_cols].copy()\ny_pciat = train['PCIAT-PCIAT_Total'].copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T18:05:11.208863Z","iopub.execute_input":"2025-01-03T18:05:11.209184Z","iopub.status.idle":"2025-01-03T18:05:18.145770Z","shell.execute_reply.started":"2025-01-03T18:05:11.209158Z","shell.execute_reply":"2025-01-03T18:05:18.144567Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Light_Params = {\n    'learning_rate': 0.020781216498091417,\n    'max_depth': 12, \n    'num_leaves': 124,\n    'min_data_in_leaf': 44, \n    'feature_fraction': 0.7738396785928227, \n    'bagging_fraction': 0.9094982620399213, \n    'bagging_freq': 4,\n    'lambda_l1': 8.867251253806305,\n    'lambda_l2': 8.039317955277623, \n    'n_estimators': 457\n}\n\npciat_model = LGBMRegressor(**Light_Params, seed=2025,verbose = -1)\npciat_model.fit(X_pciat, y_pciat)\n\n# Predict PCIAT for the train set (for optional inspection):\ntrain['PCIAT_pred'] = pciat_model.predict(X_pciat)\n\n# Now predict PCIAT for the test set\n# (the same columns used in X_pciat must exist in test)\ntest['PCIAT-PCIAT_Total'] = pciat_model.predict(test[pciat_feature_cols])\ntrain = feature_engineering(train)\ntest = feature_engineering(test)\n\ntrain = train.replace([np.inf, -np.inf], np.nan)\ntest = test.replace([np.inf, -np.inf], np.nan)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T18:05:18.146761Z","iopub.execute_input":"2025-01-03T18:05:18.147059Z","iopub.status.idle":"2025-01-03T18:05:22.184831Z","shell.execute_reply.started":"2025-01-03T18:05:18.147015Z","shell.execute_reply":"2025-01-03T18:05:22.183787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot distribution of total scores which determine the sii\n# Note the excess zeros -> consider other objective functions\nsns.set_theme(style=\"whitegrid\")\nplt.hist(train['PCIAT-PCIAT_Total'], bins=50, color=\"darkorange\")\nplt.title('Score Distribution')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T18:05:22.185930Z","iopub.execute_input":"2025-01-03T18:05:22.186330Z","iopub.status.idle":"2025-01-03T18:05:22.631291Z","shell.execute_reply.started":"2025-01-03T18:05:22.186289Z","shell.execute_reply":"2025-01-03T18:05:22.630030Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\ntime_series_cols = df_train.columns.tolist() \ntrain = train.drop('id', axis=1)\ntest  = test.drop('id', axis=1)   \n\n\nfeaturesCols = [\n    'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score',\n                # 'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC','BIA-BIA_ECW',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW',  'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw','SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'FGC_Zones_mean','FGC_Zones_min','FGC_Zones_max',\n                'GS_max','GS_min',\"CU_norm\",\"PU_norm\",\"TL_norm\",\"SR_min\",\"SR_max\",\"BMR_norm\",\"DEE_norm\",\"DEE_BMR\",\"ICW_ECW\",\"FFM_norm\",\n                'BMI_mean_norm','PCIAT-PCIAT_Total'\n]\n\nfeaturesCols += time_series_cols\n\ntrainFeatures = featuresCols + ['sii']\n\ntrain = train[trainFeatures]\ntrain = train.dropna(subset='sii')\ntest = test[featuresCols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T18:05:22.633846Z","iopub.execute_input":"2025-01-03T18:05:22.634166Z","iopub.status.idle":"2025-01-03T18:05:22.654170Z","shell.execute_reply.started":"2025-01-03T18:05:22.634141Z","shell.execute_reply":"2025-01-03T18:05:22.652587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_col = train.drop(['sii'], axis=1).columns\nall_importances = pd.DataFrame({'feature': feature_col})\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n    \ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    feature_names = X.columns\n\n    lgb_importances_list = []\n    xgb_importances_list = []\n    cat_importances_list = []\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n        named_estimators = model.named_estimators_\n        \n        # LightGBM\n        if 'lightgbm' in named_estimators and hasattr(named_estimators['lightgbm'], 'feature_importances_'):\n            lgb_importances_list.append(named_estimators['lightgbm'].feature_importances_)\n\n        # XGBoost\n        if 'xgboost' in named_estimators and hasattr(named_estimators['xgboost'], 'feature_importances_'):\n            xgb_importances_list.append(named_estimators['xgboost'].feature_importances_)\n\n        # CatBoost\n        if 'catboost' in named_estimators and hasattr(named_estimators['catboost'], 'get_feature_importance'):\n            cat_importances_list.append(named_estimators['catboost'].get_feature_importance())\n\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    def mean_importances(importances_list):\n        if len(importances_list) > 0:\n            return np.mean(importances_list, axis=0)\n        else:\n            return None\n\n    lgb_mean = mean_importances(lgb_importances_list)\n    xgb_mean = mean_importances(xgb_importances_list)\n    cat_mean = mean_importances(cat_importances_list)\n\n    def normalize_importances(importance_array):\n        if importance_array is not None:\n            return importance_array / importance_array.sum()\n        else:\n            return None\n\n    lgb_mean_normalized = normalize_importances(lgb_mean)\n    xgb_mean_normalized = normalize_importances(xgb_mean)\n    cat_mean_normalized = normalize_importances(cat_mean)\n\n    if lgb_mean_normalized is not None:\n        lgb_df = pd.DataFrame({'feature': feature_names, 'importance': lgb_mean_normalized}).sort_values('importance', ascending=False)\n        plt.figure(figsize=(10,20))\n        plt.barh(lgb_df['feature'], lgb_df['importance'])\n        plt.gca().invert_yaxis()\n        plt.title(\"LightGBM Feature Importance\")\n        plt.show()\n        all_importances['LightGBM'] = lgb_mean_normalized\n\n    if xgb_mean_normalized is not None:\n        xgb_df = pd.DataFrame({'feature': feature_names, 'importance': xgb_mean_normalized}).sort_values('importance', ascending=False)\n        plt.figure(figsize=(10,20))\n        plt.barh(xgb_df['feature'], xgb_df['importance'])\n        plt.gca().invert_yaxis()\n        plt.title(\"XGBoost Feature Importance\")\n        plt.show()\n        all_importances['XGBoost'] = xgb_mean_normalized\n\n    if cat_mean_normalized is not None:\n        cat_df = pd.DataFrame({'feature': feature_names, 'importance': cat_mean_normalized}).sort_values('importance', ascending=False)\n        plt.figure(figsize=(10,20))\n        plt.barh(cat_df['feature'], cat_df['importance'])\n        plt.gca().invert_yaxis()\n        plt.title(\"CatBoost Feature Importance\")\n        plt.show()\n        all_importances['CatBoost'] = cat_mean_normalized\n\n    return submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T18:05:22.655817Z","iopub.execute_input":"2025-01-03T18:05:22.656228Z","iopub.status.idle":"2025-01-03T18:05:22.686835Z","shell.execute_reply.started":"2025-01-03T18:05:22.656179Z","shell.execute_reply":"2025-01-03T18:05:22.685561Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 2025\nn_splits = 5\n# Model parameters for LightGBM\nLight_Params = {\n    'learning_rate': 0.09166688365135439,\n    'max_depth': 12,\n    'num_leaves': 432,\n    'min_data_in_leaf': 17, \n    'feature_fraction': 0.9718570020641087, \n    'bagging_fraction': 0.8811731013548587,\n    'bagging_freq': 3, \n    'lambda_l1': 0.4980199306309456,\n    'lambda_l2': 0.6452996657412411, \n    'n_estimators': 424,\n    'random_state':SEED,\n    'device': 'cpu'\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.15782155024687242,\n    'max_depth': 4, \n    'n_estimators': 365, \n    'subsample': 0.9226564408362041,\n    'colsample_bytree': 0.7198200130798219, \n    'reg_alpha': 0.3728342426755272, \n    'reg_lambda': 3.662453322506941,\n    'random_state': SEED,\n    'tree_method': 'auto'\n\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10,  # Increase this value\n    'task_type': 'CPU'\n\n}\n\n# Create model instances\nLight = LGBMRegressor(**Light_Params,  verbose=-1)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])\nsubmission = TrainML(voting_model, test)\nsubmission.to_csv('submission.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T18:06:03.069102Z","iopub.execute_input":"2025-01-03T18:06:03.070320Z","iopub.status.idle":"2025-01-03T18:06:38.133538Z","shell.execute_reply.started":"2025-01-03T18:06:03.070271Z","shell.execute_reply":"2025-01-03T18:06:38.132165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}