{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch\n\nimport numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.decomposition import PCA\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T19:05:34.341134Z","iopub.execute_input":"2025-01-02T19:05:34.343296Z","iopub.status.idle":"2025-01-02T19:05:34.370793Z","shell.execute_reply.started":"2025-01-02T19:05:34.343224Z","shell.execute_reply":"2025-01-02T19:05:34.368944Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_parquet_file_activity_day_night(filename, dirname, daytime_start='06:00:00', daytime_end='20:00:00',nighttime_start='20:00:00', nighttime_end='06:00:00'):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    daytime_start_dt  = pd.to_datetime(daytime_start, format='%H:%M:%S').time()\n    daytime_start_sec = (daytime_start_dt.hour * 3600 + daytime_start_dt.minute * 60 + daytime_start_dt.second) * 10 ** 9\n    daytime_end_dt  = pd.to_datetime(daytime_end, format='%H:%M:%S').time()\n    daytime_end_sec = (daytime_end_dt.hour * 3600 + daytime_end_dt.minute * 60 + daytime_end_dt.second) * 10 ** 9\n    \n    nighttime_start_dt  = pd.to_datetime(nighttime_start, format='%H:%M:%S').time()\n    nighttime_start_sec = (nighttime_start_dt.hour * 3600 + nighttime_start_dt.minute * 60 + nighttime_start_dt.second) * 10 ** 9\n    nighttime_end_dt  = pd.to_datetime(nighttime_end, format='%H:%M:%S').time()\n    nighttime_end_sec = (nighttime_end_dt.hour * 3600 + nighttime_end_dt.minute * 60 + nighttime_end_dt.second) * 10 ** 9\n    \n    df['is_daytime'] = (df['time_of_day'] >= daytime_start_sec) & (df['time_of_day'] < daytime_end_sec)\n    df['is_nighttime'] = (df['time_of_day'] >= nighttime_start_sec) | (df['time_of_day'] < nighttime_end_sec)\n    \n    df_worn = df[df['non-wear_flag'] == 0]\n    \n    grouped_day = df_worn[df_worn['is_daytime']].groupby('relative_date_PCIAT')['enmo'].mean().reset_index()\n    grouped_day = grouped_day.rename(columns={'enmo': 'mean_enmo_daytime'})\n    \n    grouped_night = df_worn[df_worn['is_nighttime']].groupby('relative_date_PCIAT')['enmo'].mean().reset_index()\n    grouped_night = grouped_night.rename(columns={'enmo': 'mean_enmo_nighttime'})\n    \n    grouped = pd.merge(grouped_day, grouped_night, on='relative_date_PCIAT', how='outer')\n    \n    time_features = {\n        'id': filename.split('=')[1],\n        'mean_month_of_mean_enmo_daytime': grouped['mean_enmo_daytime'].mean(),\n        'mean_month_of_mean_enmo_nighttime': grouped['mean_enmo_nighttime'].mean(),\n        'var_month_of_mean_enmo_daytime': grouped['mean_enmo_daytime'].var(),\n        'var_month_of_mean_enmo_nighttime': grouped['mean_enmo_nighttime'].var()\n    }\n    \n    time_series_df = pd.DataFrame([time_features])\n    return time_series_df\ndef load_time_series_activity_day_night(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_parquet_file_activity_day_night(fname, dirname), ids), total=len(ids)))\n    \n    return pd.concat(results, ignore_index=True)\n\ndef time_features(df):\n    \"\"\"Function extracting Features from ActiGraph data of an individual\"\"\"\n    # Convert time_of_day to hours\n    df[\"hours\"] = df[\"time_of_day\"] // (3_600 * 1_000_000_000)\n    # Basic features \n    features = [\n        df[\"non-wear_flag\"].mean(),\n        df[\"enmo\"][df[\"enmo\"] >= 0.05].sum(),\n    ]\n    \n    # Define conditions for night, day, and no mask (full data)\n    night = ((df[\"hours\"] >= 22) | (df[\"hours\"] <= 5))\n    day = ((df[\"hours\"] <= 20) & (df[\"hours\"] >= 7))\n    no_mask = np.ones(len(df), dtype=bool)\n    \n    # List of columns of interest and masks\n    keys = [\"enmo\", \"anglez\", \"light\", \"battery_voltage\"]\n    masks = [no_mask, night, day]\n    \n    # Helper function for feature extraction\n    def extract_stats(data):\n        return [\n            data.mean(), \n            data.std(), \n            data.max(), \n            data.min(), \n            data.diff().mean(), \n            data.diff().std()\n        ]\n    \n    # Iterate over keys and masks to generate the statistics\n    for key in keys:\n        for mask in masks:\n            filtered_data = df.loc[mask, key]\n            features.extend(extract_stats(filtered_data))\n\n    return features\n\n# Code for parallelized computation of time series data from: Sheikh Muhammad Abdullah \n# https://www.kaggle.com/code/abdmental01/cmi-best-single-model\ndef process_file(filename, dirname):\n    # Process file and extract time features\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return time_features(df), filename.split('=')[1]\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\ndef feature_engineering(df):\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1) \n    # Optionally drop the mapped columns if not needed\n    mapped_cols = [c for c in df.columns if c.endswith('_mapped')]\n    df = df.drop(mapped_cols, axis=1)\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    # df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    # df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    # df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    # df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    # df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    # df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    # df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    # df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    # df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    # df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    # df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    # df['BMI_PHR'] = df['Physical-BMI'] * df['Physical-HeartRate']\n    df['Fat_Age'] = df['BIA-BIA_Fat'] * df['Basic_Demos-Age']\n    fgc_zone_cols = [col for col in df.columns if 'FGC_' in col and col.endswith('_Zone')]\n    \n    df['FGC_Healthy_Count'] = df[fgc_zone_cols].sum(axis=1)\n    df['FGC_Healthy_Ratio'] = df['FGC_Healthy_Count'] / len(fgc_zone_cols)\n\n    df['BMI_Category'] = pd.cut(df['Physical-BMI'], \n                                bins=[0,18.5,25,30,100], \n                                labels=['Underweight','Normal','Overweight','Obese'])\n    # One-hot encode BMI categories\n    df = pd.get_dummies(df, columns=['BMI_Category'], prefix='BMI', dtype=int)\n    df['SMM_to_Fat'] = df['BIA-BIA_SMM'] / (df['BIA-BIA_Fat'] + 1e-5)\n    df['Sleep_PAQ_A_Interaction'] = df['SDS-SDS_Total_Raw'] * df['PAQ_A-PAQ_A_Total']\n    df['Sleep_PAQ_C_Interaction'] = df['SDS-SDS_Total_Raw'] * df['PAQ_C-PAQ_C_Total']\n    return df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T19:05:34.373429Z","iopub.execute_input":"2025-01-02T19:05:34.373770Z","iopub.status.idle":"2025-01-02T19:05:34.418551Z","shell.execute_reply.started":"2025-01-02T19:05:34.373742Z","shell.execute_reply":"2025-01-02T19:05:34.417220Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T19:05:34.420636Z","iopub.execute_input":"2025-01-02T19:05:34.421047Z","iopub.status.idle":"2025-01-02T19:07:04.320190Z","shell.execute_reply.started":"2025-01-02T19:05:34.421004Z","shell.execute_reply":"2025-01-02T19:07:04.319034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T19:07:04.321793Z","iopub.execute_input":"2025-01-02T19:07:04.322148Z","iopub.status.idle":"2025-01-02T19:07:04.390510Z","shell.execute_reply.started":"2025-01-02T19:07:04.322118Z","shell.execute_reply":"2025-01-02T19:07:04.389390Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"imputer = KNNImputer(n_neighbors=5)\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\n# infer NaN values of numerical cols by using KNNImputer\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\n# keep the non-numerical columns\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n\n\ntrain = train_imputed\ntrain = feature_engineering(train)\n# drop rows with less than 10 non-NaN values\ntrain = train.dropna(thresh=10, axis=0)\ntest = feature_engineering(test)\ntrain = train.replace([np.inf, -np.inf], np.nan)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-02T19:07:04.391533Z","iopub.execute_input":"2025-01-02T19:07:04.391803Z","iopub.status.idle":"2025-01-02T19:07:15.096905Z","shell.execute_reply.started":"2025-01-02T19:07:04.391780Z","shell.execute_reply":"2025-01-02T19:07:15.095746Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\ntime_series_cols = df_train.columns.tolist() \ntrain = train.drop('id', axis=1)\ntest  = test .drop('id', axis=1)   \n\n\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'Fat_Age', 'FGC_Healthy_Count', 'FGC_Healthy_Ratio', 'SMM_to_Fat', \n                'Sleep_PAQ_A_Interaction']\n\nfeaturesCols += time_series_cols\n# filter feature columns\ntrain = train[featuresCols]\n#drop train data's rows that does not have sii value\ntrain = train.dropna(subset='sii')\n\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'Fat_Age', 'FGC_Healthy_Count', 'FGC_Healthy_Ratio', 'SMM_to_Fat', \n                'Sleep_PAQ_A_Interaction']\n\nfeaturesCols += time_series_cols\ntest = test[featuresCols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T19:07:15.099068Z","iopub.execute_input":"2025-01-02T19:07:15.099420Z","iopub.status.idle":"2025-01-02T19:07:15.118868Z","shell.execute_reply.started":"2025-01-02T19:07:15.099389Z","shell.execute_reply":"2025-01-02T19:07:15.117603Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_col = train.drop(['sii'], axis=1).columns\nall_importances = pd.DataFrame({'feature': feature_col})\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n    \ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    feature_names = X.columns\n\n    lgb_importances_list = []\n    xgb_importances_list = []\n    cat_importances_list = []\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n        named_estimators = model.named_estimators_\n        \n        # LightGBM\n        if 'lightgbm' in named_estimators and hasattr(named_estimators['lightgbm'], 'feature_importances_'):\n            lgb_importances_list.append(named_estimators['lightgbm'].feature_importances_)\n\n        # XGBoost\n        if 'xgboost' in named_estimators and hasattr(named_estimators['xgboost'], 'feature_importances_'):\n            xgb_importances_list.append(named_estimators['xgboost'].feature_importances_)\n\n        # CatBoost\n        if 'catboost' in named_estimators and hasattr(named_estimators['catboost'], 'get_feature_importance'):\n            cat_importances_list.append(named_estimators['catboost'].get_feature_importance())\n\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    def mean_importances(importances_list):\n        if len(importances_list) > 0:\n            return np.mean(importances_list, axis=0)\n        else:\n            return None\n\n    lgb_mean = mean_importances(lgb_importances_list)\n    xgb_mean = mean_importances(xgb_importances_list)\n    cat_mean = mean_importances(cat_importances_list)\n\n    def normalize_importances(importance_array):\n        if importance_array is not None:\n            return importance_array / importance_array.sum()\n        else:\n            return None\n\n    lgb_mean_normalized = normalize_importances(lgb_mean)\n    xgb_mean_normalized = normalize_importances(xgb_mean)\n    cat_mean_normalized = normalize_importances(cat_mean)\n\n    if lgb_mean_normalized is not None:\n        lgb_df = pd.DataFrame({'feature': feature_names, 'importance': lgb_mean_normalized}).sort_values('importance', ascending=False)\n        plt.figure(figsize=(10,20))\n        plt.barh(lgb_df['feature'], lgb_df['importance'])\n        plt.gca().invert_yaxis()\n        plt.title(\"LightGBM Feature Importance\")\n        plt.show()\n        all_importances['LightGBM'] = lgb_mean_normalized\n\n    if xgb_mean_normalized is not None:\n        xgb_df = pd.DataFrame({'feature': feature_names, 'importance': xgb_mean_normalized}).sort_values('importance', ascending=False)\n        plt.figure(figsize=(10,20))\n        plt.barh(xgb_df['feature'], xgb_df['importance'])\n        plt.gca().invert_yaxis()\n        plt.title(\"XGBoost Feature Importance\")\n        plt.show()\n        all_importances['XGBoost'] = xgb_mean_normalized\n\n    if cat_mean_normalized is not None:\n        cat_df = pd.DataFrame({'feature': feature_names, 'importance': cat_mean_normalized}).sort_values('importance', ascending=False)\n        plt.figure(figsize=(10,20))\n        plt.barh(cat_df['feature'], cat_df['importance'])\n        plt.gca().invert_yaxis()\n        plt.title(\"CatBoost Feature Importance\")\n        plt.show()\n        all_importances['CatBoost'] = cat_mean_normalized\n\n    return submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T19:07:15.120308Z","iopub.execute_input":"2025-01-02T19:07:15.120635Z","iopub.status.idle":"2025-01-02T19:07:15.148088Z","shell.execute_reply.started":"2025-01-02T19:07:15.120607Z","shell.execute_reply":"2025-01-02T19:07:15.146773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 2025\nn_splits = 5\n# Model parameters for LightGBM\nLight_Params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01,  # Increased from 2.68e-06\n    'device': 'cpu'\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED,\n    'tree_method': 'auto',\n\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10,  # Increase this value\n    'task_type': 'CPU'\n\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T19:07:15.149300Z","iopub.execute_input":"2025-01-02T19:07:15.149718Z","iopub.status.idle":"2025-01-02T19:07:15.169215Z","shell.execute_reply.started":"2025-01-02T19:07:15.149672Z","shell.execute_reply":"2025-01-02T19:07:15.167880Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create model instances\nLight = LGBMRegressor(**Light_Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\nvoting_model = VotingRegressor(estimators=[\n   ('lightgbm', Light),\n   ('xgboost', XGB_Model),\n   ('catboost', CatBoost_Model)\n])\nsubmission = TrainML(voting_model, test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T19:07:15.170500Z","iopub.execute_input":"2025-01-02T19:07:15.171070Z","iopub.status.idle":"2025-01-02T19:08:08.795564Z","shell.execute_reply.started":"2025-01-02T19:07:15.171036Z","shell.execute_reply":"2025-01-02T19:08:08.794200Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T19:08:08.796941Z","iopub.execute_input":"2025-01-02T19:08:08.797366Z","iopub.status.idle":"2025-01-02T19:08:08.804476Z","shell.execute_reply.started":"2025-01-02T19:08:08.797336Z","shell.execute_reply":"2025-01-02T19:08:08.803487Z"}},"outputs":[],"execution_count":null}]}