{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","version":"3.11.0"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":50160,"databundleVersionId":7921029,"isSourceIdPinned":false}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"cell-0","cell_type":"markdown","source":"# Submission 6 — Improved LightGBM + CatBoost Ensemble (Kaggle)\n\n**Before running, in Kaggle notebook settings:**\n- Accelerator: **GPU T4 x2** (CatBoost uses GPU; LightGBM runs on CPU)\n- Internet: **Off** (competition rule)\n- Add competition input: `home-credit-credit-risk-model-stability`\n\n**What this notebook does:**\n1. Loads the same features as Sub 2 / Sub 3 / Sub 5 (depth-0 + 4 depth-1 aggregations + 10% stability filter)\n2. Loads improved LightGBM test predictions (from Sub 5 / notebook 06)\n3. Trains a CatBoost model on the same data\n4. Blends the two test-set probability outputs: `ensemble = 0.5 * lgbm + 0.5 * catboost`\n5. Writes one combined `submission.csv`\n\n**Why ensembling helps the Gini stability metric specifically:**\nthe two models make different errors → averaging reduces variance → lower residual std → higher score.","metadata":{}},{"id":"cell-1","cell_type":"code","source":"import glob\nimport polars as pl\nimport pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nfrom catboost import CatBoostClassifier, Pool\nfrom sklearn.metrics import roc_auc_score\nfrom scipy.stats import linregress\nimport matplotlib.pyplot as plt\nimport gc\nimport os\nimport warnings\nwarnings.filterwarnings('ignore')\n\nDATA = '/kaggle/input/competitions/home-credit-credit-risk-model-stability/parquet_files'\nOUTPUT = '/kaggle/working/'\nTRAIN_CUTOFF = 60\n\nprint(os.listdir(f'{DATA}/train')[:10])","metadata":{},"outputs":[],"execution_count":null},{"id":"cell-2","cell_type":"markdown","source":"## 1. Aggregate depth-1 tables (lazy — low memory)","metadata":{}},{"id":"cell-3","cell_type":"code","source":"def aggregate_lazy(pattern: str, name: str) -> pl.DataFrame:\n    files = sorted(glob.glob(pattern))\n    if not files:\n        raise FileNotFoundError(f'No files found: {pattern}')\n    print(f'{name}: {len(files)} file(s)')\n\n    ref_schema = None\n    for f in files:\n        s = pl.read_parquet(f, n_rows=10).schema\n        if any(v != pl.Null for v in s.values()):\n            ref_schema = s\n            break\n\n    parts = []\n    for f in files:\n        part = pl.read_parquet(f)\n        exprs = []\n        for c in part.columns:\n            if c == 'case_id':\n                continue\n            dtype = part[c].dtype\n            if dtype == pl.Null and ref_schema and c in ref_schema:\n                target = ref_schema[c]\n                exprs.append(pl.col(c).cast(target if target != pl.Null else pl.Utf8))\n            elif dtype in [pl.Float64, pl.Int64, pl.Int32, pl.Int16, pl.Int8]:\n                exprs.append(pl.col(c).cast(pl.Float32))\n        if exprs:\n            part = part.with_columns(exprs)\n        parts.append(part)\n\n    df = pl.concat(parts, how='diagonal_relaxed')\n    del parts\n    gc.collect()\n\n    schema = df.schema\n    num_cols = [c for c, dtype in schema.items() if c != 'case_id' and dtype == pl.Float32]\n    cat_cols = [c for c, dtype in schema.items() if c != 'case_id' and dtype == pl.Utf8]\n\n    aggs = [pl.len().alias(f'{name}_count').cast(pl.Int32)]\n    for col in num_cols:\n        aggs += [\n            pl.col(col).mean().alias(f'{name}_{col}_mean'),\n            pl.col(col).max().alias(f'{name}_{col}_max'),\n            pl.col(col).min().alias(f'{name}_{col}_min'),\n            pl.col(col).std().alias(f'{name}_{col}_std'),\n        ]\n    for col in cat_cols:\n        aggs.append(pl.col(col).n_unique().cast(pl.Int16).alias(f'{name}_{col}_nunique'))\n\n    result = df.group_by('case_id').agg(aggs)\n    del df\n    gc.collect()\n    print(f'  -> {result.shape}')\n    return result","metadata":{},"outputs":[],"execution_count":null},{"id":"cell-4","cell_type":"code","source":"agg_applprev      = aggregate_lazy(f'{DATA}/train/train_applprev_1_*.parquet',        'applprev')\nagg_cb_a          = aggregate_lazy(f'{DATA}/train/train_credit_bureau_a_1_*.parquet', 'cb_a')\nagg_cb_b          = aggregate_lazy(f'{DATA}/train/train_credit_bureau_b_1.parquet',   'cb_b')\nagg_person        = aggregate_lazy(f'{DATA}/train/train_person_1.parquet',            'person')\n\nagg_applprev_test = aggregate_lazy(f'{DATA}/test/test_applprev_1_*.parquet',          'applprev')\nagg_cb_a_test     = aggregate_lazy(f'{DATA}/test/test_credit_bureau_a_1_*.parquet',   'cb_a')\nagg_cb_b_test     = aggregate_lazy(f'{DATA}/test/test_credit_bureau_b_1.parquet',     'cb_b')\nagg_person_test   = aggregate_lazy(f'{DATA}/test/test_person_1.parquet',              'person')","metadata":{},"outputs":[],"execution_count":null},{"id":"cell-5","cell_type":"markdown","source":"## 2. Load depth-0 tables and join everything","metadata":{}},{"id":"cell-6","cell_type":"code","source":"def cast_static(df: pl.DataFrame) -> pl.DataFrame:\n    return df.with_columns([\n        pl.col(c).cast(pl.Float32)\n        for c in df.columns\n        if df[c].dtype in [pl.Float64, pl.Int64, pl.Int32]\n        and c not in ('case_id', 'target', 'WEEK_NUM', 'MONTH')\n    ])\n\nbase      = pl.read_parquet(f'{DATA}/train/train_base.parquet')\nstatic0   = pl.scan_parquet(sorted(glob.glob(f'{DATA}/train/train_static_0_*.parquet'))).collect().pipe(cast_static)\nstatic_cb = pl.read_parquet(f'{DATA}/train/train_static_cb_0.parquet').pipe(cast_static)\n\ntest_base      = pl.read_parquet(f'{DATA}/test/test_base.parquet')\ntest_static0   = pl.scan_parquet(sorted(glob.glob(f'{DATA}/test/test_static_0_*.parquet'))).collect().pipe(cast_static)\ntest_static_cb = pl.read_parquet(f'{DATA}/test/test_static_cb_0.parquet').pipe(cast_static)\n\ndef build(base_df, s0, scb, aggs):\n    df = base_df.join(s0, on='case_id', how='left').join(scb, on='case_id', how='left')\n    for agg in aggs:\n        df = df.join(agg, on='case_id', how='left')\n    return df\n\ndf   = build(base, static0, static_cb, [agg_applprev, agg_cb_a, agg_cb_b, agg_person])\ntest = build(test_base, test_static0, test_static_cb, [agg_applprev_test, agg_cb_a_test, agg_cb_b_test, agg_person_test])\n\ndel static0, static_cb, test_static0, test_static_cb\ndel agg_applprev, agg_cb_a, agg_cb_b, agg_person\ndel agg_applprev_test, agg_cb_a_test, agg_cb_b_test, agg_person_test\ngc.collect()\n\nprint(f'Train: {df.shape}')\nprint(f'Test:  {test.shape}')","metadata":{},"outputs":[],"execution_count":null},{"id":"cell-7","cell_type":"markdown","source":"## 3. Preprocessing\n\nSame preprocessing for both models — string columns are integer-encoded (with `'NA'` for nulls).\nLightGBM treats those integers as ordinal numeric; CatBoost gets them as `cat_features` and\ntreats them as unordered categories.","metadata":{}},{"id":"cell-8","cell_type":"code","source":"DROP_COLS = ['case_id', 'date_decision', 'MONTH', 'WEEK_NUM', 'target']\nTEST_DROP = ['case_id', 'date_decision', 'MONTH', 'WEEK_NUM']\n\ndef preprocess(df: pl.DataFrame, drop_cols: list):\n    date_cols = [c for c in df.columns if c.endswith('_D') and c not in drop_cols]\n    for col in date_cols:\n        df = df.with_columns(\n            pl.col(col).str.to_date(strict=False).cast(pl.Int32).alias(col)\n        )\n\n    # Treat both Utf8 strings AND Boolean columns as categoricals.\n    # Boolean ones (_L suffix) become 'true'/'false'/'NA' strings, then int16 codes.\n    # Without the Boolean branch they end up as pandas 'object' dtype → LightGBM rejects.\n    cat_col_names = [\n        c for c in df.columns\n        if df[c].dtype in [pl.Utf8, pl.Boolean] and c not in drop_cols\n    ]\n    for col in cat_col_names:\n        df = df.with_columns(\n            pl.col(col).cast(pl.Utf8).fill_null('NA').cast(pl.Categorical).to_physical().cast(pl.Int16).alias(col)\n        )\n\n    df = df.drop([c for c in drop_cols if c in df.columns])\n    return df, [c for c in cat_col_names if c in df.columns]\n\nweek_num = df['WEEK_NUM'].to_numpy()\ny = df['target'].to_numpy()\n\nX_df,      cat_cols      = preprocess(df,   DROP_COLS)\nX_test_df, cat_cols_test = preprocess(test, TEST_DROP)\ndel df, test\ngc.collect()\n\nfeature_cols = [c for c in X_df.columns if c in X_test_df.columns]\ncat_cols     = [c for c in cat_cols if c in feature_cols]\nX_df         = X_df.select(feature_cols)\nX_test_df    = X_test_df.select(feature_cols)\n\nprint(f'Features before stability filter: {len(feature_cols)}')\nprint(f'Categorical columns (Utf8 + Boolean): {len(cat_cols)}')","metadata":{},"outputs":[],"execution_count":null},{"id":"cell-9","cell_type":"markdown","source":"## 4. Stability-first feature filtering","metadata":{}},{"id":"cell-10","cell_type":"code","source":"train_mask = week_num <= TRAIN_CUTOFF\n\nnumeric_cols = [\n    c for c in X_df.columns\n    if X_df[c].dtype in [pl.Float32, pl.Float64, pl.Int32, pl.Int16, pl.Int8]\n    and c not in cat_cols\n]\n\ndf_wk = (\n    X_df.select(numeric_cols)\n    .with_columns(pl.Series(name='WEEK_NUM', values=week_num.tolist()))\n    .filter(pl.col('WEEK_NUM') <= TRAIN_CUTOFF)\n)\nweekly_means = df_wk.group_by('WEEK_NUM').agg([pl.col(c).mean() for c in numeric_cols]).sort('WEEK_NUM')\n\ndrift = {}\nfor col in numeric_cols:\n    vals = weekly_means[col].drop_nulls().to_numpy()\n    if len(vals) < 5 or np.abs(vals.mean()) < 1e-9:\n        drift[col] = 0.0\n    else:\n        drift[col] = vals.std() / np.abs(vals.mean())\n\ndel df_wk, weekly_means\ngc.collect()\n\ndrift_df = pl.DataFrame({'feature': list(drift.keys()), 'drift': list(drift.values())}).sort('drift', descending=True)\nn_drop = max(1, int(len(drift_df) * 0.10))\nunstable = set(drift_df.head(n_drop)['feature'].to_list())\nstable_features = [c for c in feature_cols if c not in unstable]\ncat_cols_stable = [c for c in cat_cols if c in stable_features]\n\nprint(f'Dropped {len(unstable)} unstable features, {len(stable_features)} remaining')\nprint(f'Categorical features remaining: {len(cat_cols_stable)}')","metadata":{},"outputs":[],"execution_count":null},{"id":"cell-11","cell_type":"code","source":"X      = X_df.select(stable_features).to_pandas()\nX_test = X_test_df.select(stable_features).to_pandas()\ndel X_df, X_test_df\ngc.collect()\n\n# Defensive cleanup: convert any object / bool / nullable dtypes that survived\n# preprocessing into int16. Catches edge cases where a column was all-null in\n# test (Polars Null dtype) or had an unusual dtype that the preprocess missed.\ndef cleanup_object_dtypes(df, cat_list):\n    for c in df.columns:\n        dt_str = str(df[c].dtype)\n        if dt_str in ('object', 'bool', 'boolean', 'category') or 'Bool' in dt_str:\n            df[c] = df[c].astype('object').fillna('NA').astype('category').cat.codes.astype('int16')\n            if c not in cat_list:\n                cat_list.append(c)\n    return df, cat_list\n\nX,      cat_cols_stable = cleanup_object_dtypes(X,      cat_cols_stable)\nX_test, _               = cleanup_object_dtypes(X_test, cat_cols_stable)\n\n# Defensive: also handle X_test cat cols whose Polars Null dtype came through\nfor c in cat_cols_stable:\n    if X_test[c].dtype == 'object' or X_test[c].isna().any():\n        X_test[c] = X_test[c].fillna('NA').astype('category').cat.codes.astype('int16')\n\nprint(f'Feature matrix: {X.shape}, Test: {X_test.shape}')\nprint(f'Categorical features (final): {len(cat_cols_stable)}')\nprint(f'X dtype counts: {X.dtypes.value_counts().to_dict()}')","metadata":{},"outputs":[],"execution_count":null},{"id":"cell-12","cell_type":"markdown","source":"## 5. Gini stability metric","metadata":{}},{"id":"cell-13","cell_type":"code","source":"def gini_stability(week_nums, targets, preds):\n    weeks = np.unique(week_nums)\n    weekly_gini = []\n    for w in weeks:\n        mask = week_nums == w\n        if targets[mask].sum() == 0:\n            continue\n        weekly_gini.append(2 * roc_auc_score(targets[mask], preds[mask]) - 1)\n    weekly_gini = np.array(weekly_gini)\n    mean_gini = weekly_gini.mean()\n    slope, intercept, *_ = linregress(np.arange(len(weekly_gini)), weekly_gini)\n    residuals = weekly_gini - (intercept + slope * np.arange(len(weekly_gini)))\n    return mean_gini + 88.0 * min(0, slope) - 0.5 * residuals.std(), mean_gini, slope, residuals.std()","metadata":{},"outputs":[],"execution_count":null},{"id":"cell-14","cell_type":"markdown","source":"## 6. Time-based split","metadata":{}},{"id":"cell-15","cell_type":"code","source":"val_mask = week_num > TRAIN_CUTOFF\n\nX_train, y_train = X[train_mask], y[train_mask]\nX_val,   y_val   = X[val_mask],   y[val_mask]\nweek_val = week_num[val_mask]\n\nprint(f'Train: {X_train.shape[0]:,} ({y_train.mean():.3%} pos)')\nprint(f'Val:   {X_val.shape[0]:,} ({y_val.mean():.3%} pos)')","metadata":{},"outputs":[],"execution_count":null},{"id":"cell-16","cell_type":"markdown","source":"## 7. Train LightGBM (CPU)","metadata":{}},{"id":"cell-17","cell_type":"code","source":"# LightGBM with best params from Optuna (Sub 5)\nlgb_params = {\n    'objective': 'binary', 'metric': 'auc',\n    'n_estimators': 1000,\n    'learning_rate':     0.022812809566861304,\n    'num_leaves':        55,\n    'min_child_samples': 38,\n    'feature_fraction':  0.774816395214887,\n    'bagging_fraction':  0.5187196516747448,\n    'bagging_freq':      1,\n    'reg_alpha':         0.7931156231178,\n    'reg_lambda':        0.3782056934716548,\n    'scale_pos_weight':  (y_train == 0).sum() / (y_train == 1).sum(),\n    'n_jobs': -1, 'verbose': -1, 'random_state': 42,\n}\n\nlgb_model = lgb.LGBMClassifier(**lgb_params)\nlgb_model.fit(\n    X_train, y_train,\n    eval_set=[(X_val, y_val)],\n    callbacks=[lgb.early_stopping(50, verbose=False), lgb.log_evaluation(100)]\n)\nprint(f'LightGBM best iteration: {lgb_model.best_iteration_}')\nlgb_val_preds = lgb_model.predict_proba(X_val)[:, 1]\nlgb_best_iter = lgb_model.best_iteration_\n","metadata":{},"outputs":[],"execution_count":null},{"id":"cell-18","cell_type":"markdown","source":"## 8. Train CatBoost (GPU)","metadata":{}},{"id":"cell-19","cell_type":"code","source":"train_pool = Pool(X_train, label=y_train, cat_features=cat_cols_stable)\nval_pool   = Pool(X_val,   label=y_val,   cat_features=cat_cols_stable)\n\ncat_params = {\n    'iterations': 1000, 'learning_rate': 0.05, 'depth': 6,\n    'l2_leaf_reg': 3.0, 'random_strength': 1.0, 'border_count': 128,\n    'auto_class_weights': 'Balanced', 'eval_metric': 'AUC',\n    'random_seed': 42, 'task_type': 'GPU', 'devices': '0',\n    'verbose': 100,\n}\n\ncat_model = CatBoostClassifier(**cat_params)\ncat_model.fit(train_pool, eval_set=val_pool, early_stopping_rounds=50)\nprint(f'\\nCatBoost best iteration: {cat_model.best_iteration_}')\n\ncat_val_preds = cat_model.predict_proba(val_pool)[:, 1]\ncat_best_iter = cat_model.best_iteration_\n\ndel train_pool, val_pool\ngc.collect()","metadata":{},"outputs":[],"execution_count":null},{"id":"cell-20","cell_type":"markdown","source":"## 9. Compare individual models vs ensemble on validation","metadata":{}},{"id":"cell-21","cell_type":"code","source":"ensemble_val_preds = 0.5 * lgb_val_preds + 0.5 * cat_val_preds\n\nfor name, preds in [('LightGBM alone', lgb_val_preds),\n                    ('CatBoost alone', cat_val_preds),\n                    ('Ensemble 50/50', ensemble_val_preds)]:\n    auc = roc_auc_score(y_val, preds)\n    score, mean_gini, slope, resid_std = gini_stability(week_val, y_val, preds)\n    print(f'{name:18s}  AUC={auc:.4f}  Gini-stab={score:.4f}  mean-gini={mean_gini:.4f}  resid-std={resid_std:.4f}')\n","metadata":{},"outputs":[],"execution_count":null},{"id":"cell-22","cell_type":"markdown","source":"## 10. Retrain both models on full data & generate ensembled submission","metadata":{}},{"id":"cell-23","cell_type":"code","source":"# Free validation predictions and per-model objects from training before retraining\ndel lgb_model, cat_model\ndel X_train, X_val, y_train, y_val\ngc.collect()\n\n# --- LightGBM final (with Optuna best params) ---\nlgb_final = lgb.LGBMClassifier(**{**lgb_params, 'n_estimators': lgb_best_iter})\nlgb_final.fit(X, y)\nlgb_test_preds = lgb_final.predict_proba(X_test)[:, 1]\ndel lgb_final\ngc.collect()\nprint(f'LightGBM test predictions: shape {lgb_test_preds.shape}, mean {lgb_test_preds.mean():.3f}')\n\n# --- CatBoost final ---\nfull_pool = Pool(X, label=y, cat_features=cat_cols_stable)\ntest_pool = Pool(X_test,      cat_features=cat_cols_stable)\n\ncat_final = CatBoostClassifier(**{**cat_params, 'iterations': cat_best_iter})\ncat_final.fit(full_pool)\ncat_test_preds = cat_final.predict_proba(test_pool)[:, 1]\ndel cat_final, full_pool, test_pool\ngc.collect()\nprint(f'CatBoost test predictions: shape {cat_test_preds.shape}, mean {cat_test_preds.mean():.3f}')\n\n# --- Blend ---\nensemble_test_preds = 0.5 * lgb_test_preds + 0.5 * cat_test_preds\n\nsubmission = pl.DataFrame({\n    'case_id': test_base['case_id'],\n    'score': ensemble_test_preds\n})\nsubmission.write_csv(f'{OUTPUT}/submission.csv')\nprint(f'\\nSaved {len(submission):,} ensembled predictions to {OUTPUT}/submission.csv')\nprint(submission.describe())","metadata":{},"outputs":[],"execution_count":null}]}