{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":50160,"databundleVersionId":7921029},{"sourceType":"datasetVersion","sourceId":15381455,"datasetId":9839565,"databundleVersionId":16294267},{"sourceType":"datasetVersion","sourceId":15381526,"datasetId":9839609,"databundleVersionId":16294348}],"dockerImageVersionId":31286,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":22.995521,"end_time":"2026-03-26T14:26:53.056098","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-03-26T14:26:30.060577","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Home Credit Credit Risk Model Stability — train locally, infer anywhere\nThis notebook implements a clean, student-readable version of the approach described in `reference/suggestSolution.md`:\n- Convert many relational tables into a **single row per `case_id`** using **time-agnostic aggregations**\n- Train an ensemble of **3 models** locally:\n  - **LightGBM**\n  - **CatBoost**\n  - **PyTorch TabResNet** (residual MLP for tabular data)\n- Save **artifacts** to `./models/` (so you can upload them to Kaggle as a Dataset)\n- Run inference (local or Kaggle) and write **`submission.csv`**\n### Why this pipeline is structured this way\n- Training is expensive; we do it **once locally**.\n- Kaggle inference should be lightweight and reproducible by loading the saved artifacts.\n- Feature names are saved explicitly (`models/feature_names.json`) so inference is always aligned.\n### Data sources\n- **Local (this repo)**: `home-credit-credit-risk-model-stability/csv_files/{train,test}`\n- **Kaggle**: `/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/{train,test}`\n### References in this repo\n- `reference/suggestSolution.md`\n- `reference/fork_of_home_credit_risk_lightgbm.py`\n- `reference/fork_of_home_credit_catboost_inference.py`\n**Variant**: `submission_v2` uses direct probability blending instead of percentile-rank blending.\n","metadata":{"papermill":{"duration":0.004306,"end_time":"2026-03-26T14:26:32.588597","exception":false,"start_time":"2026-03-26T14:26:32.584291","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from __future__ import annotations\n\nimport os\nimport sys\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nIS_KAGGLE = Path(\"/kaggle\").exists() and Path(\"/kaggle/input\").exists()\n\n\ndef find_repo_root(start: Path) -> Path:\n    for p in [start, *start.parents]:\n        if (p / \"src\").exists():\n            return p\n    return start\n\n\n# Make `src/` importable.\n# - Locally: infer repo root.\n# - Kaggle: point at the attached \"code bundle\" dataset.\nif IS_KAGGLE:\n    REPO_ROOT = Path(\"/kaggle/input/datasets/malcolmfongchenghong/code-bundle/kaggle_code_bundle\")\nelse:\n    REPO_ROOT = find_repo_root(Path.cwd())\n\nif str(REPO_ROOT) not in sys.path:\n    sys.path.insert(0, str(REPO_ROOT))\n\n# Kaggle's /kaggle/input is read-only. Keep CWD writable.\nos.chdir(\"/kaggle/working\" if IS_KAGGLE else REPO_ROOT)\n\nprint(\"IS_KAGGLE:\", IS_KAGGLE)\nprint(\"REPO_ROOT:\", REPO_ROOT)\nprint(\"CWD:\", Path.cwd())\n\nfrom src.home_credit_io import resolve_competition_root, available_train_test_files\nfrom src.home_credit_features import (\n    FeatureBuildConfig,\n    build_feature_matrix,\n    read_table,\n    read_tables,\n)\nfrom src.home_credit_models import (\n    align_features_for_model,\n    blend_by_percentile_rank,\n    load_artifacts,\n    resolve_model_root,\n    predict_proba_average,\n)\n\n# Training utilities (used only if DO_TRAIN_* enabled)\nfrom src.home_credit_training import (\n    TrainingConfig,\n    infer_categorical_columns,\n    make_cv_splits_stratified_group_kfold,\n    save_metadata,\n    train_catboost_folds,\n    train_lightgbm_folds,\n    train_tabresnet_folds,\n)\n\npd.set_option(\"display.max_columns\", 200)\npl.Config.set_tbl_cols(50)\npl.Config.set_tbl_rows(10)\n","metadata":{"execution":{"iopub.status.busy":"2026-03-26T15:57:12.348274Z","iopub.execute_input":"2026-03-26T15:57:12.348582Z","iopub.status.idle":"2026-03-26T15:57:12.359526Z","shell.execute_reply.started":"2026-03-26T15:57:12.348556Z","shell.execute_reply":"2026-03-26T15:57:12.358670Z"},"papermill":{"duration":1.435522,"end_time":"2026-03-26T14:26:34.026271","exception":false,"start_time":"2026-03-26T14:26:32.590749","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1) Configuration\n\nSet the **model source** and the **model type** you want to run.\n\n- If you run locally, the cleanest approach is to put artifacts under `./models/`.\n- If you run on Kaggle, upload those artifacts as a Kaggle Dataset and set `KAGGLE_MODELS_DATASET_SLUG`.\n","metadata":{"papermill":{"duration":0.002163,"end_time":"2026-03-26T14:26:34.030861","exception":false,"start_time":"2026-03-26T14:26:34.028698","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# ---- Pipeline toggles ----\n# For Kaggle submission: keep training OFF, inference ON.\nDO_TRAIN_LGBM = False\nDO_TRAIN_CATBOOST = False\nDO_TRAIN_TABRESNET = False\nDO_INFER = True\n# Inference model toggles (WARNING: enabling all 3 is slower/heavier)\nRUN_LGBM_INFER = True\nRUN_CATBOOST_INFER = True\nRUN_TABRESNET_INFER = False\n# ---- Artifact location ----\nMODEL_ROOT = \"/kaggle/input/datasets/malcolmfongchenghong/artifacts/models\"  # or \"./models\" locally\nKAGGLE_MODELS_DATASET_SLUG = None\n# Streaming batch size (lower if you still OOM)\nPREDICT_CHUNK_SIZE = 20_000 if IS_KAGGLE else 200_000\n# ---- Training config (only used if DO_TRAIN_* enabled) ----\nTRAIN_CFG = TrainingConfig(n_splits=5, random_state=42)\n# ---- Safety switches (avoid accidental overwrite) ----\nFORCE_OVERWRITE_METADATA = False\nFORCE_RETRAIN_LGBM = False\nFORCE_RETRAIN_CATBOOST = False\nFORCE_RETRAIN_TABRESNET = False\n# Direct-probability blend weights (used when dynamic weighting is off)\nBLEND_WEIGHTS = {\n    \"lgbm\": 1.0,\n    \"catboost\": 1.0,\n    \"tabresnet\": 1.0,\n}\n\n# Optional: learn blend weights from OOF (requires X_model, y, splits in memory)\nOPTIMIZE_BLEND_WEIGHTS = False\nBLEND_GRID_STEP = 0.05\n","metadata":{"execution":{"iopub.status.busy":"2026-03-26T15:57:12.361136Z","iopub.execute_input":"2026-03-26T15:57:12.361499Z","iopub.status.idle":"2026-03-26T15:57:12.378423Z","shell.execute_reply.started":"2026-03-26T15:57:12.361469Z","shell.execute_reply":"2026-03-26T15:57:12.377697Z"},"papermill":{"duration":0.008025,"end_time":"2026-03-26T14:26:34.040962","exception":false,"start_time":"2026-03-26T14:26:34.032937","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2) Locate dataset (local vs Kaggle)\n\nWe auto-detect where the competition files are mounted.\n\n- On **Kaggle**, this becomes `/kaggle/input/...`.\n- In this repo, we use the provided CSVs under `home-credit-credit-risk-model-stability/csv_files/`.\n","metadata":{"papermill":{"duration":0.002119,"end_time":"2026-03-26T14:26:34.045180","exception":false,"start_time":"2026-03-26T14:26:34.043061","status":"completed"},"tags":[]}},{"cell_type":"code","source":"paths = resolve_competition_root()\nfile_map = available_train_test_files(paths)\n\nprint(\"Resolved dataset root:\", paths.root)\nprint(\"Train dir:\", paths.train_dir)\nprint(\"Test dir:\", paths.test_dir)\nprint(\"Sample submission:\", paths.sample_submission_csv)\n\n# Show which table files we will use for TEST (inference-only).\nfor k, v in file_map[\"test\"].items():\n    if isinstance(v, list):\n        print(f\"test.{k}: {len(v)} files\")\n    else:\n        print(f\"test.{k}: {v.name}\")\n","metadata":{"execution":{"iopub.status.busy":"2026-03-26T15:57:12.379422Z","iopub.execute_input":"2026-03-26T15:57:12.379646Z","iopub.status.idle":"2026-03-26T15:57:12.409523Z","shell.execute_reply.started":"2026-03-26T15:57:12.379626Z","shell.execute_reply":"2026-03-26T15:57:12.408853Z"},"papermill":{"duration":0.052565,"end_time":"2026-03-26T14:26:34.099843","exception":false,"start_time":"2026-03-26T14:26:34.047278","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3) Feature engineering (1 row per `case_id`)\n\n### The key idea\n\nThe competition provides many tables at different granularities (multiple rows per `case_id`).\nWe **aggregate** them into summary statistics so every table becomes **one row per `case_id`**, then we **join** everything together.\n\n### Why simple aggregations work well\n\nThey are:\n\n- **Stable** across time shifts\n- **Fast** to compute\n- **Harder to overfit** than complex sequence models (a common pitfall noted in top solutions)\n","metadata":{"papermill":{"duration":0.002258,"end_time":"2026-03-26T14:26:34.104811","exception":false,"start_time":"2026-03-26T14:26:34.102553","status":"completed"},"tags":[]}},{"cell_type":"code","source":"cfg = FeatureBuildConfig()\n\n# ---- Base ----\nbase_test = read_table(Path(file_map[\"test\"][\"base\"]), depth=None, table_name=\"test_base\")\n\n# ---- Depth 0 (already 1 row per case_id, typically) ----\nstatic_cb_0 = read_table(Path(file_map[\"test\"][\"static_cb_0\"]), depth=None, table_name=\"test_static_cb_0\")\nstatic_0 = read_tables([Path(p) for p in file_map[\"test\"][\"static_0_parts\"]], depth=None, table_name=\"test_static_0_parts\")\n\ndepth_0_tables = [static_cb_0, static_0]\n\n# ---- Depth 1 (aggregate to case_id) ----\napplprev_1 = read_tables([Path(p) for p in file_map[\"test\"][\"applprev_1_parts\"]], depth=1, table_name=\"test_applprev_1_parts\")\ncb_a_1 = read_tables([Path(p) for p in file_map[\"test\"][\"credit_bureau_a_1_parts\"]], depth=1, table_name=\"test_credit_bureau_a_1_parts\")\ncb_b_1 = read_table(Path(file_map[\"test\"][\"credit_bureau_b_1\"]), depth=1, table_name=\"test_credit_bureau_b_1\")\n\ndeposit_1 = read_table(Path(file_map[\"test\"][\"deposit_1\"]), depth=1, table_name=\"test_deposit_1\")\ndebitcard_1 = read_table(Path(file_map[\"test\"][\"debitcard_1\"]), depth=1, table_name=\"test_debitcard_1\")\n\nperson_1 = read_table(Path(file_map[\"test\"][\"person_1\"]), depth=1, table_name=\"test_person_1\")\nother_1 = read_table(Path(file_map[\"test\"][\"other_1\"]), depth=1, table_name=\"test_other_1\")\n\ntax_a_1 = read_table(Path(file_map[\"test\"][\"tax_registry_a_1\"]), depth=1, table_name=\"test_tax_registry_a_1\")\ntax_b_1 = read_table(Path(file_map[\"test\"][\"tax_registry_b_1\"]), depth=1, table_name=\"test_tax_registry_b_1\")\ntax_c_1 = read_table(Path(file_map[\"test\"][\"tax_registry_c_1\"]), depth=1, table_name=\"test_tax_registry_c_1\")\n\ndepth_1_tables = [\n    applprev_1,\n    cb_a_1,\n    cb_b_1,\n    deposit_1,\n    debitcard_1,\n    tax_a_1,\n    tax_b_1,\n    tax_c_1,\n    person_1,\n    other_1,\n]\n\n# ---- Depth 2 (aggregate to case_id) ----\ncb_a_2 = read_tables([Path(p) for p in file_map[\"test\"][\"credit_bureau_a_2_parts\"]], depth=2, table_name=\"test_credit_bureau_a_2_parts\")\ncb_b_2 = read_table(Path(file_map[\"test\"][\"credit_bureau_b_2\"]), depth=2, table_name=\"test_credit_bureau_b_2\")\napplprev_2 = read_table(Path(file_map[\"test\"][\"applprev_2\"]), depth=2, table_name=\"test_applprev_2\")\nperson_2 = read_table(Path(file_map[\"test\"][\"person_2\"]), depth=2, table_name=\"test_person_2\")\n\ndepth_2_tables = [cb_a_2, cb_b_2, applprev_2, person_2]\n\n# ---- Build final matrix ----\nfeatures_test_pl = build_feature_matrix(\n    base_table=base_test,\n    depth_0_tables=depth_0_tables,\n    depth_1_tables=depth_1_tables,\n    depth_2_tables=depth_2_tables,\n    config=cfg,\n    is_train=False,\n)\n\nprint(\"Feature matrix (polars) shape:\", features_test_pl.shape)\nprint(\"Columns:\", len(features_test_pl.columns))\n\n# Free intermediate tables early (Kaggle hidden test can be larger)\nimport gc\n\ndel (\n    base_test,\n    static_cb_0,\n    static_0,\n    depth_0_tables,\n    applprev_1,\n    cb_a_1,\n    cb_b_1,\n    deposit_1,\n    debitcard_1,\n    person_1,\n    other_1,\n    tax_a_1,\n    tax_b_1,\n    tax_c_1,\n    depth_1_tables,\n    cb_a_2,\n    cb_b_2,\n    applprev_2,\n    person_2,\n    depth_2_tables,\n)\n\ngc.collect()\n\n# Keep as Polars for memory safety on Kaggle hidden test.\n# We'll do streaming batch inference later, converting only small slices to pandas.\nfeatures_test_pl = features_test_pl.sort(\"case_id\")\n\nassert features_test_pl[\"case_id\"].is_unique\nfeatures_test_pl.head()","metadata":{"execution":{"iopub.status.busy":"2026-03-26T15:57:12.410505Z","iopub.execute_input":"2026-03-26T15:57:12.411045Z","iopub.status.idle":"2026-03-26T15:57:13.211253Z","shell.execute_reply.started":"2026-03-26T15:57:12.411022Z","shell.execute_reply":"2026-03-26T15:57:13.210549Z"},"papermill":{"duration":1.04257,"end_time":"2026-03-26T14:26:35.149586","exception":false,"start_time":"2026-03-26T14:26:34.107016","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4) Local training (split into 3 independent cells)\n\nTraining is split by model so you can iterate safely:\n\n- **LightGBM** training cell\n- **CatBoost** training cell\n- **TabResNet** training cell\n\nEach model writes its own artifacts under `./models/`:\n\n- LightGBM: `lgbm_fold_*.pkl`\n- CatBoost: `catboost_fold_*.cbm`\n- TabResNet: `tabresnet_fold_*.pt`\n\n### Critical safety behavior\n\nBy default, the notebook will **not overwrite** existing artifacts.\nTo overwrite/retrain, you must explicitly set one of the `FORCE_*` flags in the config cell.\n","metadata":{"papermill":{"duration":0.002615,"end_time":"2026-03-26T14:26:35.155114","exception":false,"start_time":"2026-03-26T14:26:35.152499","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# ---- Training data prep (run once; reused by per-model training cells) ----\n# This cell only builds the training matrix + CV splits + metadata.\n# It does NOT train any model by itself.\n\nWILL_TRAIN_ANY = bool(DO_TRAIN_LGBM or DO_TRAIN_CATBOOST or DO_TRAIN_TABRESNET)\n\nif WILL_TRAIN_ANY:\n    model_dir = Path(\"models\")\n    model_dir.mkdir(parents=True, exist_ok=True)\n\n    cfg = FeatureBuildConfig()\n\n    # Base\n    base_train = read_table(Path(file_map[\"train\"][\"base\"]), depth=None, table_name=\"train_base\")\n\n    # Depth 0\n    static_cb_0_tr = read_table(Path(file_map[\"train\"][\"static_cb_0\"]), depth=None, table_name=\"train_static_cb_0\")\n    static_0_tr = read_tables([Path(p) for p in file_map[\"train\"][\"static_0_parts\"]], depth=None, table_name=\"train_static_0_parts\")\n    depth_0_tr = [static_cb_0_tr, static_0_tr]\n\n    # Depth 1\n    applprev_1_tr = read_tables([Path(p) for p in file_map[\"train\"][\"applprev_1_parts\"]], depth=1, table_name=\"train_applprev_1_parts\")\n    cb_a_1_tr = read_tables([Path(p) for p in file_map[\"train\"][\"credit_bureau_a_1_parts\"]], depth=1, table_name=\"train_credit_bureau_a_1_parts\")\n    cb_b_1_tr = read_table(Path(file_map[\"train\"][\"credit_bureau_b_1\"]), depth=1, table_name=\"train_credit_bureau_b_1\")\n    deposit_1_tr = read_table(Path(file_map[\"train\"][\"deposit_1\"]), depth=1, table_name=\"train_deposit_1\")\n    debitcard_1_tr = read_table(Path(file_map[\"train\"][\"debitcard_1\"]), depth=1, table_name=\"train_debitcard_1\")\n    person_1_tr = read_table(Path(file_map[\"train\"][\"person_1\"]), depth=1, table_name=\"train_person_1\")\n    other_1_tr = read_table(Path(file_map[\"train\"][\"other_1\"]), depth=1, table_name=\"train_other_1\")\n    tax_a_1_tr = read_table(Path(file_map[\"train\"][\"tax_registry_a_1\"]), depth=1, table_name=\"train_tax_registry_a_1\")\n    tax_b_1_tr = read_table(Path(file_map[\"train\"][\"tax_registry_b_1\"]), depth=1, table_name=\"train_tax_registry_b_1\")\n    tax_c_1_tr = read_table(Path(file_map[\"train\"][\"tax_registry_c_1\"]), depth=1, table_name=\"train_tax_registry_c_1\")\n\n    depth_1_tr = [\n        applprev_1_tr,\n        cb_a_1_tr,\n        cb_b_1_tr,\n        deposit_1_tr,\n        debitcard_1_tr,\n        tax_a_1_tr,\n        tax_b_1_tr,\n        tax_c_1_tr,\n        person_1_tr,\n        other_1_tr,\n    ]\n\n    # Depth 2\n    cb_a_2_tr = read_tables([Path(p) for p in file_map[\"train\"][\"credit_bureau_a_2_parts\"]], depth=2, table_name=\"train_credit_bureau_a_2_parts\")\n    cb_b_2_tr = read_table(Path(file_map[\"train\"][\"credit_bureau_b_2\"]), depth=2, table_name=\"train_credit_bureau_b_2\")\n    applprev_2_tr = read_table(Path(file_map[\"train\"][\"applprev_2\"]), depth=2, table_name=\"train_applprev_2\")\n    person_2_tr = read_table(Path(file_map[\"train\"][\"person_2\"]), depth=2, table_name=\"train_person_2\")\n\n    depth_2_tr = [cb_a_2_tr, cb_b_2_tr, applprev_2_tr, person_2_tr]\n\n    features_train_pl = build_feature_matrix(\n        base_table=base_train,\n        depth_0_tables=depth_0_tr,\n        depth_1_tables=depth_1_tr,\n        depth_2_tables=depth_2_tr,\n        config=cfg,\n        is_train=True,\n    )\n\n    features_train = features_train_pl.to_pandas()\n\n    y = features_train[\"target\"].to_numpy(dtype=np.int32)\n    groups = features_train[\"WEEK_NUM\"].to_numpy(dtype=np.int32)\n\n    X = features_train.drop(columns=[\"target\"], errors=\"ignore\")\n    cat_cols = infer_categorical_columns(X)\n\n    X_model = X.drop(columns=[\"case_id\"], errors=\"ignore\")\n\n    splits = make_cv_splits_stratified_group_kfold(\n        y=y,\n        groups=groups,\n        n_splits=TRAIN_CFG.n_splits,\n        random_state=TRAIN_CFG.random_state,\n    )\n\n    # Write metadata only if missing OR explicitly forced.\n    meta_files = [\n        model_dir / \"feature_names.json\",\n        model_dir / \"cat_cols.pickle\",\n        model_dir / \"train_config.json\",\n    ]\n    if FORCE_OVERWRITE_METADATA or (not all(p.exists() for p in meta_files)):\n        save_metadata(model_dir=model_dir, feature_names=list(X_model.columns), cat_cols=cat_cols, config=TRAIN_CFG)\n        print(\"Wrote metadata to:\", model_dir.resolve())\n    else:\n        print(\"Metadata already exists; not overwriting.\")\n","metadata":{"execution":{"iopub.status.busy":"2026-03-26T15:57:13.213175Z","iopub.execute_input":"2026-03-26T15:57:13.213749Z","iopub.status.idle":"2026-03-26T15:57:13.225442Z","shell.execute_reply.started":"2026-03-26T15:57:13.213726Z","shell.execute_reply":"2026-03-26T15:57:13.224595Z"},"papermill":{"duration":15.34351,"end_time":"2026-03-26T14:26:50.501419","exception":false,"start_time":"2026-03-26T14:26:35.157909","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ---- Train LightGBM folds (independent) ----\nif DO_TRAIN_LGBM:\n    assert 'X_model' in globals() and 'y' in globals() and 'splits' in globals() and 'model_dir' in globals(), \"Run the training data prep cell first.\"\n\n    existing = sorted(Path(\"models\").glob(\"lgbm_fold_*.pkl\"))\n    if existing and not FORCE_RETRAIN_LGBM:\n        print(f\"Found {len(existing)} existing LightGBM folds; skipping retrain.\")\n    else:\n        lgbm_aucs = train_lightgbm_folds(\n            X=X_model,\n            y=y,\n            splits=splits,\n            model_dir=Path(\"models\"),\n            config=TRAIN_CFG,\n            cat_cols=cat_cols,\n        )\n        print(\"LightGBM fold AUC:\", lgbm_aucs, \"mean=\", float(np.mean(lgbm_aucs)))\n","metadata":{"execution":{"iopub.status.busy":"2026-03-26T15:57:13.226385Z","iopub.execute_input":"2026-03-26T15:57:13.226680Z","iopub.status.idle":"2026-03-26T15:57:13.240046Z","shell.execute_reply.started":"2026-03-26T15:57:13.226659Z","shell.execute_reply":"2026-03-26T15:57:13.239332Z"},"papermill":{"duration":0.474495,"end_time":"2026-03-26T14:26:50.978962","exception":false,"start_time":"2026-03-26T14:26:50.504467","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ---- Train CatBoost folds (independent) ----\n\n(Next cell)","metadata":{"papermill":{"duration":0.002645,"end_time":"2026-03-26T14:26:50.984522","exception":false,"start_time":"2026-03-26T14:26:50.981877","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# ---- Train CatBoost folds (independent) ----\nif DO_TRAIN_CATBOOST:\n    assert 'X_model' in globals() and 'y' in globals() and 'splits' in globals() and 'model_dir' in globals(), \"Run the training data prep cell first.\"\n\n    existing = sorted(Path(\"models\").glob(\"catboost_fold_*.cbm\"))\n    if existing and not FORCE_RETRAIN_CATBOOST:\n        print(f\"Found {len(existing)} existing CatBoost folds; skipping retrain.\")\n    else:\n        cat_aucs = train_catboost_folds(\n            X=X_model,\n            y=y,\n            splits=splits,\n            model_dir=Path(\"models\"),\n            config=TRAIN_CFG,\n            cat_cols=cat_cols,\n        )\n        print(\"CatBoost fold AUC:\", cat_aucs, \"mean=\", float(np.mean(cat_aucs)))\n","metadata":{"execution":{"iopub.status.busy":"2026-03-26T15:57:13.240850Z","iopub.execute_input":"2026-03-26T15:57:13.241060Z","iopub.status.idle":"2026-03-26T15:57:13.249871Z","shell.execute_reply.started":"2026-03-26T15:57:13.241040Z","shell.execute_reply":"2026-03-26T15:57:13.249180Z"},"papermill":{"duration":0.043053,"end_time":"2026-03-26T14:26:51.030224","exception":false,"start_time":"2026-03-26T14:26:50.987171","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ---- Train TabResNet folds (independent; uses MPS on Mac if available) ----\n\n(Next cell)","metadata":{}},{"cell_type":"code","source":"# ---- Train TabResNet folds (independent; uses MPS on Mac if available) ----\nif DO_TRAIN_TABRESNET:\n    import torch\n\n    assert 'X_model' in globals() and 'y' in globals() and 'splits' in globals() and 'model_dir' in globals(), \"Run the training data prep cell first.\"\n\n    existing = sorted(Path(\"models\").glob(\"tabresnet_fold_*.pt\"))\n    if existing and not FORCE_RETRAIN_TABRESNET:\n        print(f\"Found {len(existing)} existing TabResNet folds; skipping retrain.\")\n    else:\n        device = \"mps\" if getattr(torch.backends, \"mps\", None) is not None and torch.backends.mps.is_available() else (\"cuda\" if torch.cuda.is_available() else \"cpu\")\n        print(\"TabResNet device:\", device)\n\n        tab_aucs = train_tabresnet_folds(\n            X=X_model,\n            y=y,\n            splits=splits,\n            model_dir=Path(\"models\"),\n            config=TRAIN_CFG,\n            cat_cols=cat_cols,\n            device=device,\n        )\n        print(\"TabResNet fold AUC:\", tab_aucs, \"mean=\", float(np.mean(tab_aucs)))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T15:57:13.250741Z","iopub.execute_input":"2026-03-26T15:57:13.251023Z","iopub.status.idle":"2026-03-26T15:57:13.261110Z","shell.execute_reply.started":"2026-03-26T15:57:13.250994Z","shell.execute_reply":"2026-03-26T15:57:13.260574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ---- Load artifacts for inference (local or Kaggle dataset) ----\nmodel_root = resolve_model_root(model_root=MODEL_ROOT, kaggle_dataset_slug=KAGGLE_MODELS_DATASET_SLUG)\nprint(\"Resolved model root:\", model_root)\n\nart = load_artifacts(model_root)\nprint(\"Loaded artifacts:\")\nprint(\"- features:\", len(art.feature_names))\nprint(\"- cat cols:\", len(art.cat_cols))\nprint(\"- lgbm folds:\", len(art.lgbm_models))\nprint(\"- catboost folds:\", len(art.catboost_models))\nprint(\"- tabresnet folds:\", len(art.tabresnet_folds))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T15:57:13.261875Z","iopub.execute_input":"2026-03-26T15:57:13.262156Z","iopub.status.idle":"2026-03-26T15:57:14.257672Z","shell.execute_reply.started":"2026-03-26T15:57:13.262128Z","shell.execute_reply":"2026-03-26T15:57:14.257086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if DO_INFER:\n    import gc\n    import time\n    # Basic sanity: you must enable at least one model\n    if not (RUN_LGBM_INFER or RUN_CATBOOST_INFER or RUN_TABRESNET_INFER):\n        raise RuntimeError(\"Enable at least one of RUN_LGBM_INFER / RUN_CATBOOST_INFER / RUN_TABRESNET_INFER\")\n    batch_size = int(PREDICT_CHUNK_SIZE)\n    n = int(features_test_pl.height)\n    all_case_ids: list[np.ndarray] = []\n    lgbm_parts: list[np.ndarray] = []\n    cb_parts: list[np.ndarray] = []\n    tab_parts: list[np.ndarray] = []\n    # Prepare TabResNet folds once (optional)\n    tab_folds_prepared = None\n    tab_device = None\n    if RUN_TABRESNET_INFER and art.tabresnet_folds:\n        import torch\n        from src.home_credit_training import _tabular_to_numpy\n        from src.tabresnet import TabResNet, Standardizer, apply_standardizer, predict_proba as tab_predict_proba\n        tab_device = \"cpu\" if IS_KAGGLE else (\"cuda\" if torch.cuda.is_available() else \"cpu\")\n        tab_folds_prepared = []\n        for f in art.tabresnet_folds:\n            ckpt = f[\"checkpoint\"]\n            model = TabResNet(\n                num_features=int(ckpt[\"num_features\"]),\n                hidden_dim=int(ckpt[\"hidden_dim\"]),\n                num_blocks=int(ckpt[\"num_blocks\"]),\n                dropout=float(ckpt[\"dropout\"]),\n            )\n            model.load_state_dict(ckpt[\"state_dict\"], strict=True)\n            model.to(tab_device)\n            model.eval()\n            s = Standardizer(mean=ckpt[\"standardizer_mean\"], std=ckpt[\"standardizer_std\"])\n            tab_folds_prepared.append((model, s))\n    t0 = time.time()\n    for start in range(0, n, batch_size):\n        end = min(start + batch_size, n)\n        batch_pl = features_test_pl.slice(start, end - start)\n        case_ids = batch_pl[\"case_id\"].to_numpy()\n        X_base = batch_pl.drop(\"case_id\").to_pandas()\n        # Align to the training feature list.\n        X_aligned = align_features_for_model(\n            X_base,\n            feature_names=art.feature_names,\n            categorical_columns=art.cat_cols,\n        )\n        # ---- LightGBM (batched) ----\n        if RUN_LGBM_INFER and art.lgbm_models:\n            X_lgbm = X_aligned.copy()\n            obj_cols = X_lgbm.select_dtypes(include=[\"object\"]).columns.tolist()\n            for c in obj_cols:\n                X_lgbm[c] = pd.to_numeric(X_lgbm[c], errors=\"coerce\")\n            obj_cols2 = X_lgbm.select_dtypes(include=[\"object\"]).columns.tolist()\n            if obj_cols2:\n                X_lgbm[obj_cols2] = X_lgbm[obj_cols2].astype(\"category\")\n            p_lgbm = predict_proba_average(art.lgbm_models, X_lgbm, chunk_size=len(X_lgbm))\n            lgbm_parts.append(p_lgbm)\n            del X_lgbm, p_lgbm\n        # ---- CatBoost (batched) ----\n        if RUN_CATBOOST_INFER and art.catboost_models:\n            X_cb = X_aligned.copy()\n            cb_cat_cols = [c for c in art.cat_cols if c in X_cb.columns]\n            if cb_cat_cols:\n                X_cb[cb_cat_cols] = X_cb[cb_cat_cols].astype(\"string\")\n            p_cb = predict_proba_average(art.catboost_models, X_cb, chunk_size=len(X_cb))\n            cb_parts.append(p_cb)\n            del X_cb, p_cb\n        # ---- TabResNet (batched) ----\n        if tab_folds_prepared is not None:\n            import torch\n            from src.home_credit_training import _tabular_to_numpy\n            from src.tabresnet import apply_standardizer, predict_proba as tab_predict_proba\n            with torch.no_grad():\n                X_np = _tabular_to_numpy(X_aligned, cat_cols=art.cat_cols)\n                fold_ps = []\n                for model, s in tab_folds_prepared:\n                    X_std = apply_standardizer(X_np, s)\n                    p = tab_predict_proba(model, X_std, device=tab_device)\n                    fold_ps.append(p)\n                p_tab = np.mean(np.vstack(fold_ps), axis=0)\n                tab_parts.append(p_tab)\n            del X_np, fold_ps, p_tab\n        all_case_ids.append(case_ids)\n        done = end\n        elapsed = time.time() - t0\n        if done == n or done % (batch_size * 5) == 0:\n            rate = done / max(elapsed, 1e-6)\n            print(f\"Pred progress: {done}/{n} rows ({done/n:.1%}) | {rate:,.0f} rows/s\")\n        del batch_pl, X_base, X_aligned, case_ids\n        gc.collect()\n    case_id_pred = np.concatenate(all_case_ids)\n    preds: dict[str, np.ndarray] = {}\n    if lgbm_parts:\n        preds[\"lgbm\"] = np.concatenate(lgbm_parts)\n    if cb_parts:\n        preds[\"catboost\"] = np.concatenate(cb_parts)\n    if tab_parts:\n        preds[\"tabresnet\"] = np.concatenate(tab_parts)\n    for k, v in preds.items():\n        print(k, v.shape, \"min/max=\", float(np.min(v)), float(np.max(v)))\n    # Blend using direct probabilities (weighted mean)\n    available = [k for k in [\"lgbm\", \"catboost\", \"tabresnet\"] if k in preds]\n    if not available:\n        raise RuntimeError(\"No predictions available for blending\")\n\n    use_dynamic = bool(OPTIMIZE_BLEND_WEIGHTS and {\"X_model\", \"y\", \"splits\"}.issubset(set(globals())))\n\n    if use_dynamic:\n        from sklearn.metrics import roc_auc_score\n\n        # Build OOF predictions from already-trained fold models, then grid-search weights.\n        oof_preds: dict[str, np.ndarray] = {}\n\n        if \"lgbm\" in available and art.lgbm_models:\n            oof = np.zeros(len(y), dtype=np.float64)\n            for fold_idx, (_, va_idx) in enumerate(splits):\n                X_va = X_model.iloc[va_idx].copy()\n                obj_cols = X_va.select_dtypes(include=[\"object\"]).columns.tolist()\n                for c in obj_cols:\n                    X_va[c] = pd.to_numeric(X_va[c], errors=\"coerce\")\n                obj_cols2 = X_va.select_dtypes(include=[\"object\"]).columns.tolist()\n                if obj_cols2:\n                    X_va[obj_cols2] = X_va[obj_cols2].astype(\"category\")\n                oof[va_idx] = art.lgbm_models[fold_idx].predict_proba(X_va)[:, 1]\n            oof_preds[\"lgbm\"] = oof\n\n        if \"catboost\" in available and art.catboost_models:\n            oof = np.zeros(len(y), dtype=np.float64)\n            for fold_idx, (_, va_idx) in enumerate(splits):\n                X_va = X_model.iloc[va_idx].copy()\n                cb_cat_cols = [c for c in art.cat_cols if c in X_va.columns]\n                if cb_cat_cols:\n                    X_va[cb_cat_cols] = X_va[cb_cat_cols].astype(\"string\")\n                oof[va_idx] = art.catboost_models[fold_idx].predict_proba(X_va)[:, 1]\n            oof_preds[\"catboost\"] = oof\n\n        # TabResNet OOF optimization is expensive; keep test-time tabresnet but skip it in OOF search.\n        if \"tabresnet\" in available and art.tabresnet_folds:\n            print(\"Skipping TabResNet in dynamic OOF weight optimization for speed; using static weight fallback.\")\n\n        opt_models = [m for m in [\"lgbm\", \"catboost\"] if m in oof_preds]\n\n        if len(opt_models) >= 2:\n            step = float(BLEND_GRID_STEP)\n            grid = np.arange(0.0, 1.0 + 1e-12, step)\n            best_auc = -np.inf\n            best_w = None\n\n            # 2-model grid for robust and fast search\n            for w0 in grid:\n                w1 = 1.0 - w0\n                ws = np.array([w0, w1], dtype=np.float64)\n                blend = ws[0] * oof_preds[opt_models[0]] + ws[1] * oof_preds[opt_models[1]]\n                auc = roc_auc_score(y, blend)\n                if auc > best_auc:\n                    best_auc = auc\n                    best_w = ws\n\n            learned_weights = {k: 0.0 for k in available}\n            learned_weights[opt_models[0]] = float(best_w[0])\n            learned_weights[opt_models[1]] = float(best_w[1])\n\n            # If tabresnet is enabled, keep a small fallback share from static weights.\n            if \"tabresnet\" in available:\n                tab_w = float(BLEND_WEIGHTS.get(\"tabresnet\", 0.0))\n                tab_w = max(tab_w, 0.0)\n                base = 1.0 - min(tab_w, 0.3)\n                learned_weights[opt_models[0]] *= base\n                learned_weights[opt_models[1]] *= base\n                learned_weights[\"tabresnet\"] = 1.0 - base\n\n            print(\"Dynamic OOF AUC:\", float(best_auc), \"weights:\", learned_weights)\n        else:\n            learned_weights = {k: float(BLEND_WEIGHTS.get(k, 0.0)) for k in available}\n            print(\"Dynamic weighting requested but insufficient OOF models; fallback weights:\", learned_weights)\n\n        w = np.array([learned_weights[k] for k in available], dtype=np.float64)\n    else:\n        w = np.array([float(BLEND_WEIGHTS.get(k, 0.0)) for k in available], dtype=np.float64)\n\n    if np.any(w < 0) or np.sum(w) <= 0:\n        raise ValueError(f\"Invalid blend weights for available models: {available}\")\n\n    w = w / np.sum(w)\n\n    y_pred = np.zeros_like(next(iter(preds.values())), dtype=np.float64)\n    for k, wk in zip(available, w):\n        y_pred += wk * preds[k]\n\n    print(\"Blend models:\", available, \"weights:\", {k: float(v) for k, v in zip(available, w)})\n    print(\"Blended:\", y_pred.shape, \"min/max=\", float(np.min(y_pred)), float(np.max(y_pred)))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T15:57:14.258702Z","iopub.execute_input":"2026-03-26T15:57:14.259181Z","iopub.status.idle":"2026-03-26T15:57:14.896910Z","shell.execute_reply.started":"2026-03-26T15:57:14.259149Z","shell.execute_reply":"2026-03-26T15:57:14.896121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub = pd.read_csv(paths.sample_submission_csv)\n\npred_df = pd.DataFrame({\"case_id\": case_id_pred, \"score\": y_pred})\nsub = sample_sub[[\"case_id\"]].merge(pred_df, on=\"case_id\", how=\"left\")\n\nassert sub[\"score\"].notna().all(), \"Some case_id in sample submission missing predictions\"\n\nout_path = Path(\"/kaggle/working/submission.csv\")\nsub.to_csv(out_path, index=False)\nprint(\"Wrote:\", out_path)\nsub.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T15:57:14.899203Z","iopub.execute_input":"2026-03-26T15:57:14.899425Z","iopub.status.idle":"2026-03-26T15:57:14.914833Z","shell.execute_reply.started":"2026-03-26T15:57:14.899406Z","shell.execute_reply":"2026-03-26T15:57:14.914137Z"}},"outputs":[],"execution_count":null}]}