{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":50160,"databundleVersionId":7921029}],"dockerImageVersionId":31286,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nimport json\nwarnings.simplefilter(action=\"ignore\", category=FutureWarning)\n\nimport gc\nfrom glob import glob\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-24T09:59:16.415547Z","iopub.execute_input":"2026-03-24T09:59:16.415883Z","iopub.status.idle":"2026-03-24T09:59:18.207113Z","shell.execute_reply.started":"2026-03-24T09:59:16.415845Z","shell.execute_reply":"2026-03-24T09:59:18.206234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# Part 1 — Prepare functions for data preprocessing\n# ============================================\n\nimport os\nos.environ[\"OMP_NUM_THREADS\"] = \"4\"\nos.environ[\"OPENBLAS_NUM_THREADS\"] = \"4\"\nos.environ[\"MKL_NUM_THREADS\"] = \"4\"\nos.environ[\"VECLIB_MAXIMUM_THREADS\"] = \"4\"\nos.environ[\"NUMEXPR_NUM_THREADS\"] = \"4\"\n\nimport gc\nfrom pathlib import Path\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nfrom catboost import CatBoostClassifier, Pool\nfrom sklearn.metrics import roc_auc_score\n\n\ndef set_table_dtypes(df: pl.DataFrame) -> pl.DataFrame:\n    \"\"\"\n    Cast columns by suffix convention.\n    Use lightweight dtypes where possible.\n    \"\"\"\n    for col in df.columns:\n        if col == \"case_id\":\n            continue\n\n        suf = col[-1]\n\n        if suf in (\"P\", \"A\"):\n            df = df.with_columns(pl.col(col).cast(pl.Float32))\n\n        elif suf in (\"L\", \"M\", \"T\"):\n            df = df.with_columns(pl.col(col).cast(pl.Utf8))\n\n        elif suf == \"D\":\n            # Keep as Utf8 here because this specific pipeline\n            # does not rely on date arithmetic.\n            df = df.with_columns(pl.col(col).cast(pl.Utf8))\n\n    return df\n\n\ndef select_static_cols(df: pl.DataFrame, max_cat: int = 120) -> list[str]:\n    \"\"\"\n    Keep all numeric static columns (A/P) and only a limited number\n    of categorical static columns (L/M/T) to control memory.\n    \"\"\"\n    cols = []\n    cat_cnt = 0\n\n    for c in df.columns:\n        if c == \"case_id\":\n            continue\n\n        if c[-1] in (\"A\", \"P\"):\n            cols.append(c)\n\n        elif c[-1] in (\"L\", \"M\", \"T\") and cat_cnt < max_cat:\n            cols.append(c)\n            cat_cnt += 1\n\n    return cols\n\n\ndef build_person1_features(df: pl.DataFrame) -> tuple[pl.DataFrame, pl.DataFrame]:\n    \"\"\"\n    Build two small but useful feature tables from person_1:\n    1) max income + self-employed flag\n    2) applicant house type (num_group1 == 0)\n    \"\"\"\n    feats_1 = df.group_by(\"case_id\").agg(\n        pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_384A_max\"),\n        (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").cast(pl.Int8).max().alias(\"any_selfemployed\"),\n    )\n\n    feats_2 = (\n        df.select([\"case_id\", \"num_group1\", \"housetype_905L\"])\n        .filter(pl.col(\"num_group1\") == 0)\n        .drop(\"num_group1\")\n        .rename({\"housetype_905L\": \"person_housetype\"})\n    )\n\n    return feats_1, feats_2\n\n\ndef build_cb_b2_features(df: pl.DataFrame) -> pl.DataFrame:\n    \"\"\"\n    Build two lightweight but useful features from credit_bureau_b_2.\n    \"\"\"\n    return df.group_by(\"case_id\").agg(\n        pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_635A_max\"),\n        (pl.col(\"pmts_dpdvalue_108P\") > 31).cast(pl.Int8).max().alias(\"pmts_dpdvalue_108P_over31\"),\n    )\n\n\ndef transform_features_pandas(\n    df: pd.DataFrame,\n    amount_cols: list[str],\n    dpd_cols: list[str],\n) -> pd.DataFrame:\n    \"\"\"\n    Apply the same feature engineering to train/valid/test batches.\n    This function avoids repeated column insertion by using concat once.\n    \"\"\"\n    out = df.copy()\n\n    log_block = {\n        col + \"_log\": np.log1p(out[col].clip(lower=0))\n        for col in amount_cols if col in out.columns\n    }\n\n    late_block = {\n        col + \"_late\": (out[col] > 0).astype(np.int8)\n        for col in dpd_cols if col in out.columns\n    }\n\n    week_block = {}\n    if \"WEEK_NUM\" in out.columns:\n        week_block = {\n            \"week_sin\": np.sin(out[\"WEEK_NUM\"] * 2 * np.pi / 52),\n            \"week_cos\": np.cos(out[\"WEEK_NUM\"] * 2 * np.pi / 52),\n        }\n\n    out = pd.concat(\n        [\n            out,\n            pd.DataFrame(log_block, index=out.index),\n            pd.DataFrame(late_block, index=out.index),\n            pd.DataFrame(week_block, index=out.index),\n        ],\n        axis=1,\n    )\n\n    out = out.copy()  # defragment memory\n    return out\n\n\ndef prepare_catboost_dataframe(df: pd.DataFrame, cat_cols: list[str]) -> pd.DataFrame:\n    \"\"\"\n    Fill missing categorical values for CatBoost and keep numeric columns as-is.\n    \"\"\"\n    out = df.copy()\n\n    for c in cat_cols:\n        if c in out.columns:\n            out[c] = out[c].astype(\"string\").fillna(\"Unknown\")\n\n    return out\n\n\ndef predict_in_batches_catboost(\n    model: CatBoostClassifier,\n    df_pl: pl.DataFrame,\n    feature_cols_before_fe: list[str],\n    amount_cols: list[str],\n    dpd_cols: list[str],\n    final_model_cols: list[str],\n    cat_cols_final: list[str],\n    batch_size: int = 50000,\n) -> np.ndarray:\n    \"\"\"\n    Submission-safe batch prediction on Polars test data.\n    Keeps hidden test from blowing up memory.\n    \"\"\"\n    n = df_pl.height\n    preds = np.zeros(n, dtype=np.float32)\n\n    cat_idx_final = [final_model_cols.index(c) for c in cat_cols_final]\n\n    for start in range(0, n, batch_size):\n        end = min(start + batch_size, n)\n\n        batch_pl = df_pl.slice(start, end - start)\n        batch_pd = batch_pl.select([\"case_id\", \"WEEK_NUM\"] + feature_cols_before_fe).to_pandas()\n\n        batch_pd = transform_features_pandas(batch_pd, amount_cols, dpd_cols)\n\n        # keep only model columns\n        batch_X = batch_pd[final_model_cols]\n        batch_X = prepare_catboost_dataframe(batch_X, cat_cols_final)\n\n        batch_pool = Pool(batch_X, cat_features=cat_idx_final)\n        preds[start:end] = model.predict_proba(batch_pool)[:, 1]\n\n        del batch_pl, batch_pd, batch_X, batch_pool\n        gc.collect()\n\n    return preds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T09:59:18.209082Z","iopub.execute_input":"2026-03-24T09:59:18.209495Z","iopub.status.idle":"2026-03-24T09:59:20.605574Z","shell.execute_reply.started":"2026-03-24T09:59:18.209465Z","shell.execute_reply":"2026-03-24T09:59:20.604577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# Part 2 — Load data sets\n# ============================================\n\nROOT = Path(\"/kaggle/input/competitions/home-credit-credit-risk-model-stability\")\nTRAIN_DIR = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR  = ROOT / \"parquet_files\" / \"test\"\n\nprint(\"ROOT exists:\", ROOT.exists())\nprint(\"TRAIN_DIR exists:\", TRAIN_DIR.exists())\nprint(\"TEST_DIR exists:\", TEST_DIR.exists())\n\n\n# -------------------------\n# Load base + static in Polars\n# -------------------------\n\ntrain_base = pl.read_parquet(TRAIN_DIR / \"train_base.parquet\")\ntest_base  = pl.read_parquet(TEST_DIR  / \"test_base.parquet\")\n\ntrain_static = pl.concat(\n    [\n        pl.read_parquet(TRAIN_DIR / \"train_static_0_0.parquet\").pipe(set_table_dtypes),\n        pl.read_parquet(TRAIN_DIR / \"train_static_0_1.parquet\").pipe(set_table_dtypes),\n    ],\n    how=\"vertical_relaxed\",\n)\n\ntest_static = pl.concat(\n    [\n        pl.read_parquet(TEST_DIR / \"test_static_0_0.parquet\").pipe(set_table_dtypes),\n        pl.read_parquet(TEST_DIR / \"test_static_0_1.parquet\").pipe(set_table_dtypes),\n        pl.read_parquet(TEST_DIR / \"test_static_0_2.parquet\").pipe(set_table_dtypes),\n    ],\n    how=\"vertical_relaxed\",\n)\n\ntrain_static_cb = pl.read_parquet(TRAIN_DIR / \"train_static_cb_0.parquet\").pipe(set_table_dtypes)\ntest_static_cb  = pl.read_parquet(TEST_DIR  / \"test_static_cb_0.parquet\").pipe(set_table_dtypes)\n\nselected_static_cols = select_static_cols(train_static, max_cat=120)\nselected_static_cb_cols = select_static_cols(train_static_cb, max_cat=120)\n\ntrain_static = train_static.select([\"case_id\"] + selected_static_cols)\ntest_static  = test_static.select([\"case_id\"] + selected_static_cols)\n\ntrain_static_cb = train_static_cb.select([\"case_id\"] + selected_static_cb_cols)\ntest_static_cb  = test_static_cb.select([\"case_id\"] + selected_static_cb_cols)\n\n\n# -------------------------\n# Load small depth features\n# -------------------------\n\ntrain_person_1 = pl.read_parquet(TRAIN_DIR / \"train_person_1.parquet\").pipe(set_table_dtypes)\ntest_person_1  = pl.read_parquet(TEST_DIR  / \"test_person_1.parquet\").pipe(set_table_dtypes)\n\ntrain_person_1_feats_1, train_person_1_feats_2 = build_person1_features(train_person_1)\ntest_person_1_feats_1, test_person_1_feats_2 = build_person1_features(test_person_1)\n\ndel train_person_1, test_person_1\ngc.collect()\n\ntrain_cb_b2 = pl.read_parquet(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\").pipe(set_table_dtypes)\ntest_cb_b2  = pl.read_parquet(TEST_DIR  / \"test_credit_bureau_b_2.parquet\").pipe(set_table_dtypes)\n\ntrain_cb_b2_feats = build_cb_b2_features(train_cb_b2)\ntest_cb_b2_feats  = build_cb_b2_features(test_cb_b2)\n\ndel train_cb_b2, test_cb_b2\ngc.collect()\n\n\n# -------------------------\n# Join all in Polars\n# -------------------------\n\ntrain_pl = (\n    train_base\n    .join(train_static, how=\"left\", on=\"case_id\")\n    .join(train_static_cb, how=\"left\", on=\"case_id\")\n    .join(train_person_1_feats_1, how=\"left\", on=\"case_id\")\n    .join(train_person_1_feats_2, how=\"left\", on=\"case_id\")\n    .join(train_cb_b2_feats, how=\"left\", on=\"case_id\")\n)\n\ntest_pl = (\n    test_base\n    .join(test_static, how=\"left\", on=\"case_id\")\n    .join(test_static_cb, how=\"left\", on=\"case_id\")\n    .join(test_person_1_feats_1, how=\"left\", on=\"case_id\")\n    .join(test_person_1_feats_2, how=\"left\", on=\"case_id\")\n    .join(test_cb_b2_feats, how=\"left\", on=\"case_id\")\n)\n\nprint(\"train_pl shape:\", train_pl.shape)\nprint(\"test_pl shape:\", test_pl.shape)\n\ndel train_base, test_base, train_static, test_static, train_static_cb, test_static_cb\ndel train_person_1_feats_1, train_person_1_feats_2, test_person_1_feats_1, test_person_1_feats_2\ndel train_cb_b2_feats, test_cb_b2_feats\ngc.collect()\n\n\n# -------------------------\n# Convert TRAIN to pandas only\n# Keep TEST in Polars for submit-safe inference later\n# -------------------------\n\nnon_features = {\"case_id\", \"WEEK_NUM\", \"target\", \"date_decision\"}\nfeature_cols_before_fe = [c for c in train_pl.columns if c not in non_features]\n\ntrain_df = train_pl.select([\"case_id\", \"WEEK_NUM\", \"target\"] + feature_cols_before_fe).to_pandas()\n\nprint(\"train_df shape:\", train_df.shape)\nprint(\"test_pl shape:\", test_pl.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T09:59:20.607043Z","iopub.execute_input":"2026-03-24T09:59:20.607640Z","iopub.status.idle":"2026-03-24T09:59:42.933933Z","shell.execute_reply.started":"2026-03-24T09:59:20.607596Z","shell.execute_reply":"2026-03-24T09:59:42.933151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# Part 3 — EDA and Feature Engineering\n# ============================================\n\nprint(\"Train shape:\", train_df.shape)\nprint(\"\\nColumns preview:\")\nprint(train_df.columns[:20])\n\nprint(\"\\nTarget distribution:\")\nprint(train_df[\"target\"].value_counts())\n\nprint(\"\\nDefault rate:\")\nprint(train_df[\"target\"].mean())\n\n\n# -------------------------\n# Decide engineered columns\n# -------------------------\n\namount_cols = [c for c in feature_cols_before_fe if c.endswith(\"A\")]\ndpd_cols = [c for c in feature_cols_before_fe if c.endswith(\"P\")]\n\n# To keep feature count controlled, cap them\namount_cols = amount_cols[:120]\ndpd_cols = dpd_cols[:120]\n\nprint(\"\\nNumber of amount cols used for log features:\", len(amount_cols))\nprint(\"Number of dpd cols used for late flags:\", len(dpd_cols))\n\n\n# -------------------------\n# Apply FE to train only here\n# -------------------------\n\ntrain_df = transform_features_pandas(train_df, amount_cols, dpd_cols)\n\nprint(\"\\nAfter feature engineering, train shape:\", train_df.shape)\n\n\n# -------------------------\n# Light EDA\n# -------------------------\n\nmissing_ratio = train_df.isna().mean().sort_values(ascending=False)\nprint(\"\\nTop 20 missing features:\")\nprint(missing_ratio.head(20))\n\ncat_cols_after_fe = train_df.select_dtypes(include=[\"object\", \"string\", \"category\"]).columns.tolist()\nprint(\"\\nNumber of categorical features after FE:\", len(cat_cols_after_fe))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T09:59:42.935047Z","iopub.execute_input":"2026-03-24T09:59:42.935795Z","iopub.status.idle":"2026-03-24T10:00:04.741937Z","shell.execute_reply.started":"2026-03-24T09:59:42.935764Z","shell.execute_reply":"2026-03-24T10:00:04.741037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# Part 4 — Train Models\n# ============================================\n\nX_all = train_df.drop(columns=[\"target\", \"case_id\"]).copy()\ny_all = train_df[\"target\"].astype(int).copy()\n\nweeks = train_df[\"WEEK_NUM\"].copy()\n\n# Time split\nunique_weeks = np.sort(weeks.unique())\ncut = unique_weeks[int(len(unique_weeks) * 0.8)]\n\ntr_idx = weeks < cut\nva_idx = ~tr_idx\n\nX_train = X_all.loc[tr_idx].copy()\ny_train = y_all.loc[tr_idx].copy()\n\nX_valid = X_all.loc[va_idx].copy()\ny_valid = y_all.loc[va_idx].copy()\n\nprint(\"X_train shape:\", X_train.shape)\nprint(\"X_valid shape:\", X_valid.shape)\n\n\n# Keep the final model columns\nfinal_model_cols = X_train.columns.tolist()\n\n# CatBoost categorical columns\ncat_cols_final = [c for c in final_model_cols if str(X_train[c].dtype) in [\"object\", \"string\", \"category\"]]\n\nX_train = prepare_catboost_dataframe(X_train, cat_cols_final)\nX_valid = prepare_catboost_dataframe(X_valid, cat_cols_final)\n\ncat_idx_final = [final_model_cols.index(c) for c in cat_cols_final]\n\ntrain_pool = Pool(X_train, y_train, cat_features=cat_idx_final)\nvalid_pool = Pool(X_valid, y_valid, cat_features=cat_idx_final)\n\nmodel = CatBoostClassifier(\n    iterations=300,\n    learning_rate=0.06,\n    depth=5,\n    loss_function=\"Logloss\",\n    eval_metric=\"AUC\",\n    random_seed=42,\n    l2_leaf_reg=5.0,\n    subsample=0.7,\n    rsm=0.7,\n    od_type=\"Iter\",\n    od_wait=100,\n    verbose=100,\n    allow_writing_files=False,\n)\n\nmodel.fit(train_pool, eval_set=valid_pool, use_best_model=True)\n\nvalid_pred = model.predict_proba(valid_pool)[:, 1]\nprint(\"Valid AUC:\", roc_auc_score(y_valid, valid_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T10:00:04.743171Z","iopub.execute_input":"2026-03-24T10:00:04.744177Z","iopub.status.idle":"2026-03-24T11:01:43.486277Z","shell.execute_reply.started":"2026-03-24T10:00:04.744132Z","shell.execute_reply":"2026-03-24T11:01:43.484848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# Part 5 — CatBoost for submissions\n# ============================================\n\n# IMPORTANT:\n# test_pl is still Polars here, which is much safer for hidden test.\n\ntest_case_ids = test_pl.select(\"case_id\").to_pandas()[\"case_id\"].values\n\ntest_pred = predict_in_batches_catboost(\n    model=model,\n    df_pl=test_pl,\n    feature_cols_before_fe=feature_cols_before_fe,\n    amount_cols=amount_cols,\n    dpd_cols=dpd_cols,\n    final_model_cols=final_model_cols,\n    cat_cols_final=cat_cols_final,\n    batch_size=50000,\n)\n\nsubmission = pd.DataFrame({\n    \"case_id\": test_case_ids,\n    \"score\": test_pred,\n})\n\nprint(submission.head())\nprint(\"Submission shape:\", submission.shape)\n\nsubmission.to_csv(\"submission.csv\", index=False)\nprint(\"Saved submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T11:01:43.488999Z","iopub.execute_input":"2026-03-24T11:01:43.489927Z","iopub.status.idle":"2026-03-24T11:01:43.496453Z","shell.execute_reply.started":"2026-03-24T11:01:43.489870Z","shell.execute_reply":"2026-03-24T11:01:43.495209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# Save artifacts\n# ============================================\n\nSAVE_DIR = \"/kaggle/working/cat_model_artifacts_1\"\nos.makedirs(SAVE_DIR, exist_ok=True)\n\n# 1. save model\nmodel.save_model(f\"{SAVE_DIR}/cat_model.cbm\")\n\n# 2. save feature column order\nwith open(f\"{SAVE_DIR}/final_model_cols.json\", \"w\") as f:\n    json.dump(final_model_cols, f)\n\n# 3. save categorical columns\nwith open(f\"{SAVE_DIR}/cat_cols_final.json\", \"w\") as f:\n    json.dump(cat_cols_final, f)\n\nprint(\"Saved files:\", os.listdir(SAVE_DIR))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T11:01:43.499312Z","iopub.execute_input":"2026-03-24T11:01:43.499620Z","iopub.status.idle":"2026-03-24T11:01:44.037402Z","shell.execute_reply.started":"2026-03-24T11:01:43.499594Z","shell.execute_reply":"2026-03-24T11:01:44.036467Z"}},"outputs":[],"execution_count":null}]}