{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":50160,"databundleVersionId":7921029,"isSourceIdPinned":false},{"sourceType":"modelInstanceVersion","sourceId":791766,"databundleVersionId":16163319,"modelInstanceId":604321,"modelId":616421,"isSourceIdPinned":false}],"dockerImageVersionId":31286,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Home Credit - Credit Risk Model Stability\nPipeline: load data, aggregate features, train LightGBM, produce submission.\nRuns locally or on Kaggle (paths auto-detect).","metadata":{}},{"cell_type":"code","source":"# On Kaggle: discover actual file layout (run this cell first)\nimport os\nfrom pathlib import Path\n\nif Path(\"/kaggle/input\").exists():\n    # Walk entire /kaggle/input in case data is in a different subfolder\n    input_dir = Path(\"/kaggle/input\")\n    all_files = []\n    for root, _, files in os.walk(input_dir):\n        for f in files:\n            if f.endswith((\".csv\", \".parquet\")):\n                all_files.append(Path(root) / f)\n\n    print(\"Top-level folders:\", os.listdir(input_dir))\n    print(\"Total CSV/Parquet files:\", len(all_files))\n    if all_files:\n        print(\"First 40 files:\")\n        for p in sorted(all_files)[:40]:\n            print(\" \", p)\n\n    BASE = Path(\"/kaggle/input/competition\")\n    train_base_path = test_base_path = submission_path = None\n    for p in all_files:\n        name = p.stem.lower()\n        if (\"train\" in name and \"base\" in name) or name == \"train\":\n            train_base_path = p\n        elif (\"test\" in name and \"base\" in name) or name == \"test\":\n            test_base_path = p\n        elif \"sample_submission\" in name or \"sample\" in name:\n            submission_path = p\n\n    if train_base_path is None:\n        for p in all_files:\n            if \"train\" in p.stem.lower():\n                train_base_path = p\n                break\n    if test_base_path is None:\n        for p in all_files:\n            if \"test\" in p.stem.lower() and \"submission\" not in p.stem.lower():\n                test_base_path = p\n                break\n    if submission_path is None:\n        for p in all_files:\n            if \"sample\" in p.name.lower():\n                submission_path = p\n                break\n\n    TRAIN_PATH = train_base_path.parent if train_base_path else BASE\n    TEST_PATH = test_base_path.parent if test_base_path else BASE\n    SUBMISSION_PATH = submission_path or (BASE / \"sample_submission.csv\")\n    OUT_DIR = Path(\"/kaggle/working\")\n\n    print(\"Discovered:\")\n    print(\"  train_base:\", train_base_path)\n    print(\"  test_base:\", test_base_path)\n    print(\"  submission:\", SUBMISSION_PATH)\n    if not train_base_path:\n        print(\"  All CSV/Parquet files found:\")\n        for f in sorted(all_files)[:30]:\n            print(\"    \", f)\nelse:\n    BASE = Path(\"home-credit-credit-risk-model-stability\")\n    TRAIN_PATH = BASE / \"csv_files\" / \"train\"\n    TEST_PATH = BASE / \"csv_files\" / \"test\"\n    SUBMISSION_PATH = BASE / \"sample_submission.csv\"\n    OUT_DIR = Path(\".\")\n    train_base_path = None\n    test_base_path = None\n\nTARGET = \"target\"\nID_COL = \"case_id\"\nFILLNA_VALUE = -999.0\n# On Kaggle: fewer aggregates + limit extra tables to avoid OOM\nAGGS = [\"mean\", \"max\"] if Path(\"/kaggle/input\").exists() else [\"mean\", \"max\", \"min\", \"std\"]\nMAX_EXTRA_TABLES = 12 if Path(\"/kaggle/input\").exists() else None  # None = process all","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-18T20:22:23.964007Z","iopub.execute_input":"2026-03-18T20:22:23.964237Z","iopub.status.idle":"2026-03-18T20:22:24.279441Z","shell.execute_reply.started":"2026-03-18T20:22:23.964208Z","shell.execute_reply":"2026-03-18T20:22:24.278606Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\nimport os\nimport time\nfrom pathlib import Path\n\nimport lightgbm as lgb\nimport numpy as np\nimport pandas as pd\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import StratifiedKFold\n\n# On Kaggle: reduce memory usage to avoid OOM\nON_KAGGLE = Path(\"/kaggle/input\").exists()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-18T20:22:24.281052Z","iopub.execute_input":"2026-03-18T20:22:24.281294Z","iopub.status.idle":"2026-03-18T20:22:30.703507Z","shell.execute_reply.started":"2026-03-18T20:22:24.281270Z","shell.execute_reply":"2026-03-18T20:22:30.702439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reduce_mem(df):\n    for c in df.columns:\n        if c == ID_COL:\n            continue\n        t = df[c].dtype\n        if t == np.float64:\n            df[c] = df[c].astype(np.float32)\n        elif t == np.int64:\n            df[c] = pd.to_numeric(df[c], downcast=\"integer\")\n    return df\n\ndef resolve_table_path(directory, stem):\n    \"\"\"Find a table either directly in directory or nested under it.\"\"\"\n    directory = Path(directory)\n\n    for ext in (\".parquet\", \".csv\"):\n        p = directory / f\"{stem}{ext}\"\n        if p.exists():\n            return p\n\n    # Kaggle can mount competition data under nested folders.\n    # Fall back to a recursive search.\n    for ext in (\".parquet\", \".csv\"):\n        matches = sorted(directory.rglob(f\"{stem}{ext}\"))\n        if matches:\n            return matches[0]\n\n    raise FileNotFoundError(\n        f\"No {stem}.parquet or {stem}.csv found under {directory} (including subfolders)\"\n    )\n\ndef read_table(path):\n    path = Path(path)\n    if path.suffix == \".parquet\":\n        return pd.read_parquet(path)\n    return pd.read_csv(path, low_memory=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-18T20:22:30.708801Z","iopub.execute_input":"2026-03-18T20:22:30.709121Z","iopub.status.idle":"2026-03-18T20:22:30.719488Z","shell.execute_reply.started":"2026-03-18T20:22:30.709091Z","shell.execute_reply":"2026-03-18T20:22:30.718155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Loading base tables...\")\nif train_base_path is None or test_base_path is None:\n    train_base_path = resolve_table_path(TRAIN_PATH, \"train_base\")\n    test_base_path = resolve_table_path(TEST_PATH, \"test_base\")\n\ntrain = read_table(train_base_path)\ntest = read_table(test_base_path)\nprint(f\"Base train: {train.shape} | Base test: {test.shape}\")\n\ny = train[TARGET].astype(np.int8).values\ntrain = train.drop(columns=[TARGET])\nnumeric_feats = [c for c in train.columns if c != ID_COL and pd.api.types.is_numeric_dtype(train[c])]\ntrain = reduce_mem(train)\ntest = reduce_mem(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-18T20:22:30.720711Z","iopub.execute_input":"2026-03-18T20:22:30.721043Z","iopub.status.idle":"2026-03-18T20:22:31.688140Z","shell.execute_reply.started":"2026-03-18T20:22:30.721015Z","shell.execute_reply":"2026-03-18T20:22:31.687321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def aggregate_table(train_file, test_file):\n    tr = read_table(train_file)\n    te = read_table(test_file)\n    num_cols = [c for c in tr.columns if c != ID_COL and pd.api.types.is_numeric_dtype(tr[c])]\n    if len(num_cols) == 0:\n        return tr[[ID_COL]].drop_duplicates(), te[[ID_COL]].drop_duplicates(), []\n    agg_dict = {c: AGGS for c in num_cols}\n    tr_agg = tr.groupby(ID_COL).agg(agg_dict)\n    te_agg = te.groupby(ID_COL).agg(agg_dict)\n    del tr, te\n    if ON_KAGGLE:\n        gc.collect()\n    prefix = Path(train_file).stem\n    new_cols = [f\"{prefix}__{col}__{stat}\" for (col, stat) in tr_agg.columns]\n    tr_agg.columns = te_agg.columns = new_cols\n    tr_agg = tr_agg.reset_index()\n    te_agg = te_agg.reset_index()\n    tr_agg = reduce_mem(tr_agg)\n    te_agg = reduce_mem(te_agg)\n    return tr_agg, te_agg, new_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-18T20:22:31.688991Z","iopub.execute_input":"2026-03-18T20:22:31.689227Z","iopub.status.idle":"2026-03-18T20:22:31.696120Z","shell.execute_reply.started":"2026-03-18T20:22:31.689203Z","shell.execute_reply":"2026-03-18T20:22:31.695321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base_ext = train_base_path.suffix\ntrain_files = sorted(TRAIN_PATH.glob(f\"*{base_ext}\"))\ntrain_files = [f for f in train_files if f.name != train_base_path.name]\nif MAX_EXTRA_TABLES:\n    train_files = train_files[:MAX_EXTRA_TABLES]\n    print(f\"Extra tables (limited to {MAX_EXTRA_TABLES}): {len(train_files)}\")\nelse:\n    print(f\"Extra tables: {len(train_files)}\")\n\nfor i, tr_file in enumerate(train_files, 1):\n    te_file = TEST_PATH / tr_file.name.replace(\"train_\", \"test_\")\n    if not te_file.exists():\n        continue\n    tr_agg, te_agg, new_cols = aggregate_table(tr_file, te_file)\n    train = train.merge(tr_agg, on=ID_COL, how=\"left\")\n    test = test.merge(te_agg, on=ID_COL, how=\"left\")\n    numeric_feats.extend(new_cols)\n    del tr_agg, te_agg\n    if i % 2 == 0 and ON_KAGGLE:\n        gc.collect()\n    if i % 3 == 0:\n        train, test = train.copy(), test.copy()\n    print(f\"  [{i}/{len(train_files)}] {tr_file.name}\")\n\nif ON_KAGGLE:\n    gc.collect()\nprint(f\"Train: {train.shape} | Test: {test.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-18T20:22:31.698838Z","iopub.execute_input":"2026-03-18T20:22:31.699188Z","iopub.status.idle":"2026-03-18T20:28:31.113492Z","shell.execute_reply.started":"2026-03-18T20:22:31.699114Z","shell.execute_reply":"2026-03-18T20:28:31.112029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train[numeric_feats].astype(np.float32).fillna(FILLNA_VALUE)\nX_test = test[numeric_feats].astype(np.float32).fillna(FILLNA_VALUE)\ndel train, test\nif ON_KAGGLE:\n    gc.collect()\nprint(f\"X: {X.shape} | X_test: {X_test.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-18T20:28:31.115003Z","iopub.execute_input":"2026-03-18T20:28:31.115266Z","iopub.status.idle":"2026-03-18T20:28:36.799307Z","shell.execute_reply.started":"2026-03-18T20:28:31.115247Z","shell.execute_reply":"2026-03-18T20:28:36.797283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"n_splits = 3 if ON_KAGGLE else 5\nkf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)\noof = np.zeros(len(X), dtype=np.float32)\npred = np.zeros(len(X_test), dtype=np.float32)\n\nfor fold, (tr_idx, va_idx) in enumerate(kf.split(X, y), 1):\n    X_tr, y_tr = X.iloc[tr_idx], y[tr_idx]\n    X_va, y_va = X.iloc[va_idx], y[va_idx]\n    model = lgb.LGBMClassifier(\n        n_estimators=1500, learning_rate=0.05, num_leaves=31,\n        subsample=0.8, colsample_bytree=0.8, reg_lambda=1.0,\n        random_state=42, n_jobs=-1, max_bin=255\n    )\n    model.fit(X_tr, y_tr, eval_set=[(X_va, y_va)], eval_metric=\"auc\",\n              callbacks=[lgb.early_stopping(100, verbose=False)])\n    oof[va_idx] = model.predict_proba(X_va)[:, 1].astype(np.float32)\n    pred += model.predict_proba(X_test)[:, 1].astype(np.float32) / n_splits\n    print(f\"Fold {fold} AUC: {roc_auc_score(y_va, oof[va_idx]):.4f}\")\n\nprint(f\"\\nOverall CV AUC: {roc_auc_score(y, oof):.4f}\")\nprint(f\"Gini: {2 * roc_auc_score(y, oof) - 1:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-18T20:28:36.802272Z","iopub.execute_input":"2026-03-18T20:28:36.802625Z","iopub.status.idle":"2026-03-18T20:49:44.225943Z","shell.execute_reply.started":"2026-03-18T20:28:36.802596Z","shell.execute_reply":"2026-03-18T20:49:44.224978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = pd.read_csv(SUBMISSION_PATH)\nsub[\"score\"] = pred\nout_path = OUT_DIR / \"submission.csv\"\nsub.to_csv(out_path, index=False)\nprint(f\"Saved: {out_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-18T20:49:44.293715Z","iopub.execute_input":"2026-03-18T20:49:44.294124Z","iopub.status.idle":"2026-03-18T20:49:44.309289Z","shell.execute_reply.started":"2026-03-18T20:49:44.294090Z","shell.execute_reply":"2026-03-18T20:49:44.308485Z"}},"outputs":[],"execution_count":null}]}