{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -*- coding: utf-8 -*-\n\"\"\"Home Credit 2024 - Credit Risk Model Stability Starter Notebook\"\"\"\n\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score\nimport lightgbm as lgb\nimport polars as pl\nimport os\nfrom tqdm import tqdm\n\n# =============================================================================\n# 1. 配置路径\n# =============================================================================\nTRAIN_DATA_DIR = \"/kaggle/input/competitions/home-credit-credit-risk-model-stability/csv_files/train\"\nTEST_DATA_DIR  = \"/kaggle/input/competitions/home-credit-credit-risk-model-stability/csv_files/test\"\n\nprint(\"=\"*50)\nprint(\"1. 开始加载数据...\")\nprint(\"=\"*50)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-29T05:34:41.367330Z","iopub.execute_input":"2026-06-29T05:34:41.367782Z","iopub.status.idle":"2026-06-29T05:34:53.904789Z","shell.execute_reply.started":"2026-06-29T05:34:41.367750Z","shell.execute_reply":"2026-06-29T05:34:53.903798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 2. 加载训练集和测试集的基础表（假设文件名与官方一致，但扩展名为 .csv）\n# =============================================================================\n# 注意：请根据实际文件名调整，可以先用 os.listdir() 查看目录内容\nbase_train = pl.read_csv(os.path.join(TRAIN_DATA_DIR, \"train_base.csv\"))\nbase_test  = pl.read_csv(os.path.join(TEST_DATA_DIR, \"test_base.csv\"))\n\nprint(f\"训练集基础表 shape: {base_train.shape}\")\nprint(f\"测试集基础表 shape: {base_test.shape}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-29T05:35:30.658171Z","iopub.execute_input":"2026-06-29T05:35:30.658543Z","iopub.status.idle":"2026-06-29T05:35:30.836782Z","shell.execute_reply.started":"2026-06-29T05:35:30.658514Z","shell.execute_reply":"2026-06-29T05:35:30.835611Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 3. 加载静态特征表（同样路径下）\n# =============================================================================\nstatic_train = pl.read_csv(os.path.join(TRAIN_DATA_DIR, \"train_static_0_0.csv\"))\nstatic_test  = pl.read_csv(os.path.join(TEST_DATA_DIR, \"test_static_0_0.csv\"))\n\nprint(f\"训练集静态表 shape: {static_train.shape}\")\nprint(f\"测试集静态表 shape: {static_test.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-29T05:36:56.737364Z","iopub.execute_input":"2026-06-29T05:36:56.737767Z","iopub.status.idle":"2026-06-29T05:37:01.057736Z","shell.execute_reply.started":"2026-06-29T05:36:56.737738Z","shell.execute_reply":"2026-06-29T05:37:01.056666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4. 合并数据 (Merge)\n# =============================================================================\n# 将所有表通过 case_id 进行合并\ntrain_df = base_train.join(static_train, on=\"case_id\", how=\"left\")\ntest_df = base_test.join(static_test, on=\"case_id\", how=\"left\")\n\n# 将Polars DataFrame转换为Pandas，以便与sklearn等库兼容\n# 注意：评论中提到，如果转换出错，可以先将case_id转为list[reference:6]\ntrain_df = train_df.to_pandas()\ntest_df = test_df.to_pandas()\n\nprint(f\"合并后训练集 shape: {train_df.shape}\")\nprint(f\"合并后测试集 shape: {test_df.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-29T05:37:34.093094Z","iopub.execute_input":"2026-06-29T05:37:34.094157Z","iopub.status.idle":"2026-06-29T05:37:38.847068Z","shell.execute_reply.started":"2026-06-29T05:37:34.094112Z","shell.execute_reply":"2026-06-29T05:37:38.845887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 5. 数据预处理\n# =============================================================================\nprint(\"=\"*50)\nprint(\"2. 开始数据预处理...\")\nprint(\"=\"*50)\n\n# 将目标变量分离\ntarget = train_df[\"target\"]\ntrain_df = train_df.drop(columns=[\"target\"])\n\n# 保存case_id用于提交\ncase_ids_test = test_df[\"case_id\"]\n\n# 删除不必要的列\ndrop_cols = [\"case_id\"]\ntrain_df = train_df.drop(columns=drop_cols, errors=\"ignore\")\ntest_df = test_df.drop(columns=drop_cols, errors=\"ignore\")\n\n# 处理缺失值：用-999填充（LightGBM可以原生处理NaN，但填充一个极端值有时效果更好）\ntrain_df = train_df.fillna(-999)\ntest_df = test_df.fillna(-999)\n\n# 处理非数值列：转为类别型（category）或进行编码\nfor col in train_df.select_dtypes(include=[\"object\"]).columns:\n    if col in test_df.columns:\n        # 合并后统一编码，防止测试集出现训练集未见的类别\n        combined = pd.concat([train_df[col], test_df[col]], axis=0)\n        combined = combined.astype(\"category\")\n        train_df[col] = combined.iloc[:len(train_df)].values\n        test_df[col] = combined.iloc[len(train_df):].values\n\nprint(f\"预处理后训练集 shape: {train_df.shape}\")\nprint(f\"预处理后测试集 shape: {test_df.shape}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-29T05:38:07.772876Z","iopub.execute_input":"2026-06-29T05:38:07.773200Z","iopub.status.idle":"2026-06-29T05:38:48.401192Z","shell.execute_reply.started":"2026-06-29T05:38:07.773175Z","shell.execute_reply":"2026-06-29T05:38:48.399252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 6. 划分训练集与验证集\n# =============================================================================\nprint(\"=\"*50)\nprint(\"3. 划分训练集与验证集...\")\nprint(\"=\"*50)\n\n# 为了避免数据泄露，按case_id进行划分[reference:7]\ncase_ids = train_df.index.values\n# 使用60%的数据作为训练集，40%作为验证集[reference:8]\ncase_ids_train, case_ids_valid = train_test_split(\n    case_ids, \n    train_size=0.6, \n    random_state=42\n)\n\nX_train = train_df.loc[case_ids_train]\ny_train = target.loc[case_ids_train]\n\nX_valid = train_df.loc[case_ids_valid]\ny_valid = target.loc[case_ids_valid]\n\nprint(f\"训练集样本数: {len(X_train)}\")\nprint(f\"验证集样本数: {len(X_valid)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-29T05:39:18.452673Z","iopub.execute_input":"2026-06-29T05:39:18.453272Z","iopub.status.idle":"2026-06-29T05:39:22.300670Z","shell.execute_reply.started":"2026-06-29T05:39:18.453236Z","shell.execute_reply":"2026-06-29T05:39:22.299623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 7. 模型训练 (LightGBM)\n# =============================================================================\nprint(\"=\"*50)\nprint(\"4. 开始训练LightGBM模型...\")\nprint(\"=\"*50)\n\n# LightGBM 数据集格式\nlgb_train = lgb.Dataset(X_train, y_train)\nlgb_valid = lgb.Dataset(X_valid, y_valid, reference=lgb_train)\n\n# 模型参数 (可调)\nparams = {\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"boosting_type\": \"gbdt\",\n    \"num_leaves\": 31,\n    \"learning_rate\": 0.05,\n    \"feature_fraction\": 0.9,\n    \"bagging_fraction\": 0.8,\n    \"bagging_freq\": 5,\n    \"verbose\": 0,\n    \"n_jobs\": -1,\n    \"random_state\": 42,\n}\n\n# 训练模型\ngbm = lgb.train(\n    params,\n    lgb_train,\n    num_boost_round=1000,\n    valid_sets=[lgb_train, lgb_valid],\n    callbacks=[lgb.early_stopping(50), lgb.log_evaluation(100)]\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-29T05:39:42.181226Z","iopub.execute_input":"2026-06-29T05:39:42.181575Z","iopub.status.idle":"2026-06-29T05:40:59.264618Z","shell.execute_reply.started":"2026-06-29T05:39:42.181547Z","shell.execute_reply":"2026-06-29T05:40:59.263642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 8. 模型评估 (AUC)\n# =============================================================================\nprint(\"=\"*50)\nprint(\"5. 模型评估...\")\nprint(\"=\"*50)\n\n# 在训练集和验证集上进行预测\ny_pred_train = gbm.predict(X_train, num_iteration=gbm.best_iteration)\ny_pred_valid = gbm.predict(X_valid, num_iteration=gbm.best_iteration)\n\n# 计算AUC\nauc_train = roc_auc_score(y_train, y_pred_train)\nauc_valid = roc_auc_score(y_valid, y_pred_valid)\n\nprint(f\"训练集 AUC: {auc_train:.6f}\")\nprint(f\"验证集 AUC: {auc_valid:.6f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-29T05:41:22.030648Z","iopub.execute_input":"2026-06-29T05:41:22.031829Z","iopub.status.idle":"2026-06-29T05:41:32.128934Z","shell.execute_reply.started":"2026-06-29T05:41:22.031782Z","shell.execute_reply":"2026-06-29T05:41:32.127953Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 9. 生成测试集预测并提交\n# =============================================================================\nprint(\"=\"*50)\nprint(\"6. 生成测试集预测...\")\nprint(\"=\"*50)\n\n# 对测试集进行预测\ny_pred_test = gbm.predict(test_df, num_iteration=gbm.best_iteration)\n\n# 生成提交文件\nsubmission = pd.DataFrame({\n    \"case_id\": case_ids_test,\n    \"score\": y_pred_test\n})\n\n# 保存文件\nsubmission.to_csv(\"submission.csv\", index=False)\nprint(\"提交文件已生成: submission.csv\")\nprint(\"=\"*50)\nprint(\"运行完成！\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-29T05:41:39.143966Z","iopub.execute_input":"2026-06-29T05:41:39.144716Z","iopub.status.idle":"2026-06-29T05:41:39.640535Z","shell.execute_reply.started":"2026-06-29T05:41:39.144686Z","shell.execute_reply":"2026-06-29T05:41:39.639522Z"}},"outputs":[],"execution_count":null}]}