{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":8375862,"sourceType":"datasetVersion","datasetId":4980177},{"sourceId":8397527,"sourceType":"datasetVersion","datasetId":4995901},{"sourceId":8397615,"sourceType":"datasetVersion","datasetId":4995963},{"sourceId":176853727,"sourceType":"kernelVersion"},{"sourceId":176966318,"sourceType":"kernelVersion"}],"dockerImageVersionId":30699,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import sys\nfrom pathlib import Path\nimport subprocess\nimport os\nimport gc\nfrom glob import glob\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nfrom datetime import datetime\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nROOT = '/kaggle/input/home-credit-credit-risk-model-stability'\n\nfrom sklearn.model_selection import TimeSeriesSplit, GroupKFold, StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.metrics import roc_auc_score\nimport lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:20:52.184244Z","iopub.execute_input":"2024-05-13T08:20:52.184559Z","iopub.status.idle":"2024-05-13T08:20:58.799542Z","shell.execute_reply.started":"2024-05-13T08:20:52.184532Z","shell.execute_reply":"2024-05-13T08:20:58.798493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pl.read_parquet(\"/kaggle/input/home-credit-dataprocessor/train_df.parquet\")\ndf_train","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:20:58.801657Z","iopub.execute_input":"2024-05-13T08:20:58.802336Z","iopub.status.idle":"2024-05-13T08:21:20.970773Z","shell.execute_reply.started":"2024-05-13T08:20:58.802300Z","shell.execute_reply":"2024-05-13T08:21:20.969796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install /kaggle/input/input-whl/kaggle_home_credit_risk_model_stability-0.3-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:21:20.972156Z","iopub.execute_input":"2024-05-13T08:21:20.972539Z","iopub.status.idle":"2024-05-13T08:21:53.925229Z","shell.execute_reply.started":"2024-05-13T08:21:20.972504Z","shell.execute_reply":"2024-05-13T08:21:53.924104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import polars as pl\nimport gc\nimport pickle\n\nimport kaggle_home_credit_risk_model_stability.libs as hcr\nfrom kaggle_home_credit_risk_model_stability.libs.env import Env\nfrom kaggle_home_credit_risk_model_stability.libs.input.dataset import Dataset\nfrom kaggle_home_credit_risk_model_stability.libs.input.data_loader import DataLoader\nfrom kaggle_home_credit_risk_model_stability.libs.preprocessor.preprocessor import Preprocessor\nfrom kaggle_home_credit_risk_model_stability.libs.preprocessor.steps import *\nfrom kaggle_home_credit_risk_model_stability.libs.preprocessor.columns_info import ColumnsInfo","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:22:04.876904Z","iopub.execute_input":"2024-05-13T08:22:04.877740Z","iopub.status.idle":"2024-05-13T08:22:04.883432Z","shell.execute_reply.started":"2024-05-13T08:22:04.877701Z","shell.execute_reply":"2024-05-13T08:22:04.882607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from kaggle_home_credit_risk_model_stability.libs.lightgbm.kfold_model import KFoldLightGbmModel\nfrom kaggle_home_credit_risk_model_stability.libs.catboost.kfold_model import KFoldCatboostModel\nfrom kaggle_home_credit_risk_model_stability.libs.feature_selection.correlation_groups import CorrelationGroupsFeatureSelector","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:22:05.343306Z","iopub.execute_input":"2024-05-13T08:22:05.343729Z","iopub.status.idle":"2024-05-13T08:22:05.349143Z","shell.execute_reply.started":"2024-05-13T08:22:05.343698Z","shell.execute_reply":"2024-05-13T08:22:05.348229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"env = Env(\n    \"/kaggle/input/\",\n    \"/kaggle/working/\"\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:22:06.713383Z","iopub.execute_input":"2024-05-13T08:22:06.713738Z","iopub.status.idle":"2024-05-13T08:22:06.717622Z","shell.execute_reply.started":"2024-05-13T08:22:06.713709Z","shell.execute_reply":"2024-05-13T08:22:06.716726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# feature_selector = CorrelationGroupsFeatureSelector(numerical_threshold=0.85, categorical_threashold=0.9)\n# selected_features = feature_selector.select(df_train, features)","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:22:07.232155Z","iopub.execute_input":"2024-05-13T08:22:07.232809Z","iopub.status.idle":"2024-05-13T08:22:07.237678Z","shell.execute_reply.started":"2024-05-13T08:22:07.232778Z","shell.execute_reply":"2024-05-13T08:22:07.236726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature selection","metadata":{}},{"cell_type":"code","source":"with open('/kaggle/input/home-credit-train-lgb-corr/home-credit-lgb-trained/model_lgb.pkl', 'rb') as rf:\n    lgb_voting_model = pickle.load(rf)\n\n# Numerical Feature, Cateogrical feature\nnumerical_features = [col for col in df_train.columns if df_train[col].dtype != pl.Enum]\ncategorical_features = [col for col in df_train.columns if df_train[col].dtype == pl.Enum]\n\n# Importance dataframe from lgbm\ndef map_feature_to_type(f):\n    if f in numerical_features:\n        return 'num'\n    elif f  in categorical_features:\n        return 'cat'\n    else:\n        return 'service'\n    \nim_df = pd.DataFrame()\nim_df['feature'] = [f[1] for f in lgb_voting_model.get_feature_importance()]\nim_df['importance'] = [f[0] for f in lgb_voting_model.get_feature_importance()]\nim_df['importance'] /= 5 # 5 fold\nim_df['type'] = im_df['feature'].apply(map_feature_to_type)\n\n# Select features with importance >= 50\nthresh = 50\nfil_im_df = im_df[im_df['importance'] >= thresh]\n\n# Statistic\nprint('==== Features after data preprocessing step =====')\nprint('Total: ', df_train.shape)\nprint('Numerical: ', len(numerical_features))\nprint('Categorical: ', len(categorical_features))\n\nprint('==== Features after selection with correlation ====')\nprint('Total: ', im_df.shape[0])\nprint('Numerical: ', (im_df['type'] == 'num').sum())\nprint('Categorical: ', (im_df['type'] == 'cat').sum())\n\nprint('==== Features after selection with importances in lgbm ====')\nprint('Total: ', fil_im_df.shape[0])\nprint('Numerical: ', (fil_im_df['type'] == 'num').sum())\nprint('Categorical: ', (fil_im_df['type'] == 'cat').sum())","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:22:09.269099Z","iopub.execute_input":"2024-05-13T08:22:09.269920Z","iopub.status.idle":"2024-05-13T08:22:10.034920Z","shell.execute_reply.started":"2024-05-13T08:22:09.269886Z","shell.execute_reply":"2024-05-13T08:22:10.033897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"im_df","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:22:10.036315Z","iopub.execute_input":"2024-05-13T08:22:10.036590Z","iopub.status.idle":"2024-05-13T08:22:10.052245Z","shell.execute_reply.started":"2024-05-13T08:22:10.036567Z","shell.execute_reply":"2024-05-13T08:22:10.051327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_features = fil_im_df['feature'].to_list()\nnumerical_features = fil_im_df[fil_im_df['type'] == 'num']['feature']\ncategorical_features = fil_im_df[fil_im_df['type'] == 'cat']['feature']","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:22:11.594641Z","iopub.execute_input":"2024-05-13T08:22:11.595030Z","iopub.status.idle":"2024-05-13T08:22:11.601936Z","shell.execute_reply.started":"2024-05-13T08:22:11.595005Z","shell.execute_reply":"2024-05-13T08:22:11.600949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(selected_features)","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:22:51.446991Z","iopub.execute_input":"2024-05-13T08:22:51.447650Z","iopub.status.idle":"2024-05-13T08:22:51.455404Z","shell.execute_reply.started":"2024-05-13T08:22:51.447611Z","shell.execute_reply":"2024-05-13T08:22:51.454207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# with open('/kaggle/input/selected-features/selected_features.pkl', 'rb') as f:\n#         selected_features = pickle.load(f)\n#         print(\"Danh sách các features được đọc từ file pickle:\")\n#         print(selected_features)","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:22:56.048329Z","iopub.execute_input":"2024-05-13T08:22:56.048964Z","iopub.status.idle":"2024-05-13T08:22:56.053009Z","shell.execute_reply.started":"2024-05-13T08:22:56.048933Z","shell.execute_reply":"2024-05-13T08:22:56.052031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.select([\"target\", 'case_id', \"WEEK_NUM\"] + selected_features)\ngc.collect()\ndf_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:23:15.982897Z","iopub.execute_input":"2024-05-13T08:23:15.983606Z","iopub.status.idle":"2024-05-13T08:23:16.103877Z","shell.execute_reply.started":"2024-05-13T08:23:15.983572Z","shell.execute_reply":"2024-05-13T08:23:16.102930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:23:17.801938Z","iopub.execute_input":"2024-05-13T08:23:17.802760Z","iopub.status.idle":"2024-05-13T08:23:17.826915Z","shell.execute_reply.started":"2024-05-13T08:23:17.802726Z","shell.execute_reply":"2024-05-13T08:23:17.826027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"enum_cols = []\nfor col in df_train.columns:\n    if df_train[col].dtype == pl.datatypes.Enum:\n        enum_cols.append(col)\n\n\n# Tạo cột mới và loại bỏ cột cũ\nfor col in enum_cols:\n    df_train = df_train.with_columns(\n        pl.col(col).cast(pl.Categorical).to_physical().alias(col + \"_ordinal\")\n    )\n\ndf_train = df_train.drop(enum_cols)","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:23:20.391309Z","iopub.execute_input":"2024-05-13T08:23:20.391659Z","iopub.status.idle":"2024-05-13T08:23:20.418458Z","shell.execute_reply.started":"2024-05-13T08:23:20.391633Z","shell.execute_reply":"2024-05-13T08:23:20.417714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:23:21.812036Z","iopub.execute_input":"2024-05-13T08:23:21.812742Z","iopub.status.idle":"2024-05-13T08:23:21.818445Z","shell.execute_reply.started":"2024-05-13T08:23:21.812711Z","shell.execute_reply":"2024-05-13T08:23:21.817512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train","metadata":{}},{"cell_type":"code","source":"from kaggle_home_credit_risk_model_stability.libs.xgboost.kfold_model import KFoldXgboostModel","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:23:23.437171Z","iopub.execute_input":"2024-05-13T08:23:23.437829Z","iopub.status.idle":"2024-05-13T08:23:23.443595Z","shell.execute_reply.started":"2024-05-13T08:23:23.437798Z","shell.execute_reply":"2024-05-13T08:23:23.442631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_features = [col for col in df_train.columns if col not in [\"target\", \"WEEK_NUM\", \"case_id\"]]","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:23:24.378578Z","iopub.execute_input":"2024-05-13T08:23:24.378943Z","iopub.status.idle":"2024-05-13T08:23:24.383729Z","shell.execute_reply.started":"2024-05-13T08:23:24.378915Z","shell.execute_reply":"2024-05-13T08:23:24.382788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(selected_features)","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:44:23.328457Z","iopub.execute_input":"2024-05-13T08:44:23.328879Z","iopub.status.idle":"2024-05-13T08:44:23.337413Z","shell.execute_reply.started":"2024-05-13T08:44:23.328849Z","shell.execute_reply":"2024-05-13T08:44:23.336445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = 'gpu'","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:44:47.949809Z","iopub.execute_input":"2024-05-13T08:44:47.950178Z","iopub.status.idle":"2024-05-13T08:44:47.954459Z","shell.execute_reply.started":"2024-05-13T08:44:47.950146Z","shell.execute_reply":"2024-05-13T08:44:47.953551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {\n    \"booster\": \"gbtree\",\n    \"objective\": \"binary:logistic\",\n    \"eval_metric\": \"auc\",\n    \"max_depth\": 10,\n    \"learning_rate\": 0.05,\n    \"n_estimators\": 600,\n    \"colsample_bytree\": 0.8,\n    \"colsample_bynode\": 0.8,\n    \"alpha\": 0.1,  \n    \"lambda\": 10,  \n    \"tree_method\": 'gpu_hist' if device == 'gpu' else 'auto',\n    \"random_state\": 42,\n    \"verbosity\": 0\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:44:51.113531Z","iopub.execute_input":"2024-05-13T08:44:51.113892Z","iopub.status.idle":"2024-05-13T08:44:51.119235Z","shell.execute_reply.started":"2024-05-13T08:44:51.113862Z","shell.execute_reply":"2024-05-13T08:44:51.118295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_model = KFoldXgboostModel(env, selected_features, model_params = params)\nxgb_model.train(df_train, n_splits=5)","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:23:26.750900Z","iopub.execute_input":"2024-05-13T08:23:26.751582Z","iopub.status.idle":"2024-05-13T08:26:30.861762Z","shell.execute_reply.started":"2024-05-13T08:23:26.751553Z","shell.execute_reply":"2024-05-13T08:26:30.860839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle as pkl\n\nwith open(f'/kaggle/working/model_xgb.pkl', 'wb') as fout:\n    pkl.dump(xgb_model, fout)\n    \nprint('saved model done.')","metadata":{"execution":{"iopub.status.busy":"2024-05-13T08:35:35.233616Z","iopub.execute_input":"2024-05-13T08:35:35.234357Z","iopub.status.idle":"2024-05-13T08:35:35.267290Z","shell.execute_reply.started":"2024-05-13T08:35:35.234325Z","shell.execute_reply":"2024-05-13T08:35:35.266373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}