{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":8346684,"sourceType":"datasetVersion","datasetId":4958380},{"sourceId":8346765,"sourceType":"datasetVersion","datasetId":4958444},{"sourceId":8346838,"sourceType":"datasetVersion","datasetId":4958496},{"sourceId":8347465,"sourceType":"datasetVersion","datasetId":4958988}],"dockerImageVersionId":30648,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\nRecently, I wrote a discussion about [gini_stability_custom_metric](https://www.kaggle.com/competitions/home-credit-credit-risk-model-stability/discussion/500868), in this notebook, I will use this and BoostRFE to do feature selection.\n\nin this notebook, you will find:\n1. gini_stability_custom_metric_lgb_sklearnapi. This function is used in lightgbm with sklearn api version\n2. way to use BoostRFE from shaphypetune to do feature selection, I will provide a detailed explanation below.\n3. with above change, the score improved a little!😂(you can see version3)\n\nThis notebook bases on the great notebook, thanks him firstly😃https://www.kaggle.com/code/rsmits/feature-selection-with-boruta\n\nIn the end, if you find this useful, pleaaaaaase give me an upvote!🥳🥳🥳🥳🥳","metadata":{}},{"cell_type":"code","source":"!pip install file:///kaggle/input/hyperopt025/hyperopt-0.2.5-py2.py3-none-any.whl\n!pip install file:///kaggle/input/shap-hypetune027/shap_hypetune-0.2.7-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:17:55.294890Z","iopub.execute_input":"2024-05-08T04:17:55.295461Z","iopub.status.idle":"2024-05-08T04:19:01.406926Z","shell.execute_reply.started":"2024-05-08T04:17:55.295434Z","shell.execute_reply":"2024-05-08T04:19:01.405947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from shaphypetune import BoostRFE, BoostBoruta, BoostSearch","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:19:01.408653Z","iopub.execute_input":"2024-05-08T04:19:01.408962Z","iopub.status.idle":"2024-05-08T04:19:07.669988Z","shell.execute_reply.started":"2024-05-08T04:19:01.408933Z","shell.execute_reply":"2024-05-08T04:19:07.669197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\nimport time\nfrom contextlib import contextmanager\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, ClassifierMixin\nfrom sklearn.metrics import roc_auc_score\n\nimport lightgbm as lgb\n\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-08T04:19:07.671423Z","iopub.execute_input":"2024-05-08T04:19:07.672258Z","iopub.status.idle":"2024-05-08T04:19:11.299650Z","shell.execute_reply.started":"2024-05-08T04:19:07.672224Z","shell.execute_reply":"2024-05-08T04:19:11.298630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## utils","metadata":{}},{"cell_type":"markdown","source":"### gini_stability_custom_metric_lgb_sklearnapi","metadata":{}},{"cell_type":"code","source":"def gini_stability_custom_metric_lgb_sklearnapi(y_pred: np.array, y_true: np.array, week: np.array):\n    \n    '''\n    \n    :param y_pred:\n    :param y_true:\n    :param week: \n    :return eval_name: str\n    :return eval_result: float\n    :return is_higher_better: bool\n    '''\n    \n    w_fallingrate = 88.0\n    w_resstd = -0.5\n    \n    base = pd.DataFrame()\n    base['WEEK_NUM'] = week\n    base['target'] = y_true\n    base['score'] = y_pred\n\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", \"score\"]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", \"score\"]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[\"score\"])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    \n    final_score = avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std\n    \n    return 'gini_stability', final_score, True\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### tools","metadata":{}},{"cell_type":"code","source":"@contextmanager\ndef simple_timer(message):\n    print('Now processing→', message)\n    start_time = time.time()\n    yield\n    elapsed_time = time.time() - start_time\n    print(\"Processed→ {}: {:.3f} [s]\".format(message, elapsed_time))","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:19:11.301774Z","iopub.execute_input":"2024-05-08T04:19:11.302071Z","iopub.status.idle":"2024-05-08T04:19:11.307473Z","shell.execute_reply.started":"2024-05-08T04:19:11.302047Z","shell.execute_reply":"2024-05-08T04:19:11.306293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Pre-Fitted Voting Model","metadata":{}},{"cell_type":"code","source":"class VotingModel(BaseEstimator, ClassifierMixin):\n    def __init__(self, estimators):\n        super().__init__()\n        self.estimators = estimators\n        \n    def fit(self, X, y=None):\n        return self\n    \n    def predict(self, X):\n        y_preds = [estimator.predict(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)\n    \n    def predict_proba(self, X):\n        y_preds = [estimator.predict_proba(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:19:11.308390Z","iopub.execute_input":"2024-05-08T04:19:11.308619Z","iopub.status.idle":"2024-05-08T04:19:11.325795Z","shell.execute_reply.started":"2024-05-08T04:19:11.308600Z","shell.execute_reply":"2024-05-08T04:19:11.324922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Pipeline","metadata":{}},{"cell_type":"code","source":"class Pipeline:\n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int32))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                df = df.with_columns(pl.col(col).cast(pl.Float32))\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.95:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n\n        return df","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:19:11.326801Z","iopub.execute_input":"2024-05-08T04:19:11.327122Z","iopub.status.idle":"2024-05-08T04:19:11.339747Z","shell.execute_reply.started":"2024-05-08T04:19:11.327098Z","shell.execute_reply":"2024-05-08T04:19:11.338914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Automatic Aggregation","metadata":{}},{"cell_type":"code","source":"class Aggregator:\n    @staticmethod\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\",)]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n    \n    @staticmethod\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n\n        return exprs","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:19:11.340775Z","iopub.execute_input":"2024-05-08T04:19:11.341051Z","iopub.status.idle":"2024-05-08T04:19:11.354281Z","shell.execute_reply.started":"2024-05-08T04:19:11.341023Z","shell.execute_reply":"2024-05-08T04:19:11.353431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### File I/O","metadata":{}},{"cell_type":"code","source":"def read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    \n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    \n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob(str(regex_path)):\n        df = pl.read_parquet(path)\n        df = df.pipe(Pipeline.set_table_dtypes)\n        \n        if depth in [1, 2]:\n            df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        \n        chunks.append(df)\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:19:11.355239Z","iopub.execute_input":"2024-05-08T04:19:11.355471Z","iopub.status.idle":"2024-05-08T04:19:11.367517Z","shell.execute_reply.started":"2024-05-08T04:19:11.355451Z","shell.execute_reply":"2024-05-08T04:19:11.366686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature Engineering","metadata":{}},{"cell_type":"code","source":"def feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n        \n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n        \n    df_base = df_base.pipe(Pipeline.handle_dates)\n    \n    return df_base","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:19:11.368764Z","iopub.execute_input":"2024-05-08T04:19:11.369348Z","iopub.status.idle":"2024-05-08T04:19:11.377505Z","shell.execute_reply.started":"2024-05-08T04:19:11.369317Z","shell.execute_reply":"2024-05-08T04:19:11.376627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n    return df_data, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:19:11.381611Z","iopub.execute_input":"2024-05-08T04:19:11.381857Z","iopub.status.idle":"2024-05-08T04:19:11.390276Z","shell.execute_reply.started":"2024-05-08T04:19:11.381837Z","shell.execute_reply":"2024-05-08T04:19:11.389513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Configuration","metadata":{}},{"cell_type":"code","source":"ROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:19:11.391104Z","iopub.execute_input":"2024-05-08T04:19:11.391329Z","iopub.status.idle":"2024-05-08T04:19:11.399740Z","shell.execute_reply.started":"2024-05-08T04:19:11.391310Z","shell.execute_reply":"2024-05-08T04:19:11.398915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train Files Read & Feature Engineering","metadata":{}},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_2_*.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:19:11.400920Z","iopub.execute_input":"2024-05-08T04:19:11.401264Z","iopub.status.idle":"2024-05-08T04:21:28.644037Z","shell.execute_reply.started":"2024-05-08T04:19:11.401234Z","shell.execute_reply":"2024-05-08T04:21:28.643247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = feature_eng(**data_store)\n\nprint(\"train data shape:\\t\", df_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:21:28.645202Z","iopub.execute_input":"2024-05-08T04:21:28.645986Z","iopub.status.idle":"2024-05-08T04:21:41.039321Z","shell.execute_reply.started":"2024-05-08T04:21:28.645950Z","shell.execute_reply":"2024-05-08T04:21:41.038387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test Files Read & Feature Engineering","metadata":{}},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_files(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n        read_files(TEST_DIR / \"test_credit_bureau_a_2_*.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:21:41.040442Z","iopub.execute_input":"2024-05-08T04:21:41.040737Z","iopub.status.idle":"2024-05-08T04:21:41.618756Z","shell.execute_reply.started":"2024-05-08T04:21:41.040712Z","shell.execute_reply":"2024-05-08T04:21:41.617799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = feature_eng(**data_store)\n\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:21:41.619991Z","iopub.execute_input":"2024-05-08T04:21:41.620306Z","iopub.status.idle":"2024-05-08T04:21:41.663478Z","shell.execute_reply.started":"2024-05-08T04:21:41.620281Z","shell.execute_reply":"2024-05-08T04:21:41.662631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Elimination1","metadata":{}},{"cell_type":"code","source":"df_train = df_train.pipe(Pipeline.filter_cols)\ndf_test = df_test.select([col for col in df_train.columns if col != \"target\"])\n\nprint(\"train data shape:\\t\", df_train.shape)\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:21:41.664627Z","iopub.execute_input":"2024-05-08T04:21:41.664979Z","iopub.status.idle":"2024-05-08T04:21:45.156620Z","shell.execute_reply.started":"2024-05-08T04:21:41.664945Z","shell.execute_reply":"2024-05-08T04:21:45.155656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train, cat_cols = to_pandas(df_train)\ndf_test, cat_cols = to_pandas(df_test, cat_cols)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:21:45.157997Z","iopub.execute_input":"2024-05-08T04:21:45.158421Z","iopub.status.idle":"2024-05-08T04:22:04.264582Z","shell.execute_reply.started":"2024-05-08T04:21:45.158387Z","shell.execute_reply":"2024-05-08T04:22:04.263679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del data_store\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:22:04.265887Z","iopub.execute_input":"2024-05-08T04:22:04.266581Z","iopub.status.idle":"2024-05-08T04:22:04.457271Z","shell.execute_reply.started":"2024-05-08T04:22:04.266543Z","shell.execute_reply":"2024-05-08T04:22:04.456070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training with BoostRFE\n\nBoostRFE:\n- In source code, the while loop reduces the number of features in each iteration by eliminating the poorer ones. Then, the score is computed in each iteration and compared with the best score.\n- In each iteration, if the new score is better than the best score, the best score and best feature selection are updated. This means that the best feature selection corresponding to the best score is the best one achieved during the iteration process.\n- Finally, when all iterations are completed and the specified minimum number of features is reached, the final estimator is trained using the feature selection corresponding to the best score. This implies that the features selected in the end are the versions corresponding to the best score obtained during the iteration process.\n- Therefore, this code selects the best feature subset based on the best score, aiming to achieve the best performance for the model.\"","metadata":{}},{"cell_type":"code","source":"X = df_train.drop(columns=[\"target\", \"case_id\", \"WEEK_NUM\"])\ny = df_train[\"target\"]\nweeks = df_train[\"WEEK_NUM\"]\n\nall_list = X.columns.tolist()\n\ncat_cols = [all_list.index(x) for x in cat_cols]\n\ndel df_train","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:23:09.970607Z","iopub.execute_input":"2024-05-08T04:23:09.971305Z","iopub.status.idle":"2024-05-08T04:23:09.975444Z","shell.execute_reply.started":"2024-05-08T04:23:09.971275Z","shell.execute_reply":"2024-05-08T04:23:09.974367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ncv = StratifiedGroupKFold(n_splits=5, shuffle=False)\n\nlgb_params = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"custom\",\n    \"max_depth\": 8,\n    \"learning_rate\": 0.05,\n    \"n_estimators\": 3000,\n    \"colsample_bytree\": 0.8, \n    \"colsample_bynode\": 0.8,\n    \"verbose\": -1,\n    \"random_state\": 42,\n    \"device\": \"gpu\",\n}\n\nfitted_models = []\n\nfold_token = 0\nfor idx_train, idx_valid in cv.split(X, y, groups=weeks):\n    print(f'*********************** {fold_token} ***********************')\n    X_train, y_train = X.iloc[idx_train], y.iloc[idx_train]\n    X_valid, y_valid, week_valid = X.iloc[idx_valid], y.iloc[idx_valid], weeks[idx_valid]\n    \n    model = BoostRFE(lgb.LGBMClassifier(**lgb_params)\n                 , greater_is_better=True\n                 , min_features_to_select=1  # The minimum number of features to be selected.\n                 , importance_type='feature_importances'  # also can use 'shap_importances'\n                 , train_importance=True\n                 , verbose=1\n                 \n                 # des of step:\n                 # If greater than or equal to 1, then `step` corresponds to the (integer) number of features to remove at each iteration.\n                 # If within (0.0, 1.0), then `step` corresponds to the percentage (rounded down) of features to remove at each iteration.\n                 , step=0.2\n                )\n\n    model.fit(X=X_train\n              , y=y_train\n              , eval_set=[(X_valid, y_valid)]\n              , categorical_feature=cat_cols\n              , callbacks=[\n                    lgb.callback.early_stopping(stopping_rounds=10),\n                    lgb.callback.log_evaluation(period=10),\n                ]\n              , eval_metric=lambda y_true, y_pred: gini_stability_custom_metric_lgb_sklearnapi(y_pred, y_true, week_valid)\n             )\n    \n    \n    '''\n    model  # model is the best model selected by RFE, you can use it to predict without retraining it using selected features\n    \n    model.ranking_  # you can call this attribute to see the feature ranking evaluated by the RFE method. rank==1 is the most useful feature \n    \n    model.score_history_  # you can use this attribute to see iteration result(score of metric) of every model\n    '''\n\n    fitted_models.append(model)\n    \n    fold_token += 1\n\nmodel = VotingModel(fitted_models)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T06:49:00.073045Z","iopub.execute_input":"2024-05-08T06:49:00.073921Z","iopub.status.idle":"2024-05-08T07:07:33.272161Z","shell.execute_reply.started":"2024-05-08T06:49:00.073885Z","shell.execute_reply":"2024-05-08T07:07:33.271132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prediction","metadata":{}},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T06:29:09.544146Z","iopub.execute_input":"2024-05-08T06:29:09.545020Z","iopub.status.idle":"2024-05-08T06:29:09.876466Z","shell.execute_reply.started":"2024-05-08T06:29:09.544974Z","shell.execute_reply":"2024-05-08T06:29:09.875491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = df_test.drop(columns=[\"WEEK_NUM\"])\nX_test = X_test.set_index(\"case_id\")\n\ny_pred = pd.Series(model.predict_proba(X_test)[:, 1], index=X_test.index)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T07:19:24.209418Z","iopub.execute_input":"2024-05-08T07:19:24.209835Z","iopub.status.idle":"2024-05-08T07:19:24.443644Z","shell.execute_reply.started":"2024-05-08T07:19:24.209802Z","shell.execute_reply":"2024-05-08T07:19:24.442614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"df_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\n\ndf_subm[\"score\"] = y_pred","metadata":{"execution":{"iopub.status.busy":"2024-05-08T07:19:27.165131Z","iopub.execute_input":"2024-05-08T07:19:27.165990Z","iopub.status.idle":"2024-05-08T07:19:27.185260Z","shell.execute_reply.started":"2024-05-08T07:19:27.165946Z","shell.execute_reply":"2024-05-08T07:19:27.184212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Check null: \", df_subm[\"score\"].isnull().any())\n\ndf_subm.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T07:19:30.136253Z","iopub.execute_input":"2024-05-08T07:19:30.136715Z","iopub.status.idle":"2024-05-08T07:19:30.154149Z","shell.execute_reply.started":"2024-05-08T07:19:30.136682Z","shell.execute_reply":"2024-05-08T07:19:30.152965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-05-08T07:19:34.567689Z","iopub.execute_input":"2024-05-08T07:19:34.568548Z","iopub.status.idle":"2024-05-08T07:19:34.575521Z","shell.execute_reply.started":"2024-05-08T07:19:34.568512Z","shell.execute_reply":"2024-05-08T07:19:34.574695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"If you would like to delve deeper into this package, you can check out this link: https://github.com/cerlymarco/shap-hypetune/blob/3f73eec1e05c2027c8f9c0780d0c59a8411480d6/notebooks/LGBM_usage.ipynb\n\nThanks again for reading and happy kaggling!😁","metadata":{}}]}