{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30699,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Dependencies","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import sys\nimport os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, ClassifierMixin\nfrom sklearn.preprocessing import LabelEncoder\n\nimport lightgbm as lgb\nfrom catboost import CatBoostClassifier, Pool\nimport joblib\n\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:33:49.960202Z","iopub.execute_input":"2024-05-13T02:33:49.961150Z","iopub.status.idle":"2024-05-13T02:33:55.887936Z","shell.execute_reply.started":"2024-05-13T02:33:49.961115Z","shell.execute_reply":"2024-05-13T02:33:55.887118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration","metadata":{}},{"cell_type":"code","source":"ROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:33:55.889368Z","iopub.execute_input":"2024-05-13T02:33:55.889681Z","iopub.status.idle":"2024-05-13T02:33:55.894365Z","shell.execute_reply.started":"2024-05-13T02:33:55.889657Z","shell.execute_reply":"2024-05-13T02:33:55.893501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Collection and Preprecessing\n\n## Pipeline","metadata":{}},{"cell_type":"code","source":"class Pipeline:\n    # DataFrame의 column type 지정 method\n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    # date type column 처리 및 시차/일차 계산 method\n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\")) # 일차 계산\n                df = df.with_columns(pl.col(col).dt.total_days())\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    # 결측 비율 및 문자형 변수 cardinality을 기준으로 변수 제거\n    @staticmethod\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.7: # 결측 비율 70% 초과 시 제거\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 200): # 문자형 변수 cardinality 기준 변수 제거\n                    df = df.drop(col)\n\n        return df","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:33:55.895637Z","iopub.execute_input":"2024-05-13T02:33:55.895943Z","iopub.status.idle":"2024-05-13T02:33:55.909388Z","shell.execute_reply.started":"2024-05-13T02:33:55.895922Z","shell.execute_reply":"2024-05-13T02:33:55.908467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Aggregation ","metadata":{}},{"cell_type":"code","source":"class Aggregator:\n    @staticmethod\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\",)]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n    \n    @staticmethod\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n\n        return exprs","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:33:55.911965Z","iopub.execute_input":"2024-05-13T02:33:55.912547Z","iopub.status.idle":"2024-05-13T02:33:55.923502Z","shell.execute_reply.started":"2024-05-13T02:33:55.912522Z","shell.execute_reply":"2024-05-13T02:33:55.922693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    \n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    \n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob(str(regex_path)):\n        df = pl.read_parquet(path)\n        df = df.pipe(Pipeline.set_table_dtypes)\n        \n        if depth in [1, 2]:\n            df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        \n        chunks.append(df)\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:33:55.924532Z","iopub.execute_input":"2024-05-13T02:33:55.924856Z","iopub.status.idle":"2024-05-13T02:33:55.937047Z","shell.execute_reply.started":"2024-05-13T02:33:55.924835Z","shell.execute_reply":"2024-05-13T02:33:55.936141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{}},{"cell_type":"code","source":"def feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n        \n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n        \n    df_base = df_base.pipe(Pipeline.handle_dates)\n    \n    return df_base\n\ndef to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n    return df_data, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:33:55.938086Z","iopub.execute_input":"2024-05-13T02:33:55.938385Z","iopub.status.idle":"2024-05-13T02:33:55.947695Z","shell.execute_reply.started":"2024-05-13T02:33:55.938353Z","shell.execute_reply":"2024-05-13T02:33:55.946880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" \n    Iterate through all the columns of a dataframe and modify the data type\n    to reduce memory usage.\n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2  # Memory usage before optimization\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            continue\n    end_mem = df.memory_usage().sum() / 1024**2  # Memory usage after optimization\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:33:55.948760Z","iopub.execute_input":"2024-05-13T02:33:55.950000Z","iopub.status.idle":"2024-05-13T02:33:55.963495Z","shell.execute_reply.started":"2024-05-13T02:33:55.949970Z","shell.execute_reply":"2024-05-13T02:33:55.962615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class VotingModel(BaseEstimator, ClassifierMixin):\n    def __init__(self, estimators):\n        super().__init__()\n        self.estimators = estimators\n        \n    def fit(self, X, y=None):\n        return self\n    \n    def predict(self, X):\n        y_preds = [estimator.predict(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)\n    \n    def predict_proba(self, X):\n        # y_preds = [estimator.predict_proba(X) for estimator in self.estimators]\n        # return np.mean(y_preds, axis=0)\n        X[cat_cols] = X[cat_cols].astype(str) # for catboost\n        y_preds = [estimator.predict_proba(X) for estimator in self.estimators[:5]]\n        \n        X[cat_cols] = X[cat_cols].astype('category') # for lgbm\n        y_preds += [estimator.predict_proba(X) for estimator in self.estimators[5:]]\n        \n        return np.mean(y_preds, axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:33:55.964622Z","iopub.execute_input":"2024-05-13T02:33:55.964892Z","iopub.status.idle":"2024-05-13T02:33:55.977133Z","shell.execute_reply.started":"2024-05-13T02:33:55.964869Z","shell.execute_reply":"2024-05-13T02:33:55.976400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def gini_stability(base, score_col=\"score\", w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", score_col]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", score_col]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[score_col])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:33:55.978287Z","iopub.execute_input":"2024-05-13T02:33:55.978632Z","iopub.status.idle":"2024-05-13T02:33:55.990209Z","shell.execute_reply.started":"2024-05-13T02:33:55.978568Z","shell.execute_reply":"2024-05-13T02:33:55.989388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Data Load","metadata":{}},{"cell_type":"code","source":"%%time\ndata_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TRAIN_DIR / \"train_applprev_2.parquet\", 2),\n        read_file(TRAIN_DIR / \"train_person_2.parquet\", 2)\n    ]\n}\n\ntrain_df = feature_eng(**data_store)","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:33:55.994473Z","iopub.execute_input":"2024-05-13T02:33:55.994795Z","iopub.status.idle":"2024-05-13T02:36:08.699257Z","shell.execute_reply.started":"2024-05-13T02:33:55.994772Z","shell.execute_reply":"2024-05-13T02:36:08.698235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test Data Load","metadata":{}},{"cell_type":"code","source":"%%time\ndata_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_files(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n        read_files(TEST_DIR / \"test_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TEST_DIR / \"test_applprev_2.parquet\", 2),\n        read_file(TEST_DIR / \"test_person_2.parquet\", 2)\n    ]\n}\ntest_df = feature_eng(**data_store)\n\nprint(\"test data shape:\\t\", test_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:36:08.700499Z","iopub.execute_input":"2024-05-13T02:36:08.700869Z","iopub.status.idle":"2024-05-13T02:36:09.361563Z","shell.execute_reply.started":"2024-05-13T02:36:08.700834Z","shell.execute_reply":"2024-05-13T02:36:09.360633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Elimination","metadata":{}},{"cell_type":"code","source":"%%time\ntrain_df = train_df.pipe(Pipeline.filter_cols)\ntest_df = test_df.select([col for col in train_df.columns if col != \"target\"])\n\nprint(\"train data shape:\\t\", train_df.shape)\nprint(\"test data shape:\\t\", test_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:36:09.362798Z","iopub.execute_input":"2024-05-13T02:36:09.363149Z","iopub.status.idle":"2024-05-13T02:36:11.988146Z","shell.execute_reply.started":"2024-05-13T02:36:09.363114Z","shell.execute_reply":"2024-05-13T02:36:11.987232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Pandas Conversion","metadata":{}},{"cell_type":"code","source":"%%time\ntrain_df, cat_cols = to_pandas(train_df)\ntest_df, cat_cols = to_pandas(test_df, cat_cols)","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:36:11.989284Z","iopub.execute_input":"2024-05-13T02:36:11.989544Z","iopub.status.idle":"2024-05-13T02:36:33.055035Z","shell.execute_reply.started":"2024-05-13T02:36:11.989523Z","shell.execute_reply":"2024-05-13T02:36:33.054097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del data_store\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:36:33.056270Z","iopub.execute_input":"2024-05-13T02:36:33.056573Z","iopub.status.idle":"2024-05-13T02:36:33.167896Z","shell.execute_reply.started":"2024-05-13T02:36:33.056545Z","shell.execute_reply":"2024-05-13T02:36:33.166867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = reduce_mem_usage(train_df)\ntest_df = reduce_mem_usage(test_df)\n\nprint(\"train data shape:\\t\", train_df.shape)\nprint(\"test data shape:\\t\", test_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:36:33.169083Z","iopub.execute_input":"2024-05-13T02:36:33.169440Z","iopub.status.idle":"2024-05-13T02:36:39.291850Z","shell.execute_reply.started":"2024-05-13T02:36:33.169408Z","shell.execute_reply":"2024-05-13T02:36:39.290820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 상관관계 기반 그룹화\ndef group_columns_by_corr(matrix, threshold=0.8):\n    correlation_matrix = matrix.corr()\n    groups = []\n    remaining_cols = list(matrix.columns)\n    while remaining_cols:\n        col = remaining_cols.pop(0)\n        group = [col]\n        correlated_cols = [col]\n        \n        for c in remaining_cols:\n            if correlation_matrix.loc[col, c] >= threshold:\n                group.append(c)\n                correlated_cols.append(c)\n        groups.append(group)\n        remaining_cols = [c for c in remaining_cols if c not in correlated_cols]\n    return groups        \n\ndef reduce_groups(grps):\n    use = []\n    for g in grps:\n        mx = 0\n        vx = g[0]\n        for gg in g:\n            n = train_df[gg].nunique()\n            if n>mx:\n                mx = n\n                vx = gg\n        use.append(vx)\n    print(\"Use these\", use, \"\\n\")\n    return use","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:36:39.293137Z","iopub.execute_input":"2024-05-13T02:36:39.293450Z","iopub.status.idle":"2024-05-13T02:36:39.303366Z","shell.execute_reply.started":"2024-05-13T02:36:39.293423Z","shell.execute_reply":"2024-05-13T02:36:39.302387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnumeric_cols = train_df.select_dtypes(exclude='category').columns\n\nfrom itertools import combinations, permutations\nnans_df = train_df[numeric_cols].isna()\n# 결측치 기반 그룹화\nnans_groups = {} # 결측치 수를 key로, 해당 결측치 수를 갖는 cols를 list로 values로 저장함\nfor col in numeric_cols:\n    cur_group = nans_df[col].sum()\n    try:\n        nans_groups[cur_group].append(col)\n    except:\n        nans_groups[cur_group]=[col]\ndel nans_df\n\nuses = []\nfor k, v in nans_groups.items():\n    print(\"####### NAN count=\", k)\n    if len(v)>1:\n        Vs = nans_groups[k]\n        grps = group_columns_by_corr(train_df[Vs], threshold=0.8)\n        use = reduce_groups(grps)\n        uses = uses + use\n    else:\n        uses = uses + v\n        print(\"\\n\")\nprint(len(uses))\nuses = uses + list(train_df.select_dtypes(include='category').columns)\nprint(len(uses))","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:36:39.304667Z","iopub.execute_input":"2024-05-13T02:36:39.304992Z","iopub.status.idle":"2024-05-13T02:37:02.062988Z","shell.execute_reply.started":"2024-05-13T02:36:39.304966Z","shell.execute_reply":"2024-05-13T02:37:02.062027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df[uses]\ncols_train = [col for col in train_df.columns if col != \"target\"]\ntest_df = test_df[cols_train]\n\nprint(\"train data shape:\\t\", train_df.shape)\nprint(\"test data shape:\\t\", test_df.shape)\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:37:02.064013Z","iopub.execute_input":"2024-05-13T02:37:02.064294Z","iopub.status.idle":"2024-05-13T02:37:03.700509Z","shell.execute_reply.started":"2024-05-13T02:37:02.064270Z","shell.execute_reply":"2024-05-13T02:37:03.699635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subm_df = pd.read_csv(ROOT / \"sample_submission.csv\")\nsubm_df = subm_df.set_index(\"case_id\")\ndevice='gpu'\nn_est = 6000\n\nDRY_RUN = True if subm_df.shape[0] == 10 else False\nif DRY_RUN:\n    device = \"cpu\"\n    train_df = train_df.iloc[:50000]\n    n_est=600\n\nprint(device)","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:37:03.701667Z","iopub.execute_input":"2024-05-13T02:37:03.701939Z","iopub.status.idle":"2024-05-13T02:37:03.720689Z","shell.execute_reply.started":"2024-05-13T02:37:03.701916Z","shell.execute_reply":"2024-05-13T02:37:03.719856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"train_df_cols = train_df.columns\nX = train_df.drop(columns=['target', 'case_id', 'WEEK_NUM'])\nX[cat_cols] = X[cat_cols].astype('str') # for catboost\ny = train_df['target']\nweeks = train_df['WEEK_NUM']\n\ncv = StratifiedGroupKFold(n_splits=5, shuffle=False)\n\nlgb_params = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"max_depth\": 10,\n    \"learning_rate\": 0.05,\n    \"n_estimators\": 2000,\n    \"colsample_bytree\": 0.8, \n    \"colsample_bynode\": 0.8,\n    \"verbose\": -1,\n    \"random_state\": 42,\n    \"l1_regularization\": 0.1,\n    \"l2_regularization\": 10,\n    \"extra_trees\": True,\n    \"num_leavse\": 64,\n    \"device\": device,\n    \"min_data_in_bin\": 256\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:37:03.721585Z","iopub.execute_input":"2024-05-13T02:37:03.721862Z","iopub.status.idle":"2024-05-13T02:37:03.954400Z","shell.execute_reply.started":"2024-05-13T02:37:03.721840Z","shell.execute_reply":"2024-05-13T02:37:03.953617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nfitted_models_cat = []\n\ncv_scores_cat = []\n\noof_pred_cat = np.zeros(X.shape[0])\n\nfor train_idx, valid_idx in cv.split(X, y, groups=weeks):\n    X_train, y_train = X.iloc[train_idx], y.iloc[train_idx]\n    X_valid, y_valid = X.iloc[valid_idx], y.iloc[valid_idx]\n    train_pool = Pool(X_train, y_train, cat_features = cat_cols)\n    val_pool = Pool(X_valid, y_valid, cat_features = cat_cols)\n    \n    model_cat = CatBoostClassifier(\n        best_model_min_trees = 1000,\n        boosting_type = \"Plain\",\n        eval_metric = \"AUC\",\n        iterations = n_est,\n        learning_rate = 0.05,\n        l2_leaf_reg = 10,\n        max_leaves = 64,\n        random_seed = 42,\n        task_type = \"GPU\",\n        use_best_model = True\n    )\n    model_cat.fit(train_pool, eval_set=val_pool, verbose=200)\n    fitted_models_cat.append(model_cat)\n    \n    y_pred_valid = model_cat.predict_proba(X_valid)[:, 1]\n    oof_pred_cat[valid_idx] = y_pred_valid\n    auc_score = roc_auc_score(y_valid, y_pred_valid)\n    cv_scores_cat.append(auc_score)\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:37:03.955516Z","iopub.execute_input":"2024-05-13T02:37:03.955813Z","iopub.status.idle":"2024-05-13T02:41:47.710350Z","shell.execute_reply.started":"2024-05-13T02:37:03.955789Z","shell.execute_reply":"2024-05-13T02:41:47.709455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"CV AUC scores: \", cv_scores_cat)\nprint(\"Maximum CV AUC score: \", max(cv_scores_cat))","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:41:47.711548Z","iopub.execute_input":"2024-05-13T02:41:47.711922Z","iopub.status.idle":"2024-05-13T02:41:47.716827Z","shell.execute_reply.started":"2024-05-13T02:41:47.711883Z","shell.execute_reply":"2024-05-13T02:41:47.715900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nX[cat_cols] = X[cat_cols].astype('category')\n\nfitted_models_lgb = []\n\ncv_scores_lgb = []\n\noof_pred_lgb = np.zeros(X.shape[0])\n\nfor train_idx, valid_idx in cv.split(X, y, groups=weeks):\n    X_train, y_train = X.iloc[train_idx], y.iloc[train_idx]\n    X_valid, y_valid = X.iloc[valid_idx], y.iloc[valid_idx]\n    \n    model_lgb = lgb.LGBMClassifier(**lgb_params)\n    model_lgb.fit(\n        X_train, y_train,\n        eval_set=[(X_valid, y_valid)],\n        callbacks=[lgb.log_evaluation(100), lgb.early_stopping(100)]\n    )\n    fitted_models_lgb.append(model_lgb)\n    \n    y_pred_valid = model_lgb.predict_proba(X_valid)[:, 1]\n    oof_pred_lgb[valid_idx] = y_pred_valid\n    auc_score = roc_auc_score(y_valid, y_pred_valid)\n    cv_scores_lgb.append(auc_score)\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:41:47.718091Z","iopub.execute_input":"2024-05-13T02:41:47.718544Z","iopub.status.idle":"2024-05-13T02:42:21.860987Z","shell.execute_reply.started":"2024-05-13T02:41:47.718513Z","shell.execute_reply":"2024-05-13T02:42:21.860107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"CV AUC scores: \", cv_scores_lgb)\nprint(\"Maximum CV AUC score: \", max(cv_scores_lgb))","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:42:21.862216Z","iopub.execute_input":"2024-05-13T02:42:21.862560Z","iopub.status.idle":"2024-05-13T02:42:21.868037Z","shell.execute_reply.started":"2024-05-13T02:42:21.862527Z","shell.execute_reply":"2024-05-13T02:42:21.867227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# VotingModel","metadata":{}},{"cell_type":"code","source":"model = VotingModel(fitted_models_cat + fitted_models_lgb)","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:42:21.869282Z","iopub.execute_input":"2024-05-13T02:42:21.869546Z","iopub.status.idle":"2024-05-13T02:42:21.882207Z","shell.execute_reply.started":"2024-05-13T02:42:21.869525Z","shell.execute_reply":"2024-05-13T02:42:21.881465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.le_ = LabelEncoder().fit(y)\nmodel.classes_ = model.le_.classes_\n\njoblib.dump(model, \"HCB_cat_lgb.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:42:21.883115Z","iopub.execute_input":"2024-05-13T02:42:21.883371Z","iopub.status.idle":"2024-05-13T02:42:22.230303Z","shell.execute_reply.started":"2024-05-13T02:42:21.883350Z","shell.execute_reply":"2024-05-13T02:42:22.229333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"joblib.dump((train_df_cols, cat_cols), \"HCB_cat_lgb_train_cat_columns.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:42:22.231358Z","iopub.execute_input":"2024-05-13T02:42:22.231654Z","iopub.status.idle":"2024-05-13T02:42:22.238935Z","shell.execute_reply.started":"2024-05-13T02:42:22.231628Z","shell.execute_reply":"2024-05-13T02:42:22.238021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"joblib.dump(model.predict_proba(X)[:, 1], \"HCB_cat_lgb_pred.pkl\")\njoblib.dump(oof_pred_cat, \"HCB_cat_lgb_oof_pred_cat.pkl\")\njoblib.dump(oof_pred_lgb, \"HCB_cat_lgb_oof_pred_lgb.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:42:54.622529Z","iopub.execute_input":"2024-05-13T02:42:54.623232Z","iopub.status.idle":"2024-05-13T02:43:20.343196Z","shell.execute_reply.started":"2024-05-13T02:42:54.623203Z","shell.execute_reply":"2024-05-13T02:43:20.342277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction","metadata":{}},{"cell_type":"code","source":"X_test = test_df.drop(columns = ['WEEK_NUM'])\nX_test = X_test.set_index(\"case_id\")\n\ny_pred = pd.Series(model.predict_proba(X_test)[:, 1], index=X_test.index)\n\nsubm_df = pd.read_csv(ROOT / \"sample_submission.csv\")\nsubm_df = subm_df.set_index(\"case_id\")\n\nsubm_df['score'] = y_pred\nsubm_df","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:43:58.306841Z","iopub.execute_input":"2024-05-13T02:43:58.307219Z","iopub.status.idle":"2024-05-13T02:43:58.807282Z","shell.execute_reply.started":"2024-05-13T02:43:58.307189Z","shell.execute_reply":"2024-05-13T02:43:58.806479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subm_df.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-05-13T02:44:01.835487Z","iopub.execute_input":"2024-05-13T02:44:01.836368Z","iopub.status.idle":"2024-05-13T02:44:01.842794Z","shell.execute_reply.started":"2024-05-13T02:44:01.836334Z","shell.execute_reply":"2024-05-13T02:44:01.841835Z"},"trusted":true},"execution_count":null,"outputs":[]}]}