{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":8189312,"sourceType":"datasetVersion","datasetId":4849478}],"dockerImageVersionId":30698,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install /kaggle/input/scikit-learn-1-4-2-cp310/scikit_learn-1.4.2-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Summary\n- **Algorithm**: scikit-learn's HistGradientBoostingClassifier\n    - Pros\n        - Similar to LightGBM with missing values support\n        - For v1.4.2, categorical features can be automatically detected\n        - Faster training speed on CPU compared to other frameworks\n        - Less hyperparameters\n        - Good results\n    - Cons\n        - No GPU acceleration, hyperparameter tuning can take very long\n        - Unable to specify desired validation set, only can specify `validation_fraction` which is sampled from train set\n- Feature selection\n    - Numerical: Pearson correlation\n    - Categorical: Cramer's V","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport glob\nimport pickle\nimport psutil\nfrom time import time\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport lightgbm as lgb\nimport xgboost as xgb\nfrom catboost import CatBoostClassifier, Pool\nfrom sklearn.ensemble import HistGradientBoostingClassifier\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.metrics import roc_auc_score \nfrom scipy.stats import chi2_contingency","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_execution_time(t):\n    print(f'Total time = {time() - t:.2f}s')\n\n\ndef get_memory():\n    print(f'Available memory left: {psutil.virtual_memory().available / (1024 * 1024 * 1024):.2f}gb')\n\n    \nget_memory()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preparation","metadata":{}},{"cell_type":"code","source":"total_t = time()\nSEED = 42\nMISSING_PERCENTAGE = 0.95\nCATEGORIES_MAX = 200\nDATA_PATH = \"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/\"\nTABLES = {\n    'depth_0': ['static_0_*', 'static_cb_0'],\n    'depth_1': [\n        'applprev_1_*', \n        'credit_bureau_a_1_*',\n        'credit_bureau_b_1',\n        'debitcard_1',\n        'deposit_1',\n        'other_1',\n        'person_1',\n        'tax_registry_a_1',\n        'tax_registry_b_1',\n        'tax_registry_c_1',\n    ],\n    'depth_2': [\n        'applprev_2',\n        'credit_bureau_a_2_*',\n        'credit_bureau_b_2',   \n        'person_2',\n    ],\n}\nPROCESSED_DATA_PATH = ''\nSELECTED_COLS = []","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_memory(df):\n    return df.select(pl.all().shrink_dtype())\n\n\ndef set_dtypes(df):\n    for col in df.columns:\n        if col in (\"target\", \"case_id\", \"WEEK_NUM\", \"MONTH\"):\n            df = df.with_columns(pl.col(col).cast(pl.Int32))\n        elif col == \"date_decision\" or col[-1] == \"D\":\n            df = df.with_columns(pl.col(col).cast(pl.Date))\n        elif col[-1] in (\"P\", \"A\") or (\"num_group\" in col):\n            df = df.with_columns(pl.col(col).cast(pl.Float32))\n        elif col[-1] == \"M\":\n            df = df.with_columns(pl.col(col).cast(pl.String))\n    return df\n\n\ndef filter_columns(df):\n    cols_to_drop = set()\n\n    for col in df.columns:\n        if col in (\"target\", \"case_id\", \"WEEK_NUM\", \"MONTH\"):\n            continue\n\n        # Remove categorical columns with 1 or >CATEGORIES_MAX categories\n        if df[col].dtype == pl.String:\n            n = df[col].n_unique()\n            if (n == 1) or (n > CATEGORIES_MAX):\n                cols_to_drop.add(col)\n                continue\n\n        # Remove columns with >MISSING_PERCENTAGE missing values\n        nulls = df[col].is_null().mean()\n        if nulls > MISSING_PERCENTAGE:\n            cols_to_drop.add(col)\n\n    return df.drop(list(cols_to_drop))\n\n\ndef aggregate(df):\n    exprs = []\n\n    for col, dtype in zip(df.columns, df.dtypes):\n        if col in (\"target\", \"case_id\", \"WEEK_NUM\", \"MONTH\"):\n            continue\n        \n        exprs += [\n            pl.col(col).max().alias(f\"max_{col}\"),\n        ]\n        \n        if col[-1] in (\"P\", \"A\", \"D\"):\n            exprs += [\n                pl.col(col).mean().alias(f\"mean_{col}\"),\n                pl.col(col).var().alias(f\"var_{col}\"),\n            ]\n\n    return exprs\n\n\ndef handle_dates(df):\n    assert \"date_decision\" in df.columns, \"date_decision not in df\"\n    for col in df.columns:\n        if col[-1] == \"D\":\n            df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n            df = df.with_columns(pl.col(col).dt.total_days().cast(pl.Float32))\n        elif col in [\n            \"dpdmaxdateyear_596T\",\n            \"dpdmaxdateyear_742T\",\n            \"dpdmaxdateyear_896T\",\n            \"overdueamountmaxdateyear_2T\",\n            \"overdueamountmaxdateyear_432T\",\n            \"overdueamountmaxdateyear_994T\",\n            \"pmts_year_1139T\",\n            \"pmts_year_507T\",\n        ]:  # These columns are represented in years (ex: 2020, 2021, ...)\n            df = df.with_columns(\n                pl.col(col) - pl.col(\"date_decision\").dt.year().cast(pl.Float32)\n            )\n\n    return df.drop(\"date_decision\")\n\n\ndef read_files(split=\"train\"):\n    df = pl.read_parquet(DATA_PATH + f\"{split}/{split}_base.parquet\").pipe(set_dtypes)\n    df = df.with_columns([\n        pl.col(\"date_decision\").dt.month().alias(\"month_decision\"),\n        pl.col(\"date_decision\").dt.weekday().alias(\"weekday_decision\"),\n    ])\n    date_decision = df.select(pl.col([\"case_id\", \"date_decision\"]))\n\n    for key, item in TABLES.items():\n        print(f\"#\\tHandling {key}\")\n        t = time()\n\n        for i in item:\n            print(f\"##\\tProcessing {i}\")\n            sub_df = pl.DataFrame()\n\n            for file in glob.glob(DATA_PATH + f\"{split}/{split}_{i}.parquet\"):\n                dummy = (\n                    pl.read_parquet(file)\n                    .pipe(set_dtypes)\n                    .join(date_decision, how=\"left\", on=\"case_id\")\n                    .pipe(handle_dates)\n                )\n\n                if key != \"depth_0\" and not dummy.is_empty():\n                    dummy = dummy.group_by(\"case_id\").agg(aggregate(dummy))\n\n                sub_df = (\n                    dummy\n                    if sub_df.is_empty()\n                    else pl.concat([sub_df, dummy], how=\"diagonal_relaxed\")\n                )\n\n            df = df.join(\n                sub_df.unique(subset=[\"case_id\"]),\n                how=\"left\",\n                on=\"case_id\",\n                suffix=f\"_{i}\",\n            )\n            if split == \"train\":\n                df = df.pipe(filter_columns)\n\n    return df.drop([\"date_decision\", \"MONTH\"]).pipe(reduce_memory)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Get Train Set","metadata":{}},{"cell_type":"code","source":"def get_train():\n    t = time()\n    print(\"~~~~~~~~~~~~~~~~~~~~~~~~~~~\")\n    if PROCESSED_DATA_PATH == \"\":\n        print(\"Reading files\")\n        df = read_files().pipe(filter_columns)\n        df.write_parquet(\"./data.parquet\")\n    else:\n        print(\"Processed data found, use these instead\")\n        df = pl.read_parquet(PROCESSED_DATA_PATH).select(SELECTED_COLS)\n    get_memory()\n    get_execution_time(t)\n\n    print(f\"Unique dtypes: {set(df.dtypes)}\")\n    cat_cols, num_cols = [], []\n    for col, dtype in zip(df.columns, df.dtypes):\n        if col in [\"target\", \"WEEK_NUM\", \"case_id\"]:\n            continue\n\n        if dtype in (pl.String, pl.Boolean):\n            cat_cols.append(col)\n        else:\n            num_cols.append(col)\n\n    t = time()\n    print(\"~~~~~~~~~~~~~~~~~~~~~~~~~~~\")\n    print(\"Converting to pandas\")\n    df = df.to_pandas()\n    df[cat_cols] = df[cat_cols].astype(\"category\")\n\n    print(f\"{len(df.index)} rows and {len(df.columns)} columns\")\n    print(f\"{len(cat_cols)} categorical and {len(num_cols)} numerical columns\")\n    get_memory()\n    get_execution_time(t)\n\n    return df.copy(), cat_cols, num_cols","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nprint('Getting train set')\ndf, cat_cols, num_cols = get_train()\ndf.tail()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Selection","metadata":{}},{"cell_type":"code","source":"def cramers_v(x, y):\n    # https://stackoverflow.com/questions/20892799/using-pandas-calculate-cram%C3%A9rs-coefficient-matrix\n    cm = pd.crosstab(x, y)\n    chi2 = chi2_contingency(cm)[0]\n    n = cm.sum().sum()\n    phi2 = chi2 / n\n    r, k = cm.shape\n    phi2corr = max(0, phi2 - ((k - 1) * (r - 1)) / (n - 1))\n    rcorr = r - ((r - 1) ** 2) / (n - 1)\n    kcorr = k - ((k - 1) ** 2) / (n - 1)\n    return np.sqrt(phi2corr / min((kcorr - 1), (rcorr - 1)))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# https://www.kaggle.com/code/harrychan123/lgb-cat-ensemble-stacking\nprint(\"\\nFinding numerical columns with high correlation\")\nnans_df = df[num_cols].isna()\nnans_groups = {}\nfor col in num_cols:\n    cur_group = nans_df[col].sum()\n    try:\n        nans_groups[cur_group].append(col)\n    except:\n        nans_groups[cur_group] = [col]\ndel nans_df\ngc.collect()\n\n\ndef reduce_group(grps, df):\n    use = []\n    for g in grps:\n        mx, vx = 0, g[0]\n        for gg in g:\n            n = df[gg].nunique()\n            if n > mx:\n                mx = n\n                vx = gg\n        use.append(vx)\n    return use\n\n\ndef group_columns_by_correlation(matrix, threshold=0.9):\n    correlation_matrix = matrix.corr()\n\n    groups = []\n    remaining_cols = list(matrix.columns)\n    while remaining_cols:\n        col = remaining_cols.pop(0)\n        group = [col]\n        correlated_cols = [col]\n        for c in remaining_cols:\n            if correlation_matrix.loc[col, c] >= threshold:\n                group.append(c)\n                correlated_cols.append(c)\n        groups.append(group)\n        remaining_cols = [c for c in remaining_cols if c not in correlated_cols]\n\n    return groups\n\n\nuses = []\nfor k, v in nans_groups.items():\n    if len(v) > 1:\n        Vs = nans_groups[k]\n        grps = group_columns_by_correlation(df[Vs])\n        use = reduce_group(grps, df)\n        uses = uses + use\n    else:\n        uses = uses + v\n\nto_remove = set([col for col in num_cols if col not in uses])\nprint(f\"{len(to_remove)} columns are to be dropped\")\ndf = df.drop(list(to_remove), axis=1)\nnum_cols = [item for item in num_cols if item not in to_remove]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Remove categorical columns with high association (Cramer's V)\nprint(\"\\nFinding categorical columns with high association\")\ndummy = df[cat_cols].astype(\"str\").astype(\"category\")\n\nto_remove_cat = set()\nfor i in range(len(cat_cols)):\n    for j in range(i + 1, len(cat_cols)):\n        col1, col2 = cat_cols[i], cat_cols[j]\n        if col1 == col2 or col1 in to_remove_cat or col2 in to_remove_cat:\n            continue\n\n        corr = cramers_v(dummy[col1], dummy[col2])\n        if corr > 0.9:\n            print(f\"{col1} & {col2} = {corr}\")\n            to_remove_cat.add(col2)\n\nprint(f\"{len(to_remove_cat)} columns are to be dropped\")\ndf = df.drop(list(to_remove_cat), axis=1)\ncat_cols = [item for item in cat_cols if item not in to_remove_cat]\ndel dummy\ngc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.tail()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'\\nColumns to use {len(df.columns)}: ')\nprint(df.columns.tolist())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Get Test Set","metadata":{}},{"cell_type":"code","source":"def get_test(features, dtypes):\n    df_submission = read_files(split=\"test\").to_pandas()\n    initial_cols = df_submission.columns.to_list()\n\n    # Check for missing columns\n    missing, mapping = {}, {}\n\n    for col, dtype in zip(features, dtypes):\n        mapping[col] = dtype\n\n        if col not in initial_cols:\n            missing[col] = np.repeat(\n                np.nan if dtype.kind in \"iufc\" else None,\n                len(df_submission.index),\n            )\n\n    df_submission = pd.concat([df_submission, pd.DataFrame(missing)], axis=1).astype(\n        mapping\n    )\n\n    return df_submission[[\"case_id\"] + features].copy()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nprint('Getting test set')\nfeatures = num_cols + cat_cols\ndtypes = df[features].dtypes.to_list()\ndf_submission = get_test(features, dtypes)\nprint(f'Test set has {len(df_submission.index)} rows and {len(df_submission.columns)} columns')\nget_memory()\ndf_submission.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"def gini_stability(base, w_fallingrate=88.0, w_resstd=-0.5):\n    # https://www.kaggle.com/code/jetakow/home-credit-2024-starter-notebook\n    gini_in_time = (\n        base.loc[:, [\"WEEK_NUM\", \"target\", \"score\"]]\n        .sort_values(\"WEEK_NUM\")\n        .groupby(\"WEEK_NUM\")[[\"target\", \"score\"]]\n        .apply(lambda x: 2 * roc_auc_score(x[\"target\"], x[\"score\"]) - 1)\n        .tolist()\n    )\n\n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a * x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df[features].copy()\ny = df['target'].copy()\ngroups = df['WEEK_NUM'].copy()\ncase_id = df['case_id'].copy()\ncv = StratifiedGroupKFold(n_splits=5).split(X, y, groups)\n\ndel df\ngc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params_hist = {\n    'learning_rate': 0.05,\n    'max_iter': 2000,\n    'max_leaf_nodes': 64,\n    'max_depth': 10,\n    'max_features': 0.8,\n    'l2_regularization': 10,\n    'categorical_features': \"from_dtype\",\n    'early_stopping': True,\n    'scoring': 'roc_auc',\n    'n_iter_no_change': 50,\n    'random_state': SEED,\n    'validation_fraction': 0.2,\n}","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hist_models, rocs, ginis = [], [], []\n\nfor i, (train_index, valid_index) in enumerate(cv):\n    print(\"~~~~~~~~~~~~~~~~~~~~~~~~~~~\")\n    print(f\"Currently on fold {i + 1}\")\n\n    t = time()\n\n    # Training\n    hist = HistGradientBoostingClassifier(**params_hist)\n    hist.fit(X.iloc[train_index], y.iloc[train_index])\n\n    # Evaluation\n    y_preds = hist.predict_proba(X.iloc[valid_index])[:, 1]\n    base = pd.DataFrame({\n        \"WEEK_NUM\": groups.iloc[valid_index],\n        \"target\": y.iloc[valid_index],\n        \"score\": y_preds,\n    })\n\n    roc = roc_auc_score(y.iloc[valid_index], y_preds)\n    gini = gini_stability(base)\n\n    print(f\"AUC: {roc}\")\n    print(f\"Stability: {gini}\")\n\n    # Saving\n    hist_models.append(hist)\n    rocs.append(roc)\n    ginis.append(gini)\n\n    get_memory()\n    get_execution_time(t)\n\n    del hist\n    gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Average AUC score: {np.mean(rocs)}')\nprint(f'Average gini score: {np.mean(ginis)}')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('hist.pickle', 'wb') as handle:\n    pickle.dump(hist_models, handle, protocol=pickle.HIGHEST_PROTOCOL)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"class VotingModel:\n    # https://www.kaggle.com/code/kononenko/metric-trick-home-credit-baseline-inference\n    def __init__(self, estimators, batch_size=100000):\n        self.estimators = estimators\n        self.batch_size = batch_size\n    \n    def predict_proba_in_batches(self, model, data):\n        num_samples = len(data)\n        num_batches = int(np.ceil(num_samples / self.batch_size))\n        probabilities = np.zeros((num_samples, 2))\n\n        for batch_idx in range(num_batches):\n            start_idx = batch_idx * self.batch_size\n            end_idx = min((batch_idx + 1) * self.batch_size, num_samples)\n            probabilities[start_idx:end_idx, :] = model.predict_proba(data.iloc[start_idx:end_idx])\n            gc.collect()\n        \n        return probabilities      \n    \n    def predict_proba(self, X):\n        y_preds = [self.predict_proba_in_batches(estimator, X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)    ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nprint('Predicting on test set')\nhist_models = VotingModel(hist_models)\ndf_submission['score'] = hist_models.predict_proba(df_submission[features])[:, 1]\nprint(df_submission[['case_id', 'score']].to_string())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission[['case_id', 'score']].to_csv('./submission.csv', index=None)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Notebook completed in {(time() - total_t) / 60:.2f}min')\nget_memory()","metadata":{},"execution_count":null,"outputs":[]}]}