{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\n\nimport tqdm\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.metrics import roc_auc_score\n\nimport lightgbm as lgb\nfrom catboost import CatBoostClassifier\n# import xgboost as xgb","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-17T19:26:41.281465Z","iopub.execute_input":"2024-05-17T19:26:41.281809Z","iopub.status.idle":"2024-05-17T19:26:45.994401Z","shell.execute_reply.started":"2024-05-17T19:26:41.281782Z","shell.execute_reply":"2024-05-17T19:26:45.992702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CAT_NAN_VALUE = '_NAN_VALUE_'\nCAT_RARE_VALUE = '_RARE_VALUE_'","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:26:45.996510Z","iopub.execute_input":"2024-05-17T19:26:45.997039Z","iopub.status.idle":"2024-05-17T19:26:46.001576Z","shell.execute_reply.started":"2024-05-17T19:26:45.997011Z","shell.execute_reply":"2024-05-17T19:26:46.000446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Custom loss function","metadata":{}},{"cell_type":"code","source":"def _validate_labels_preds(labels, preds):\n    # Check y_true and y_pred\n    assert len(labels.shape) == len(preds.shape) == 1\n    assert labels.shape[0] == preds.shape[0]\n\n\ndef _split_labels_into_neg_pos(labels):\n    # Return indices of negative and positive samples\n    assert len(labels.shape) == 1\n    # assert np.all(np.in1d(labels, [0, 1]))\n\n    neg_cls_mask = labels <= 0.5\n    pos_cls_mask = ~neg_cls_mask\n\n    assert len(neg_cls_mask.shape) == len(pos_cls_mask.shape) == 1\n    assert neg_cls_mask.shape[0] == pos_cls_mask.shape[0] == labels.shape[0]\n    return neg_cls_mask, pos_cls_mask\n\n\ndef _split_labels_into_groups(samples_n, approx_group_size):\n    # Return size for each group\n    assert samples_n >= 1\n    assert approx_group_size >= 1\n\n    idx = np.arange(samples_n)\n    groups_n = max(1, samples_n // approx_group_size)\n\n    idx_groups = np.array_split(idx, groups_n)\n    assert len(idx_groups) >= 1\n\n    end_idx_groups = np.cumsum(\n        np.array([len(idx_group) for idx_group in idx_groups])\n    )\n    begin_idx_groups = np.concatenate([\n        np.array([0]),\n        end_idx_groups[:-1]\n    ])\n\n    groups = np.array([begin_idx_groups, end_idx_groups]).T\n\n    assert len(groups) == groups_n\n    assert 1 <= groups_n <= samples_n\n    return groups\n\n\nclass StableRocAucLoss:\n    \"\"\"Maximize ROC-AUC and minimize its variance simultaneously.\"\"\"\n\n    def __init__(self, approx_group_size, alpha=0.5, beta=0.5, gamma=0.2):\n        self.approx_group_size = approx_group_size\n        self.alpha = alpha\n        self.beta = beta\n        self.gamma = gamma\n\n    def validate(self):\n        \"\"\"Throw an exception in case of any inconsistency.\"\"\"\n        assert self.approx_group_size >= 1  # 2\n        assert 0.0 <= self.alpha <= 1.0\n        assert 0.0 <= self.beta <= 1.0\n        assert 0.0 <= self.gamma <= 1.0  # [0.1; 0.7]\n\n    def compute_value(self, labels, preds):\n        \"\"\"Return value of the loss at the specified point.\"\"\"\n        self.validate()\n        _validate_labels_preds(labels, preds)\n\n        if labels.shape[0] < 1:\n            return 0.0\n\n        intermediate_losses = self._compute_intermediate_losses(labels, preds)\n        groups_n = intermediate_losses.shape[0]\n        assert groups_n >= 1\n        assert intermediate_losses.shape == (groups_n,)\n        assert len(intermediate_losses.shape) == 1\n\n        mu = np.sum(intermediate_losses / groups_n)\n        term1 = np.sum(1.0 / groups_n * np.square(intermediate_losses))\n        term2 = np.sum(1.0 / groups_n * np.square(intermediate_losses - mu))\n        return self.alpha * term1 + self.beta * term2\n\n    def compute_grad_hess(self, labels, preds):\n        \"\"\"Return the first derivatives of the loss.\"\"\"\n        self.validate()\n        _validate_labels_preds(labels, preds)\n\n        if labels.shape[0] < 1:\n            return np.zeros(labels.shape), np.zeros(labels.shape)\n\n        groups = _split_labels_into_groups(\n            samples_n=labels.shape[0],\n            approx_group_size=self.approx_group_size\n        )\n\n        intermediate_losses = self._compute_intermediate_losses(labels, preds)\n        outer_grad_hess = self._compute_outer_grad_hess(intermediate_losses)\n        inner_grad_hess = [\n            self._compute_inner_grad_hess(\n                labels=labels[group[0]:group[1]],\n                preds=preds[group[0]:group[1]]\n            ) for group in groups\n        ]\n\n        assert len(outer_grad_hess) == 2\n        assert len(outer_grad_hess[0]) == len(outer_grad_hess[1])\n        assert len(outer_grad_hess[0]) == groups.shape[0]\n        assert len(inner_grad_hess) == groups.shape[0]\n\n        grad = np.concatenate([\n            outer_grad_hess[0][i] * inner_grad_hess[i][0]\n            for i in range(groups.shape[0])\n        ])\n        hess = np.concatenate([\n            outer_grad_hess[1][i] * inner_grad_hess[i][0]**2\n            + outer_grad_hess[0][i] * inner_grad_hess[i][1]\n            for i in range(groups.shape[0])\n        ])\n\n        assert grad.shape == hess.shape == labels.shape\n        return grad, hess\n\n    def _compute_intermediate_loss(self, labels, preds):\n        # Return U_i\n        self.validate()\n        _validate_labels_preds(labels, preds)\n\n        neg_cls_mask, pos_cls_mask = _split_labels_into_neg_pos(labels)\n        neg_n = np.count_nonzero(neg_cls_mask)\n        pos_n = np.count_nonzero(pos_cls_mask)\n        if neg_n < 1 or pos_n < 1:\n            return 0.0\n\n        u_i = np.subtract.outer(preds[pos_cls_mask], preds[neg_cls_mask])\n        u_i = np.square(np.fmax(self.gamma - u_i, 0.0))\n        # u_i = np.where(u_i < self.gamma, (self.gamma - u_i)**2, 0.0)\n        u_i = np.sum(u_i / neg_n / pos_n) / (1.0 + self.gamma)**2\n\n        assert u_i >= 0.0\n        return u_i\n\n    def _compute_intermediate_losses(self, labels, preds):\n        # Return U_1, U_2, ..., U_m\n        self.validate()\n        _validate_labels_preds(labels, preds)\n\n        groups = _split_labels_into_groups(\n            samples_n=labels.shape[0],\n            approx_group_size=self.approx_group_size\n        )\n\n        intermediate_losses = np.array([\n            self._compute_intermediate_loss(\n                labels=labels[group[0]:group[1]],\n                preds=preds[group[0]:group[1]]\n            ) for group in groups\n        ])\n\n        assert intermediate_losses.shape == (groups.shape[0],)\n        return intermediate_losses\n\n    def _compute_outer_grad_hess(self, intermediate_losses):\n        # Return dL / dU_i and d^2 L / dU_i^2, i = 1, 2, ..., m\n        self.validate()\n        assert len(intermediate_losses.shape) == 1\n\n        groups_n = intermediate_losses.shape[0]\n        if groups_n < 1:\n            return np.zeros(groups_n), np.zeros(groups_n)\n\n        mu = np.sum(intermediate_losses / groups_n)\n        diff = intermediate_losses - mu\n\n        grad_term1 = self.alpha / groups_n * intermediate_losses\n        grad_term2 = self.beta / groups_n * diff\n        # grad_term3 = self.beta / groups_n * np.sum(diff / groups_n)\n        grad = 2.0 * (grad_term1 + grad_term2)\n\n        hess_term1 = 2.0 * self.alpha / groups_n\n        hess_term2 = 2.0 * self.beta * (groups_n - 1) / groups_n / groups_n\n        hess = np.repeat(hess_term1 + hess_term2, groups_n)\n\n        return grad, hess\n\n    def _compute_inner_grad_hess(self, labels, preds):\n        # Return dU_i / dz and d^2 U_i / dz^2\n        self.validate()\n        _validate_labels_preds(labels, preds)\n\n        neg_cls_mask, pos_cls_mask = _split_labels_into_neg_pos(labels)\n        neg_n = np.count_nonzero(neg_cls_mask)\n        pos_n = np.count_nonzero(pos_cls_mask)\n        if neg_n < 1 or pos_n < 1:\n            return np.zeros(preds.shape[0]), np.zeros(preds.shape[0])\n\n        factor = 2.0 / (1.0 + self.gamma) / (1.0 + self.gamma)\n        factor = factor / neg_n / pos_n\n\n        r = np.subtract.outer(preds[pos_cls_mask], preds[neg_cls_mask])\n        pos_grad = np.fmin(r - self.gamma, 0.0)\n        neg_grad = -1.0 * pos_grad\n        hess = np.where(r < self.gamma, factor, 0.0)\n\n        neg_grad = np.sum(factor * neg_grad, axis=0)\n        pos_grad = np.sum(factor * pos_grad, axis=1)\n        neg_hess = np.sum(hess, axis=0)\n        pos_hess = np.sum(hess, axis=1)\n\n        grad, hess = np.zeros(labels.shape), np.zeros(labels.shape)\n        grad[neg_cls_mask], grad[pos_cls_mask] = neg_grad, pos_grad\n        hess[neg_cls_mask], hess[pos_cls_mask] = neg_hess, pos_hess\n\n        return grad, hess","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:26:46.007855Z","iopub.execute_input":"2024-05-17T19:26:46.008283Z","iopub.status.idle":"2024-05-17T19:26:46.038679Z","shell.execute_reply.started":"2024-05-17T19:26:46.008248Z","shell.execute_reply":"2024-05-17T19:26:46.037845Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"custom_loss = StableRocAucLoss(\n    approx_group_size=30000,\n    alpha=0.8,\n    beta=1.0,\n    gamma=0.25\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:26:46.039756Z","iopub.execute_input":"2024-05-17T19:26:46.040034Z","iopub.status.idle":"2024-05-17T19:26:46.058565Z","shell.execute_reply.started":"2024-05-17T19:26:46.040011Z","shell.execute_reply":"2024-05-17T19:26:46.057286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class CatBoostWrapperLoss:\n#     def calc_ders_range(self, approxes, targets, weights):\n#         # weights are ignored\n#         assert len(approxes) == len(targets)\n#         if weights is not None:\n#             assert len(weights) == len(approxes)\n\n#         grad, hess = custom_loss.compute_grad_hess(\n#             labels=np.array([label for label in targets]),\n#             preds=np.array([pred for pred in approxes])\n#         )\n\n#         result = []\n#         assert len(grad) == len(hess) == len(targets)\n#         for i in range(len(targets)):\n#             result.append((grad[i], hess[i]))\n\n#         return result","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:26:46.059791Z","iopub.execute_input":"2024-05-17T19:26:46.060123Z","iopub.status.idle":"2024-05-17T19:26:46.069411Z","shell.execute_reply.started":"2024-05-17T19:26:46.060095Z","shell.execute_reply":"2024-05-17T19:26:46.068196Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class CatBoostWrapperMetric:\n#     def get_final_error(self, error, weight):\n#         return error\n\n#     def is_max_optimal(self):\n#         return False\n\n#     def evaluate(self, approxes, target, weight):\n#         assert len(approxes) == 1\n#         assert len(target) == len(approxes[0])\n#         assert weight is None\n\n#         approx = np.mean(approxes, axis=0)\n#         return custom_loss.compute_value(\n#             labels=np.array([label for label in target]),\n#             preds=np.array([pred for pred in approx])\n#         ), target.shape[0]","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:26:46.070891Z","iopub.execute_input":"2024-05-17T19:26:46.071296Z","iopub.status.idle":"2024-05-17T19:26:46.087476Z","shell.execute_reply.started":"2024-05-17T19:26:46.071256Z","shell.execute_reply":"2024-05-17T19:26:46.086102Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utility functions","metadata":{}},{"cell_type":"code","source":"def plot_roc_curves(metric_df):\n    assert len(metric_df.shape) == 2\n\n    plt.figure(figsize=(5, 5))\n    unique_weeks = metric_df['WEEK_NUM'].unique()\n\n    for week in unique_weeks:\n        week_df = metric_df[metric_df['WEEK_NUM'] == week]\n        week_y_pred = week_df['score'].to_numpy()\n        week_y_true = week_df['target'].to_numpy()\n        fpr, tpr, _ = roc_curve(week_y_true, week_y_pred)\n        auc = roc_auc_score(week_y_true, week_y_pred)\n        if auc < 0.8:\n            plt.plot(fpr, tpr, label='week ' + str(week) + '; auc = ' + str(auc))\n        else:\n            plt.plot(fpr, tpr)\n\n    plt.plot([0.0, 1.0], [0.0, 1.0], color='black', linestyle='--')\n    plt.legend()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:26:46.088689Z","iopub.execute_input":"2024-05-17T19:26:46.089044Z","iopub.status.idle":"2024-05-17T19:26:46.099516Z","shell.execute_reply.started":"2024-05-17T19:26:46.088994Z","shell.execute_reply":"2024-05-17T19:26:46.098446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_batches(metric_df):\n    assert len(metric_df.shape) == 2\n\n    weeks = np.sort(metric_df['WEEK_NUM'].unique())\n    pos_share = []\n    samples_n = []\n\n    for week in weeks:\n        week_df = metric_df[metric_df['WEEK_NUM'] == week]\n        # week_y_pred = week_df['score'].to_numpy()\n        week_y_true = week_df['target'].to_numpy()\n        pos_share.append(np.sum(week_y_true) / week_y_true.shape[0])\n        samples_n.append(week_df.shape[0])\n    \n    fig, ax1 = plt.subplots(figsize=(16, 4))\n    ax2 = ax1.twinx()\n    ax1.plot(weeks, samples_n, marker='o', label='samples number', color='tab:blue')\n    ax2.plot(weeks, pos_share, marker='o', label='pos class share', color='tab:orange')\n    ax1.legend(loc='upper center')\n    ax2.legend(loc='upper right')","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:26:46.101426Z","iopub.execute_input":"2024-05-17T19:26:46.101867Z","iopub.status.idle":"2024-05-17T19:26:46.113897Z","shell.execute_reply.started":"2024-05-17T19:26:46.101835Z","shell.execute_reply":"2024-05-17T19:26:46.112882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def compute_gini_stability_metric(base, w_fallingrate=88.0, w_resstd=-0.5, debug=False):\n    assert len(base.shape) == 2\n\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", \"score\"]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", \"score\"]]\\\n        .apply(lambda x: 2 * roc_auc_score(x[\"target\"], x[\"score\"]) - 1).tolist()\n\n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a * x + b\n\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    score = avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std\n\n    if debug:\n        print('a =', a, '; b =', b, '; res_std =', res_std, '; avg_gini =', avg_gini)\n        print('score =', score)\n        plt.figure(figsize=(16, 3))\n        plt.plot(x, y, marker='o', label='gini_in_time')\n        plt.plot(x, y_hat, marker='o', label='y_hat')\n        plt.ylim([0.4, 0.8])\n        plt.legend()\n\n    return score","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:26:46.118583Z","iopub.execute_input":"2024-05-17T19:26:46.118957Z","iopub.status.idle":"2024-05-17T19:26:46.132128Z","shell.execute_reply.started":"2024-05-17T19:26:46.118929Z","shell.execute_reply":"2024-05-17T19:26:46.130821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Pipeline","metadata":{}},{"cell_type":"code","source":"class Pipeline:\n\n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int32))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                df = df.with_columns(pl.col(col).cast(pl.Float32))\n\n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.95:  # 0.7\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) and (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq <= 1) or (freq > 255):  # (freq <= 1) | (freq > 200)\n                    df = df.drop(col)\n\n        return df","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:26:46.133334Z","iopub.execute_input":"2024-05-17T19:26:46.133698Z","iopub.status.idle":"2024-05-17T19:26:46.146095Z","shell.execute_reply.started":"2024-05-17T19:26:46.133667Z","shell.execute_reply":"2024-05-17T19:26:46.144915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Automatic Aggregation","metadata":{}},{"cell_type":"code","source":"class Aggregator:\n\n    @staticmethod\n    def _applicant_max_expr(df, depth, cols):\n        assert depth == 1\n        assert \"num_group1\" in df.columns\n        assert \"num_group2\" not in df.columns\n        assert \"num_group1\" not in cols\n\n        redundant_cols = [\n            'creationdate_885D',\n            'recorddate_4527225D',\n            'deductiondate_4917603D',\n            'debtoutstand_525A',\n            'debtoverdue_47A',\n            'empl_employedfrom_271D',\n            'empl_employedtotal_800L',\n            'empl_industry_691L',\n            'familystate_447L',\n            'housetype_905L',\n            'incometype_1044T',\n            'safeguarantyflag_411L',\n            'sex_738L',\n            'revolvingaccount_394A',\n            'residualamount_488A',\n            'instlamount_768A',\n            'instlamount_852A',\n            'outstandingamount_362A',\n            'totaldebtoverduevalue_178A',\n            'totaloutstanddebtvalue_39A',\n            'dateofcredstart_181D',\n            'numberofoverdueinstlmaxdat_641D',\n            'overdueamountmax2date_1142D',\n            'numberofcontrsvalue_258L',\n            'maxdpdtolerance_577P',\n            'outstandingdebt_522A',\n            'actualdpd_943P',\n            'financialinstitution_591M',\n            'contractst_545M',\n            'contractst_964M',\n        ]\n\n        return [\n            pl.col(\"num_group1\", col)\n            .filter(pl.col(\"num_group1\") == 0)\n            .exclude(\"num_group1\")\n            .max()\n            .alias(f\"applicant_max_{col}\")\n            for col in cols if col not in redundant_cols\n        ]\n\n    @staticmethod\n    def _above_expr(df, depth, cols, threshold):\n        assert depth in [1, 2]\n        assert \"num_group1\" in df.columns\n        return [\n            pl.col(col)\n            .filter(pl.col(col) > threshold)\n            .len()\n            .cast(pl.Float64)\n            .truediv(1 + pl.len())\n            .alias(f\"above_{threshold}_share_{col}\")\n            for col in cols\n        ]\n\n    @staticmethod\n    def _mode_share_expr(df, depth, cols):\n        assert depth in [1, 2]\n        assert \"num_group1\" in df.columns\n        \n        additional_param_cols = [\n            'numberofoverdueinstlmax_1151L',\n            'postype_4733339M',\n            'rejectreasonclient_4145042M',\n            'purposeofcred_874M',\n            'purposeofcred_426M',\n            'purposeofcred_722M',\n            'classificationofcontr_13M',\n            'contractst_545M',\n            'financialinstitution_591M',\n            'numberofoutstandinstls_520L',\n            'employername_M',\n            'name_4527232M',\n            'name_4917606M',\n            'employername_160M',\n            'numberofcontrsvalue_358L',\n            'contaddr_district_15M',\n            'registaddr_district_1083M',\n            'empladdr_district_926M',\n            'empladdr_zipcode_114M',\n        ]\n\n        expr = []\n\n        for col in cols:\n            if df[col].dtype == pl.String:\n                mode1 = pl.col(col).drop_nulls().mode().max()\n                # mode2 = pl.col(col).filter(pl.col(col) != mode1).mode().first()\n                expr.append(mode1.alias(f\"mode1_{col}\"))\n\n                if col in additional_param_cols:\n                    expr.append(\n                        (pl.col(col) == mode1).cast(pl.Float64).sum()\n                        .truediv(1 + pl.len())\n                        .alias(f\"mode1_additional_param_{col}\")\n                    )\n            else:\n                expr.append(\n                    pl.mean(col).alias(f\"mode1_{col}\")\n                )\n\n                if col in additional_param_cols:\n                    expr.append(\n                        pl.max(col).alias(f\"mode1_additional_param_{col}\")\n                    )\n\n        return expr\n\n    @staticmethod\n    def num_dpd_expr(df, depth):\n        assert depth in [1, 2]\n        \n        redundant_cols = [\n            'avgdbdtollast24m_4525197P',\n        ]\n\n        cols = [\n            col for col in df.columns\n            if (col[-1] == \"P\") and (col not in redundant_cols)\n        ]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_mean = [pl.mean(col).alias(f\"mean_{col}\") for col in cols]\n\n        expr_above = Aggregator._above_expr(\n            df,\n            depth,\n            [col for col in cols if 'pmts_dpd' in col],\n            threshold=0.0\n        )\n        \n        if depth == 2:\n            return expr_mean + expr_max + expr_above\n\n        expr_applicant_max = Aggregator._applicant_max_expr(df, depth, cols)\n        return expr_mean + expr_max + expr_above + expr_applicant_max\n\n    @staticmethod\n    def num_amnt_expr(df, depth):\n        assert depth in [1, 2]\n        \n        redundant_cols = [\n            'avgpmtlast12m_4525200A',\n            'disbursedcredamount_1113A',\n            'maxpmtlast3m_4525190A',\n            'pmtamount_36A',\n            'totaloutstanddebtvalue_668A',\n            'mainoccupationinc_384A',\n            'credacc_credlmt_575A',\n        ]\n\n        cols = [\n            col for col in df.columns\n            if (col[-1] == \"A\") and (col not in redundant_cols)\n        ]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [\n            pl.min(col).alias(f\"min_{col}\") for col in cols\n            if col not in [\n                'debtoutstand_525A',\n                'debtoverdue_47A',\n                'outstandingamount_362A',\n                'residualamount_488A',\n                'totalamount_996A',\n                'credlmt_935A',\n                'residualamount_856A',\n                'instlamount_768A',\n                \n            ]\n        ]\n\n        expr_mean = [pl.mean(col).alias(f\"mean_{col}\") for col in cols]\n        expr_std = [\n            pl.std(col).alias(f\"std_{col}\")\n            for col in cols if 'amount' in col\n        ]\n\n        if depth == 2:\n            return expr_max + expr_min + expr_mean + expr_std\n\n        expr_applicant_max = Aggregator._applicant_max_expr(df, depth, cols)\n        return expr_max + expr_min + expr_mean + expr_std + expr_applicant_max\n\n    @staticmethod\n    def date_expr(df, depth):\n        assert depth in [1, 2]\n\n        redundant_cols = [\n            'responsedate_1012D',\n            'processingdate_168D',\n            'numberofoverdueinstlmaxdat_148D',\n        ]\n\n        cols = [\n            col for col in df.columns\n            if (col[-1] == \"D\") and (col not in redundant_cols)\n        ]\n\n        expr_max = [\n            pl.max(col).alias(f\"max_{col}\") for col in cols\n            if col not in [\n                'recorddate_4527225D',\n                'approvaldate_319D',\n                'creationdate_885D',\n            ]\n        ]\n        expr_min = [\n            pl.min(col).alias(f\"min_{col}\") for col in cols\n            if col not in [\n                'empl_employedfrom_271D',\n                'recorddate_4527225D',\n            ]\n        ]\n\n        if depth == 1:\n            expr_applicant_max = Aggregator._applicant_max_expr(df, depth, cols)\n            return expr_max + expr_min + expr_applicant_max\n\n        return expr_max + expr_min\n\n    @staticmethod\n    def str_expr(df, depth):\n        assert depth in [1, 2]\n\n        redundant_cols = [\n            'lastrejectcommodtypec_5251769M',\n            'language1_981M',\n            'subjectrole_182M',\n            'subjectrole_93M',\n            'subjectroles_name_838M',\n        ]\n\n        cols = [\n            col for col in df.columns\n            if (col[-1] == \"M\") and (col not in redundant_cols)\n        ]\n\n        expr_mode_share = Aggregator._mode_share_expr(df, depth, cols)\n\n        if depth == 2:\n            return expr_mode_share\n\n        expr_applicant_max = Aggregator._applicant_max_expr(df, depth, cols)\n        return expr_mode_share + expr_applicant_max\n\n    @staticmethod\n    def other_expr(df, depth):\n        assert depth in [1, 2]\n        \n        redundant_cols = [\n            'pmts_month_158T',\n            'pmts_year_1139T',\n            'pmts_month_706T',\n            'dpdmaxdateyear_596T',\n            'overdueamountmaxdatemonth_284T',\n            'overdueamountmaxdatemonth_365T',\n            'dpdmaxdatemonth_89T',\n            'dpdmaxdatemonth_442T',\n\n            'contractssum_5085716L',\n            'applicationscnt_629L',\n            'clientscnt_1130L',\n            'clientscnt_360L',\n            'clientscnt_533L',\n            'clientscnt_887L',\n            'clientscnt_946L',\n            'numinstpaidlastcontr_4325080L',\n            'numnotactivated_1143L',\n            'contaddr_smempladdr_334L',\n            'type_25L',\n\n            # only 1 unique value\n            'personindex_1023L',\n            'persontype_1072L',\n            'persontype_792L',\n\n            # only 2 unique values\n            'contaddr_matchlist_1032L',  # [null, false]\n            'remitter_829L',             # [ 0. null]\n\n            # duplicates\n            'tenor_203L',  # ~ pmtnum_8L\n            \n            # score decrease\n            'periodicityofpmts_837L',\n            'overdueamountmaxdateyear_2T',\n            'numberofoutstandinstls_59L',\n            'annualeffectiverate_63L',\n            'pmtnum_8L',\n            'status_219L',\n        ]\n\n        cols = [\n            col for col in df.columns\n            if (col[-1] in (\"T\", \"L\")) and (col not in redundant_cols)\n        ]\n\n        expr_mode_share = Aggregator._mode_share_expr(df, depth, cols)\n\n        if depth == 2:\n            return expr_mode_share\n\n        expr_applicant_max = Aggregator._applicant_max_expr(df, depth, cols)\n        return expr_mode_share + expr_applicant_max\n    \n    @staticmethod\n    def count_expr(df, depth):\n        assert depth in [1, 2]\n\n        cols = [\n            col for col in df.columns\n            if \"num_group\" in col\n        ]\n\n        if depth == 1:\n            assert \"num_group1\" in df.columns\n            assert \"num_group2\" not in df.columns\n            expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n            return expr_max\n\n        assert \"num_group1\" in df.columns\n        assert \"num_group2\" in df.columns\n\n        return []  # expr_n_unique + expr_applicant_share\n\n    @staticmethod\n    def get_exprs(df, depth):\n        assert depth in [1, 2]\n\n        exprs = Aggregator.num_dpd_expr(df, depth) + \\\n                Aggregator.num_amnt_expr(df, depth) + \\\n                Aggregator.date_expr(df, depth) + \\\n                Aggregator.str_expr(df, depth) + \\\n                Aggregator.other_expr(df, depth) + \\\n                Aggregator.count_expr(df, depth)\n\n        return exprs","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:26:46.147421Z","iopub.execute_input":"2024-05-17T19:26:46.147723Z","iopub.status.idle":"2024-05-17T19:26:46.179939Z","shell.execute_reply.started":"2024-05-17T19:26:46.147697Z","shell.execute_reply":"2024-05-17T19:26:46.178256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### File I/O","metadata":{}},{"cell_type":"code","source":"def read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n\n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df, depth))\n\n    return df\n\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob(str(regex_path)):\n        df = pl.read_parquet(path)\n        df = df.pipe(Pipeline.set_table_dtypes)\n\n        if depth in [1, 2]:\n            df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df, depth))\n        \n        chunks.append(df)\n\n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:26:46.181402Z","iopub.execute_input":"2024-05-17T19:26:46.181762Z","iopub.status.idle":"2024-05-17T19:26:46.199098Z","shell.execute_reply.started":"2024-05-17T19:26:46.181733Z","shell.execute_reply":"2024-05-17T19:26:46.198152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature Engineering","metadata":{}},{"cell_type":"code","source":"def feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base.with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n\n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n\n    df_base = df_base.pipe(Pipeline.handle_dates)\n\n    return df_base","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:26:46.200685Z","iopub.execute_input":"2024-05-17T19:26:46.201095Z","iopub.status.idle":"2024-05-17T19:26:46.218075Z","shell.execute_reply.started":"2024-05-17T19:26:46.201064Z","shell.execute_reply":"2024-05-17T19:26:46.217125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n\n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n\n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n\n    return df_data, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:26:46.219619Z","iopub.execute_input":"2024-05-17T19:26:46.220137Z","iopub.status.idle":"2024-05-17T19:26:46.232524Z","shell.execute_reply.started":"2024-05-17T19:26:46.220107Z","shell.execute_reply":"2024-05-17T19:26:46.230942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Configuration","metadata":{}},{"cell_type":"code","source":"ROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:26:46.233913Z","iopub.execute_input":"2024-05-17T19:26:46.234243Z","iopub.status.idle":"2024-05-17T19:26:46.247820Z","shell.execute_reply.started":"2024-05-17T19:26:46.234212Z","shell.execute_reply":"2024-05-17T19:26:46.246285Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train Files Read & Feature Engineering","metadata":{}},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_2_*.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:26:46.249097Z","iopub.execute_input":"2024-05-17T19:26:46.249444Z","iopub.status.idle":"2024-05-17T19:30:53.985147Z","shell.execute_reply.started":"2024-05-17T19:26:46.249414Z","shell.execute_reply":"2024-05-17T19:30:53.983935Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = feature_eng(**data_store)\nprint(\"train data shape:\\t\", df_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:30:53.986859Z","iopub.execute_input":"2024-05-17T19:30:53.987288Z","iopub.status.idle":"2024-05-17T19:31:06.818480Z","shell.execute_reply.started":"2024-05-17T19:30:53.987242Z","shell.execute_reply":"2024-05-17T19:31:06.816979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del data_store\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:31:06.819821Z","iopub.execute_input":"2024-05-17T19:31:06.820133Z","iopub.status.idle":"2024-05-17T19:31:07.330743Z","shell.execute_reply.started":"2024-05-17T19:31:06.820106Z","shell.execute_reply":"2024-05-17T19:31:07.329732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.pipe(Pipeline.filter_cols)\n\nassert len(df_train.shape) == 2\nprint(\"train data shape:\\t\", df_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:31:07.331890Z","iopub.execute_input":"2024-05-17T19:31:07.332157Z","iopub.status.idle":"2024-05-17T19:31:11.310865Z","shell.execute_reply.started":"2024-05-17T19:31:07.332133Z","shell.execute_reply":"2024-05-17T19:31:11.309543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train, cat_cols = to_pandas(df_train)\ntrain_cols = list(df_train.columns)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:31:11.311937Z","iopub.execute_input":"2024-05-17T19:31:11.312203Z","iopub.status.idle":"2024-05-17T19:31:25.752802Z","shell.execute_reply.started":"2024-05-17T19:31:11.312180Z","shell.execute_reply":"2024-05-17T19:31:25.751508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Preprocessor","metadata":{}},{"cell_type":"code","source":"class Preprocessor:\n\n    def __init__(self):\n        self._cat_col2valid_values = dict()\n\n    def fit(self, X):\n        assert len(X.shape) == 2\n        assert X.shape[0] >= 2\n\n        self._cat_col2valid_values.clear()\n        for col in X.columns:\n            if X[col].dtype.name == 'category':\n                value_counts = X[col].value_counts(normalize=True, sort=True)\n                max_unique_values_n = 20\n                assert max_unique_values_n >= 2\n                valid_values = value_counts.index[:max_unique_values_n]\n                valid_values = set(valid_values.tolist())\n                valid_values = valid_values.union({CAT_NAN_VALUE, CAT_RARE_VALUE})\n                self._cat_col2valid_values[col] = list(valid_values)\n\n        return self\n\n    def transform(self, X):\n        assert len(X.shape) == 2\n        assert X.shape[0] >= 1\n\n        out_df = X.copy()\n        out_df = out_df.set_index('case_id')\n\n        target = None\n        if 'target' in out_df.columns:\n            out_df = out_df.sort_values(by=['WEEK_NUM', 'case_id'])\n            target = out_df['target']\n            out_df = out_df.drop(columns=['target'])\n\n        # ============================================================\n        # Preprocess categorical features\n        for cat_col, valid_values in self._cat_col2valid_values.items():\n            assert out_df[cat_col].dtype.name == 'category'\n            out_df[cat_col] = out_df[cat_col].cat.add_categories([\n                CAT_NAN_VALUE, CAT_RARE_VALUE\n            ])\n            out_df[cat_col] = out_df[cat_col].fillna(CAT_NAN_VALUE)\n            mask = ~out_df[cat_col].isin(valid_values)\n            out_df.loc[mask, cat_col] = CAT_RARE_VALUE\n            out_df[cat_col] = out_df[cat_col].cat.set_categories(valid_values)\n            assert out_df[cat_col].dtype.name == 'category'\n        # ============================================================\n\n        # ============================================================\n        # drop redundant columns\n        out_df.drop(columns=Preprocessor._get_redundant_columns(), inplace=True)\n        # ============================================================\n\n        # ============================================================\n        # aggregate some columns\n        agg2group = Preprocessor._get_agg2group()\n        agg_df = Preprocessor._get_agg_df(out_df, agg2group)\n\n        mode2group = Preprocessor._get_mode2group()\n        mode_df = Preprocessor._get_mode_df(out_df, mode2group)\n\n        out_df.drop(\n            columns=Preprocessor._get_groups_to_drop(agg2group, mode2group),\n            inplace=True\n        )\n        out_df = pd.concat([agg_df, mode_df, out_df], axis=1)\n        # ============================================================\n\n        # ============================================================\n        # construct new features (sometimes, e.g.\n        # feature1 - feature2 > 0 => higher chance of default)\n        mix_df = pd.concat([\n            Preprocessor._get_add_df(out_df, Preprocessor._get_add_pairs()),\n            Preprocessor._get_sub_df(out_df, Preprocessor._get_sub_pairs()),\n            Preprocessor._get_mul_df(out_df, Preprocessor._get_mul_pairs()),\n            Preprocessor._get_div_df(out_df, Preprocessor._get_div_pairs()),\n            # Preprocessor._get_compound_df(out_df, agg_df),\n        ], axis=1)\n\n        out_df = pd.concat([mix_df, out_df], axis=1)\n\n        # ============================================================\n        # select columns\n        out_df = out_df[\n            ['WEEK_NUM']\n            + Preprocessor._get_primary_features()\n            + Preprocessor._get_secondary_features()\n        ]\n        # ============================================================\n\n        # ============================================================\n        # handle NaNs - fillna?\n        # out_df['missing_values_n'] = out_df.isna().sum(axis=1)\n        # ============================================================\n\n        return out_df, target\n    \n    @staticmethod\n    def _get_redundant_columns():\n        return [\n            'deferredmnthsnum_166L',         # only 1 unique value\n            'commnoinclast6m_3546845L',      # only 0 and nan\n            'mastercontrelectronic_519L',    # only 0 and nan\n            'mastercontrexist_109L',         # [ 0. nan]\n            'bankacctype_710L',              # [NaN, 'CA']\n            'isdebitcard_729L',              # [NaN, False]\n            'paytype1st_925L',               # ['OTHER', NaN]\n            'paytype_783L',                  # ['OTHER', NaN]\n            'typesuite_864L',                # [NaN, 'AL'] \n            # 'mode1_subjectrole_182M',        # [NaN, 'a55475b1']\n            # 'mode1_subjectrole_93M',         # [NaN, 'a55475b1']\n            # 'mode1_subjectroles_name_838M',  # [NaN, 'a55475b1']\n\n            # duplicated columns\n            'interestrate_311L',\n            'paytype_783L',\n            'mean_debtoutstand_525A',\n            \n            # other columns contain the same values, but have less nans\n            'assignmentdate_4527235D',\n\n            # score decrease\n            'month_decision',\n            'weekday_decision',\n            'firstclxcampaign_1125D',\n\n            'firstquarter_103L',\n            'secondquarter_766L',\n            # 'thirdquarter_1082L',\n            'fourthquarter_440L',\n            'max_num_group1_5',\n            'max_num_group1_6',\n            'max_num_group1',\n            'max_num_group1_9',\n            'max_num_group1_10',\n            'max_num_group1_11',\n\n            'pmtcount_4527229L',\n            'pmtcount_693L',\n            'pmtscount_423L',\n            'pmtssum_45A',\n            # 'assignmentdate_238D',\n            # 'responsedate_1012D',\n            'applicationscnt_464L',\n            'applicationscnt_629L',\n        ]\n\n    @staticmethod\n    def _get_mode2group():\n        return {\n            'agg_applicant_education_M': [\n                'education_1103M',\n                'applicant_max_education_1138M',\n                'education_88M',\n                'applicant_max_education_927M',\n            ],\n        }\n    \n    @staticmethod\n    def _get_mode_df(df, mode2group):\n        assert len(df.shape) == 2\n        assert type(mode2group) is dict\n        assert len(mode2group) == len(Preprocessor._get_mode2group())\n        \n        mode_data = dict()\n        for mode, group in mode2group.items():\n            mode_data[mode] = df[group].mode(axis=1).iloc[:, 0].astype('category')\n\n        return pd.DataFrame(index=df.index, data=mode_data).astype('category')\n\n    @staticmethod\n    def _get_agg2group():\n        # Return aggregation feature name -> group of features\n        agg2group = {\n            'agg_birth_D': [\n                'birthdate_574D', 'dateofbirth_337D',\n                'applicant_max_birth_259D',\n                'max_birth_259D', 'min_birth_259D',\n            ],\n            'agg_credit_bureau_queries_L': [\n                'days120_123L', 'days180_256L', 'days30_165L',\n                'days360_512L', 'days90_310L',\n            ],\n            'agg_avgdbddpdlast24m_P': [\n                'avgdpdtolclosure24_3658938P',\n                'avgdbddpdlast24m_3658932P',\n                # 'avgdbddpdlast3m_4187120P',\n                'avgdbdtollast24m_4525197P',\n            ],\n            'agg_phone_number_shares_L': [\n                'clientscnt_100L', 'clientscnt_1022L',\n                'clientscnt_1071L', 'clientscnt_1130L',\n                'clientscnt_157L', 'clientscnt_257L',\n                'clientscnt_304L', 'clientscnt_360L',\n                'clientscnt_493L', 'clientscnt_533L',\n                'clientscnt_887L', 'clientscnt_946L',\n                'clientscnt12m_3712952L',\n                'clientscnt3m_3712950L',\n                'clientscnt6m_3712949L',\n            ],\n            'agg_applicant_dtlastpmt_D': [\n                'applicant_max_dtlastpmt_581D',\n                'applicant_max_dtlastpmtallstes_3545839D',\n            ],\n            'agg_min_dtlastpmt_D': [\n                'min_dtlastpmt_581D',\n                'min_dtlastpmtallstes_3545839D',\n            ],\n            'agg_maxdbddpd_P': [  # diffs are interesting\n                'maxdbddpdlast1m_3658939P',\n                'maxdbddpdtollast12m_3658940P',\n                'maxdbddpdtollast6m_4187119P',\n            ],\n            'agg_maxdpdlast_P': [  # diffs are interesting\n                'maxdpdlast12m_727P',\n                'maxdpdlast24m_143P',\n                'maxdpdlast3m_392P',\n                'maxdpdlast6m_474P',\n                'maxdpdlast9m_1059P',\n            ],\n            'agg_mindpdlast24m_P': [\n                'mindbddpdlast24m_3658935P',\n                'mindbdtollast24m_4525191P',\n            ],\n            'agg_numactivecreds_L': [\n                'numactivecreds_622L',\n                'numactivecredschannel_414L',\n            ],\n            # 'agg_pmtcount_L': [\n            #     'pmtcount_4527229L',\n            #     'pmtcount_693L',\n            #     'pmtscount_423L',\n            # ],\n            'agg_pmtaverage_A': [\n                'pmtaverage_3A',\n                'pmtaverage_4527227A',\n            ],\n            'agg_numinstpaid_L': [\n                'numinstlsallpaid_934L',\n                'numinstpaid_4499208L',\n            ],\n            'agg_numinstnotpaid_L': [\n                'numinsttopaygr_769L',\n                'numinsttopaygrest_4493213L',\n            ],\n            'agg_numinstpaidearlymix_L': [\n                'numinstpaidearly3d_3546850L',\n                'numinstpaidearly3dest_4493216L',\n                'numinstmatpaidtearly2d_4499204L',\n                'numinstlallpaidearly3d_817L',\n                'numinstpaidearly5dobd_4499205L',\n                'numinstpaidearly_338L',\n                'numinstpaidearlyest_4493214L',\n            ],\n            'agg_numinstpaidearly5d_L': [\n                'numinstpaidearly5d_1087L',\n                'numinstpaidearly5dest_4493211L',\n            ],\n            'agg_numinstregularpaid_L': [\n                'numinstregularpaid_973L',\n                'numinstregularpaidest_4493210L',\n            ],\n            'agg_numinstunpaidmax_L': [\n                'numinstunpaidmax_3546851L',\n                'numinstunpaidmaxest_4493212L',\n            ],\n            'agg_pctinstlsallpaidlat_L': [\n                'pctinstlsallpaidlat10d_839L',\n                'pctinstlsallpaidlate1d_3546856L',\n                'pctinstlsallpaidlate4d_3546849L',\n                'pctinstlsallpaidlate6d_3546844L',\n            ],\n\n            # correlation is very close to 1 (0.9999...)\n            'agg_dpd_P': [\n                'actualdpdtolerance_344P',\n                'max_actualdpd_943P',\n            ],\n            'agg_currtotaldebt_A': [\n                'currdebt_22A',\n                'totaldebt_9A',\n            ],\n            'agg_activateddate_dateactivated_D': [\n                'lastactivateddate_801D',\n                'max_dateactivated_425D',\n                'applicant_max_dateactivated_425D',\n            ],\n            'agg_apprdate_D': [\n                'lastapprdate_640D',\n                'applicant_max_approvaldate_319D',\n            ],\n            'agg_sumoutstandtotal_A': [\n                'sumoutstandtotal_3546847A',\n                'sumoutstandtotalest_4493215A',\n            ],\n            'agg_firstnonzeroinstldate_D': [\n                'max_firstnonzeroinstldate_307D',\n                'applicant_max_firstnonzeroinstldate_307D',\n            ],\n            'agg_dpdmax_139_pmts_dpd_1073_P': [\n                'max_dpdmax_139P',\n                'max_pmts_dpd_1073P',\n            ],\n            'agg_dpdmax_757_pmts_dpd_303_P': [\n                'max_dpdmax_757P',\n                'max_pmts_dpd_303P',\n            ],\n            'agg_totaldebtoverduevalue_A': [\n                'max_overdueamount_31A',\n                'max_totaldebtoverduevalue_718A',\n                'mean_totaldebtoverduevalue_718A',\n                'min_totaldebtoverduevalue_718A',\n                'applicant_max_totaldebtoverduevalue_718A',\n            ],\n            'agg_numberofoutstandinstls_L': [\n                'mode1_numberofoutstandinstls_520L',\n                'mode1_additional_param_numberofoutstandinstls_520L',\n            ],\n\n            # correlation analysis (correlations 0.95+)\n            'agg_dtlastpmtallstes_D': [\n                'dtlastpmtallstes_4499206D',\n                'max_dtlastpmtallstes_3545839D',\n            ],\n            'agg_approvaldate_firstnonzeroinstldate_D': [\n                'lastapplicationdate_877D',\n                'applicant_max_approvaldate_319D',\n                'applicant_max_dateactivated_425D',\n                'lastapprdate_640D',\n                'applicant_max_firstnonzeroinstldate_307D',\n            ],\n            'agg_min_approvaldate_dateactivated_D': [\n                'min_approvaldate_319D',\n                'min_dateactivated_425D',\n            ],\n            'agg_creationdate_firstnonzeroinstldate_D': [\n                'min_creationdate_885D',\n                'min_firstnonzeroinstldate_307D',\n            ],\n            'agg_max_amount_A': [\n                'max_amount_4527230A',\n                'max_amount_4917619A',\n                'applicant_max_amount_4527230A',\n                'applicant_max_amount_4917619A',\n            ],\n            'agg_min_amount_A': [\n                'min_amount_4527230A',\n                'min_amount_4917619A',\n            ],\n            'agg_mean_amount_A': [\n                'mean_amount_4527230A',\n                'mean_amount_4917619A',\n            ],\n            'agg_std_amount_A': [\n                'std_amount_4527230A',\n                'std_amount_4917619A',\n            ],\n            'agg_mode1_additional_param_employername_M': [\n                'mode1_additional_param_name_4527232M',\n                'mode1_additional_param_name_4917606M',\n                'mode1_additional_param_employername_160M',\n            ],\n            'agg_max_num_group1_3_4': [\n                'max_num_group1_3',\n                'max_num_group1_4',\n            ],\n            'agg_totaldebtoverdue_A': [\n                'max_debtoverdue_47A',\n                'max_overdueamount_659A',\n                'max_totaldebtoverduevalue_178A',\n                'mean_totaldebtoverduevalue_178A',\n            ],\n            'agg_applicant_credlmt_totalamount_A': [\n                'applicant_max_credlmt_230A',\n                'applicant_max_totalamount_6A',\n            ],\n            'agg_overdueamountmaxdateyear_T': [\n                'mode1_dpdmaxdateyear_896T',\n                'mode1_overdueamountmaxdateyear_994T',\n            ],\n            'agg_numberofcontrsvalue_L': [\n                'mode1_additional_param_numberofcontrsvalue_358L',\n                'applicant_max_numberofcontrsvalue_358L',\n                'mode1_numberofcontrsvalue_358L',\n            ],\n            'agg_applicant_overdueamountmaxdateyear_T': [\n                'applicant_max_dpdmaxdateyear_896T',\n                'applicant_max_overdueamountmaxdateyear_994T',\n            ],\n            'agg_mode1_additional_param_contaddr_registaddr_district_M': [\n                'mode1_additional_param_contaddr_district_15M',\n                'mode1_additional_param_registaddr_district_1083M',\n            ],\n            'agg_mode1_additional_param_empladdr_district_zipcode_M': [\n                'mode1_additional_param_empladdr_district_926M',\n                'mode1_additional_param_empladdr_zipcode_114M',\n            ],\n            'agg_max_openingdate_D': [\n                'max_openingdate_313D',\n                'max_openingdate_857D',\n            ],\n            'agg_min_openingdate_D': [\n                'min_openingdate_313D',\n                'min_openingdate_857D',\n            ],\n            'agg_applicant_openingdate_D': [\n                'applicant_max_openingdate_313D',\n                'applicant_max_openingdate_857D',\n            ],\n            'agg_mode1_additional_param_13_545_591_426_M': [\n                'mode1_additional_param_classificationofcontr_13M',\n                'mode1_additional_param_contractst_545M',\n                'mode1_additional_param_financialinstitution_591M',\n                'mode1_additional_param_purposeofcred_426M',\n            ],\n        }\n\n        for prefix in ['max', 'min', 'mean', 'std', 'applicant_max']:\n            agg2group['agg_' + prefix + '_overdueamountmax_14155A'] = [\n                prefix + '_overdueamountmax2_14A',\n                prefix + '_overdueamountmax_155A',\n            ]\n            agg2group['agg_' + prefix + '_overdueamountmax_35398A'] = [\n                prefix + '_overdueamountmax2_398A',\n                prefix + '_overdueamountmax_35A',\n            ]\n            \n        return agg2group\n    \n    @staticmethod\n    def _get_agg_df(df, agg2group):\n        assert len(df.shape) == 2\n        assert type(agg2group) is dict\n        assert len(agg2group) == len(Preprocessor._get_agg2group())\n\n        agg_data = dict()\n        for agg, group in agg2group.items():\n            agg_data[agg] = df[group].mean(axis=1)\n\n        return pd.DataFrame(index=df.index, data=agg_data)\n\n    @staticmethod\n    def _get_add_pairs():\n        return [\n            ['agg_currtotaldebt_A', 'agg_sumoutstandtotal_A', 1.4, 1.0],\n            [\n                'mode1_numberofoverdueinstlmax_1039L',\n                'mode1_numberofoverdueinstlmax_1151L',\n                4.0, 1.0\n            ],\n            # ['mean_dpdmax_757P', 'maxdpdlast24m_143P', 0.35, 1.0],\n            # ['mean_pmts_dpd_303P', 'maxdpdlast24m_143P', 0.38, 1.0],\n            ['mean_pmts_overdue_1140A', 'mean_pmts_overdue_1152A', 3.0, 1.0],\n        ]\n    \n    @staticmethod\n    def _get_add_df(df, add_pairs):\n        assert len(df.shape) == 2\n        assert type(add_pairs) is list\n        assert len(add_pairs) == len(Preprocessor._get_add_pairs())\n        \n        for add_pair in add_pairs:\n            assert len(add_pair) == 4\n\n        add_data = dict()\n        for (f1, f2, w1, w2) in add_pairs:\n            feature_name = '_'.join(['add', f1, f2])\n            add_data[feature_name] = w1 * df[f1] + w2 * df[f2]\n        \n        return pd.DataFrame(index=df.index, data=add_data)\n\n    @staticmethod\n    def _get_sub_pairs():\n        return [\n            ['agg_currtotaldebt_A', 'max_outstandingdebt_522A', 1.27, 1.0],\n            # ['agg_totaldebtoverdue_A', 'mean_totaldebtoverduevalue_178A', 1.0, 1.0],\n            # ['agg_totaldebtoverdue_A', 'max_pmts_overdue_1140A', 2.0, 1.0],\n            ['agg_totaldebtoverdue_A', 'agg_max_overdueamountmax_14155A', 2.0, 1.0],\n            ['mean_totaldebtoverduevalue_178A', 'mean_pmts_overdue_1140A', 0.65, 1.0],\n            ['numincomingpmts_3546848L', 'numinstlswithdpd10_728L', 0.11, 1.0],\n            \n            # ['min_annuity_853A', 'min_totaldebtoverduevalue_178A', 0.75, 1.0],\n            ['numinstlswithdpd5_4187116L', 'numinstlswithoutdpd_562L', 14.1, 1.0],\n            # ['datelastinstal40dpd_247D', 'max_dateofcredstart_181D', 0.42, 1.0],\n            ['agg_min_overdueamountmax_35398A', 'min_annuity_853A', 1.3, 1.0],\n            # ['agg_totaldebtoverdue_A', 'mean_annuity_853A', 1.65, 1.0],\n            ['annuity_780A', 'mean_annuity_853A', 0.83, 1.0],\n            ['annuity_780A', 'applicant_max_annuity_853A', 0.9, 1.0],\n            ['agg_avgdbddpdlast24m_P', 'avgdbddpdlast3m_4187120P', 0.65, 1.0],\n            ['maxdbddpdlast1m_3658939P', 'maxdbddpdtollast12m_3658940P', 1.17, 1.0],\n            ['mean_overdueamount_659A', 'mean_overdueamount_31A', 0.02, 1.0],\n            # ['maxdpdlast12m_727P', 'maxdpdlast24m_143P', 1.6, 1.0],\n            # ['maxdpdlast9m_1059P', 'maxdpdlast12m_727P', 1.24, 1.0],\n            # ['maxdpdlast6m_474P', 'maxdpdlast9m_1059P', 1.33, 1.0],\n            ['maxdpdlast3m_392P', 'maxdpdlast6m_474P', 1.5, 1.0],\n            # ['maxdpdlast3m_392P', 'maxdpdlast24m_143P', 4.0, 1.0],\n            # ['currdebt_22A', 'currdebt_94A', 1.0, 1.0],\n            ['maininc_215A', 'mean_mainoccupationinc_437A', 0.83, 1.0],\n            # ['mean_dpdmax_139P', 'mean_dpdmax_757P', 4.5, 1.0],\n            ['mean_monthlyinstlamount_332A', 'mean_monthlyinstlamount_674A', 1.2, 1.0],\n            ['mean_totalamount_996A', 'mean_totalamount_6A', 0.35, 1.0],\n        ]\n\n    @staticmethod\n    def _get_sub_df(df, sub_pairs):\n        assert len(df.shape) == 2\n        assert type(sub_pairs) is list\n        assert len(sub_pairs) == len(Preprocessor._get_sub_pairs())\n        \n        for sub_pair in sub_pairs:\n            assert len(sub_pair) == 4\n\n        sub_data = dict()\n        for (f1, f2, w1, w2) in sub_pairs:\n            feature_name = '_'.join(['sub', f1, f2])\n            sub_data[feature_name] = w1 * df[f1] - w2 * df[f2]\n\n        return pd.DataFrame(index=df.index, data=sub_data)\n    \n    @staticmethod\n    def _get_mul_pairs():\n        return [\n            # ['maxdpdlast12m_727P', 'agg_sumoutstandtotal_A'],\n            # ['agg_sumoutstandtotal_A', 'applicant_max_dateofcredend_289D'],\n            # ['mean_residualamount_488A', 'mean_pmts_overdue_1152A'],\n        ]\n    \n    @staticmethod\n    def _get_mul_df(df, mul_pairs):\n        assert len(df.shape) == 2\n        assert type(mul_pairs) is list\n        assert len(mul_pairs) == len(Preprocessor._get_mul_pairs())\n        \n        for mul_pair in mul_pairs:\n            assert len(mul_pair) == 2\n\n        mul_data = dict()\n        for (f1, f2) in mul_pairs:\n            feature_name = '_'.join(['mul', f1, f2])\n            mul_data[feature_name] = df[f1] * df[f2]\n\n        return pd.DataFrame(index=df.index, data=mul_data)\n\n    @staticmethod\n    def _get_div_pairs():\n        return [\n            ['annuity_780A', 'maininc_215A'],\n            ['agg_currtotaldebt_A', 'maininc_215A'],\n            ['mean_debtoverdue_47A', 'maininc_215A'],\n            ['mean_totaloutstanddebtvalue_39A', 'maininc_215A'],\n            ['avglnamtstart24m_4525187A', 'maininc_215A'],\n            ['avgpmtlast12m_4525200A', 'maininc_215A'],\n            ['agg_sumoutstandtotal_A', 'maininc_215A'],\n            ['maxdpdlast12m_727P', 'maininc_215A'],\n            ['maxpmtlast3m_4525190A', 'maininc_215A'],\n            ['inittransactionamount_650A', 'price_1097A'],\n            ['agg_sumoutstandtotal_A', 'price_1097A'],\n            ['agg_currtotaldebt_A', 'price_1097A'],\n            # ['maxpmtlast3m_4525190A', 'price_1097A'],\n            ['mean_credacc_actualbalance_314A', 'agg_numactivecreds_L'],\n            ['mean_credacc_actualbalance_314A', 'avgpmtlast12m_4525200A'],\n            ['agg_currtotaldebt_A', 'applicant_max_dateofcredend_289D'],\n            # ['mean_debtoverdue_47A', 'applicant_max_dateofcredend_289D'],\n            # ['agg_sumoutstandtotal_A', 'applicant_max_dateofcredend_289D'],\n            ['maininc_215A', 'applicant_max_childnum_21L'],\n            ['mean_monthlyinstlamount_332A', 'inittransactionamount_650A'],\n            ['mean_monthlyinstlamount_332A', 'maininc_215A'],\n            ['mean_totaldebtoverduevalue_178A', 'maininc_215A'],\n        ]\n    \n    @staticmethod\n    def _get_div_df(df, div_pairs):\n        assert len(df.shape) == 2\n        assert type(div_pairs) is list\n        assert len(div_pairs) == len(Preprocessor._get_div_pairs())\n        \n        for div_pair in div_pairs:\n            assert len(div_pair) == 2\n\n        div_data = dict()\n        for (f1, f2) in div_pairs:\n            feature_name = '_'.join(['div', f1, f2])\n            div_data[feature_name] = df[f1] / (1.0 + df[f2])\n\n        return pd.DataFrame(index=df.index, data=div_data)\n    \n#     @staticmethod\n#     def _get_compound_df(df):\n#         assert len(df.shape) == 2\n\n#         compound_data = {\n#             'compound_coef_1_DA': (\n#                 df['mean_monthlyinstlamount_332A']\n#                 / (1.0 + df['maininc_215A'])\n#                 * df['applicant_max_dateofcredend_289D']\n#             ),\n#             'compound_coef_2_DA': (\n#                 df['maininc_215A']\n#                 * df['applicant_max_dateofcredend_289D']\n#                 / (1.0 + df['price_1097A'])\n#             ),\n#             'compound_coef_3_DA': (\n#                 df['maininc_215A']\n#                 * df['applicant_max_dateofcredend_289D']\n#                 - df['agg_sumoutstandtotal_A']\n#             ),\n#             'compound_coef_4_DA': (\n#                 df['maininc_215A'] * df['applicant_max_dateofcredend_289D']\n#                 + df['mean_credacc_actualbalance_314A']\n#                 - df['agg_sumoutstandtotal_A']\n#             ) / (1.0 + df['agg_numactivecreds_L']),\n#         }\n\n#         return pd.DataFrame(index=df.index, data=compound_data)\n\n    @staticmethod\n    def _get_groups_to_drop(agg2group, mode2group):\n        assert type(agg2group) is dict\n        assert type(mode2group) is dict\n        assert len(agg2group) == len(Preprocessor._get_agg2group())\n        assert len(mode2group) == len(Preprocessor._get_mode2group())\n\n        features_to_drop = []\n        features_to_keep = [\n            'mean_totaldebtoverduevalue_178A',\n            'maxdbddpdlast1m_3658939P',\n            'maxdbddpdtollast12m_3658940P',\n            'maxdpdlast12m_727P',\n            'maxdpdlast24m_143P',\n            'maxdpdlast9m_1059P',\n            'maxdpdlast6m_474P',\n            'maxdpdlast3m_392P',\n        ]\n\n        for _, group in agg2group.items():\n            for feature in group:\n                if feature not in features_to_keep:\n                    features_to_drop.append(feature)\n\n        for _, group in mode2group.items():\n            for feature in group:\n                if feature not in features_to_keep:\n                    features_to_drop.append(feature)\n\n        return features_to_drop\n\n    @staticmethod\n    def _get_primary_features():\n        return [\n            'annuity_780A',\n            'pmtnum_254L',\n            'eir_270L',\n            'mobilephncnt_593L',\n            'price_1097A',\n\n            'agg_birth_D',\n            'requesttype_4525192L',\n            'amtinstpaidbefduel24m_4187115A',\n            'annuitynextmonth_57A',\n            'agg_avgdbddpdlast24m_P',\n\n            'mean_pmts_dpd_1073P',\n            'mean_pmts_overdue_1140A',\n            'applicant_max_dpdmax_139P',\n            'above_0.0_share_pmts_dpd_1073P',\n            'lastrejectdate_50D',\n\n            'agg_applicant_education_M',\n            'mean_dpdmax_139P',\n            'lastcancelreason_561M',\n\n            'mode1_relationshiptoclient_415T',\n            'min_dateofcredstart_739D',\n            'agg_max_amount_A',\n            'datefirstoffer_1144D',\n            'agg_currtotaldebt_A',\n\n            'mode1_incometype_1044T',\n            'mode1_numberofoverdueinstlmax_1151L',\n            'mode1_additional_param_numberofoverdueinstlmax_1151L',\n            'agg_mean_overdueamountmax_35398A',\n            'above_0.0_share_pmts_dpd_303P',\n    \n            'monthsannuity_845L',\n            'lastdelinqdate_224D',\n            'isbidproduct_1095L',\n            'maxdpdtolerance_374P',\n            'mean_residualamount_856A',\n            'std_overdueamount_659A',\n            'agg_maxdbddpd_P',\n            'mode1_numberofoverdueinstlmax_1039L',\n            'agg_numberofcontrsvalue_L',\n            'mean_maxdpdtolerance_577P',\n            'mean_totalamount_6A',\n            'agg_numinstunpaidmax_L',\n            'applicationscnt_867L',\n\n            'sub_numincomingpmts_3546848L_numinstlswithdpd10_728L',\n            'sub_numinstlswithdpd5_4187116L_numinstlswithoutdpd_562L',\n            'div_inittransactionamount_650A_price_1097A',\n            'min_credacc_actualbalance_314A',\n            'add_mode1_numberofoverdueinstlmax_1039L_mode1_numberofoverdueinstlmax_1151L',\n            'sub_agg_currtotaldebt_A_max_outstandingdebt_522A',\n            'sub_maxdbddpdlast1m_3658939P_maxdbddpdtollast12m_3658940P',\n            'agg_totaldebtoverduevalue_A',\n            'min_mainoccupationinc_437A',\n        ]\n    \n    @staticmethod\n    def _get_secondary_features():\n        return [\n            'agg_credit_bureau_queries_L',\n            'maritalst_385M',\n            'responsedate_4527233D',\n            'max_employedfrom_700D',\n            'mode1_education_1138M',\n            'mode1_additional_param_postype_4733339M',\n            'mode1_additional_param_rejectreasonclient_4145042M',\n            'mode1_isdebitcard_527L',\n            'agg_max_num_group1_3_4',\n            'max_instlamount_852A',\n            'max_monthlyinstlamount_332A',\n            'min_totaloutstanddebtvalue_39A',\n            'mean_totalamount_996A',\n            'agg_std_overdueamountmax_35398A',\n            'applicant_max_purposeofcred_426M',\n            'mode1_collater_valueofguarantee_1124L',\n\n            'agg_mean_amount_A',\n            'agg_overdueamountmaxdateyear_T',\n            'agg_applicant_overdueamountmaxdateyear_T',\n            'agg_mode1_additional_param_13_545_591_426_M',\n            'cntincpaycont9m_3716944L',\n            'lastrejectreason_759M',\n            'posfpd30lastmonth_3976960P',\n            'max_credamount_590A',\n            'mean_credamount_590A',\n            'max_overdueamountmax2date_1142D',\n            'applicant_max_dateofcredend_289D',\n            'applicant_max_overdueamountmax2date_1002D',\n            'mode1_additional_param_purposeofcred_874M',\n            'mode1_sex_738L',\n            'max_amount_416A',\n            'mean_amount_416A',\n            'max_pmts_overdue_1152A',\n            \n            'equalitydataagreement_891L',\n            'lastapprcredamount_781A',\n            'lastrejectcredamount_222A',\n            'maxdpdfrom6mto36m_3546853P',\n            'mode1_financialinstitution_382M',  # check\n\n            # 'sub_agg_totaldebtoverdue_A_max_overdueamount_659A',\n            'sub_agg_totaldebtoverdue_A_agg_max_overdueamountmax_14155A',\n            'sub_mean_totalamount_996A_mean_totalamount_6A',\n            'div_mean_totaloutstanddebtvalue_39A_maininc_215A',\n            'div_maxdpdlast12m_727P_maininc_215A',\n            'div_agg_currtotaldebt_A_price_1097A',\n            'div_mean_monthlyinstlamount_332A_maininc_215A',\n            'agg_numinstpaid_L',\n            'agg_dtlastpmtallstes_D',\n            'agg_creationdate_firstnonzeroinstldate_D',\n            'agg_max_overdueamountmax_35398A',\n            'thirdquarter_1082L',\n            'avgdbddpdlast3m_4187120P',\n            'credamount_770A',\n            'credtype_322L',\n            'daysoverduetolerancedd_3976961L',\n            'inittransactioncode_186L',\n            'lastrejectcommoditycat_161M',\n            'maxannuity_159A',\n            'numinstls_657L',\n            'numinstpaidlate1d_3546852L',\n            'sellerplacescnt_216L',\n            'max_credacc_maxhisbal_375A',\n            'max_credacc_minhisbal_90A',\n            'min_credacc_maxhisbal_375A',\n            'min_credacc_minhisbal_90A',\n            'mean_credacc_minhisbal_90A',\n            'mean_currdebt_94A',\n            'mean_mainoccupationinc_437A',\n            'mean_revolvingaccount_394A',\n            'applicant_max_credamount_590A',\n            'applicant_max_currdebt_94A',\n            'applicant_max_downpmt_134A',\n            'mode1_credacc_status_367L',\n            'mode1_credtype_587L',\n            'mode1_familystate_726L',\n            'max_credlmt_935A',\n            'max_outstandingamount_362A',\n            'min_totalamount_6A',\n            'mean_overdueamount_659A',\n            'std_instlamount_768A',\n            'std_monthlyinstlamount_674A',\n            'applicant_max_totalamount_996A',\n            'max_dateofcredend_289D',\n            'max_dateofcredstart_739D',\n            'min_numberofoverdueinstlmaxdat_641D',\n            'mode1_nominalrate_281L',\n            'mode1_nominalrate_498L',\n            'mode1_numberofinstls_229L',\n            'mode1_periodicityofpmts_1102L',\n            'applicant_max_numberofoverdueinstlmax_1039L',\n            'mode1_collater_valueofguarantee_876L',\n            \n            'mean_outstandingdebt_522A',\n            'agg_mean_overdueamountmax_14155A',\n            'agg_std_amount_A',\n            'agg_firstnonzeroinstldate_D',\n            'applicant_max_familystate_726L',\n            'applicant_max_dateofcredstart_739D',\n            'min_overdueamountmax2date_1002D',\n            'maxdbddpdtollast12m_3658940P',\n            'maxdbddpdlast1m_3658939P',\n            'agg_maxdpdlast_P',\n            'div_annuity_780A_maininc_215A',\n            'div_agg_sumoutstandtotal_A_price_1097A',\n            'sub_mean_totaldebtoverduevalue_178A_mean_pmts_overdue_1140A',\n\n            'add_agg_currtotaldebt_A_agg_sumoutstandtotal_A',\n            'add_mean_pmts_overdue_1140A_mean_pmts_overdue_1152A',\n            'sub_annuity_780A_applicant_max_annuity_853A',\n            'sub_agg_avgdbddpdlast24m_P_avgdbddpdlast3m_4187120P',\n            'sub_mean_monthlyinstlamount_332A_mean_monthlyinstlamount_674A',\n            'div_avglnamtstart24m_4525187A_maininc_215A',\n            'div_maxpmtlast3m_4525190A_maininc_215A',\n            'agg_min_dtlastpmt_D',\n            'agg_mindpdlast24m_P',\n            'agg_numinstregularpaid_L',\n            'agg_pctinstlsallpaidlat_L',\n            'agg_activateddate_dateactivated_D',\n            'agg_sumoutstandtotal_A',\n            'agg_dpdmax_139_pmts_dpd_1073_P',\n            'agg_dpdmax_757_pmts_dpd_303_P',\n            'agg_approvaldate_firstnonzeroinstldate_D',\n            'agg_min_amount_A',\n            'agg_applicant_credlmt_totalamount_A',\n            'agg_mode1_additional_param_empladdr_district_zipcode_M',\n            'agg_min_openingdate_D',\n            'agg_max_overdueamountmax_14155A',\n            'agg_min_overdueamountmax_14155A',\n            'agg_min_overdueamountmax_35398A',\n            'agg_applicant_max_overdueamountmax_14155A',\n            'avglnamtstart24m_4525187A',\n            'avgmaxdpdlast9m_3716943P',\n            'avgpmtlast12m_4525200A',\n            'cntpmts24_3658933L',\n            'datelastinstal40dpd_247D',\n            'disbursedcredamount_1113A',\n            'homephncnt_628L',\n            'inittransactionamount_650A',\n            'lastapprcommoditycat_1041M',\n            'lastrejectreasonclient_4145040M',\n            'lastst_736L',\n            'maxdebt4_972A',\n            'maxdpdlast24m_143P',\n            'maxdpdlast9m_1059P',\n            'maxinstallast24m_3658928A',\n            'maxlnamtstart6m_4525199A',\n            'numcontrs3months_479L',\n            'numincomingpmts_3546848L',\n            'numnotactivated_1143L',\n            'numrejects9m_859L',\n            'pctinstlsallpaidearl3d_427L',\n            'totalsettled_863A',\n            'totinstallast1m_4525188A',\n            'max_maxdpdtolerance_577P',\n            'max_annuity_853A',\n            'max_mainoccupationinc_437A',\n            'max_outstandingdebt_522A',\n            'mean_annuity_853A',\n            'std_credamount_590A',\n            'applicant_max_annuity_853A',\n            'min_employedfrom_700D',\n            'applicant_max_employedfrom_700D',\n            'mode1_childnum_21L',\n            'applicant_max_credtype_587L',\n            'applicant_max_inittransactioncode_279L',\n            'mean_dpdmax_757P',\n            'applicant_max_dpdmax_757P',\n            'max_credlmt_230A',\n            'max_instlamount_768A',\n            'max_monthlyinstlamount_674A',\n            'max_residualamount_488A',\n            'max_residualamount_856A',\n            'max_totalamount_6A',\n            'max_totalamount_996A',\n            'min_credlmt_230A',\n            'min_instlamount_852A',\n            'min_monthlyinstlamount_332A',\n            'min_monthlyinstlamount_674A',\n            'min_totaldebtoverduevalue_178A',\n            'mean_credlmt_230A',\n            'mean_outstandingamount_362A',\n            'mean_residualamount_488A',\n            'mean_totaldebtoverduevalue_178A',\n            'std_instlamount_852A',\n            'std_monthlyinstlamount_332A',\n            'std_residualamount_488A',\n            'std_residualamount_856A',\n            'std_totalamount_6A',\n            'applicant_max_monthlyinstlamount_332A',\n            'max_dateofcredstart_181D',\n            'max_dateofrealrepmt_138D',\n            'max_lastupdate_388D',\n            'max_numberofoverdueinstlmaxdat_641D',\n            'max_overdueamountmax2date_1002D',\n            'min_dateofcredend_289D',\n            'min_dateofcredstart_181D',\n            'min_dateofrealrepmt_138D',\n            'applicant_max_dateofcredend_353D',\n            'applicant_max_dateofrealrepmt_138D',\n            'applicant_max_lastupdate_388D',\n            'applicant_max_refreshdate_3813885D',\n            'mode1_numberofinstls_320L',\n            'mode1_prolongationcount_1120L',\n            'applicant_max_numberofinstls_320L',\n            'applicant_max_numberofoverdueinstlmax_1151L',\n            'mean_pmts_overdue_1152A',\n            'maxdpdinstldate_3546855D',\n            'maxdpdlast3m_392P',\n            'mode1_numberofoverdueinstls_725L',\n            'agg_numinstnotpaid_L',\n\n            'mode1_rejectreason_755M',\n            'mode1_financialinstitution_591M',\n            'mode1_familystate_447L',\n            'mode1_education_927M',\n            'datelastunpaid_3546854D',\n            'mode1_cancelreason_3545846M',\n        ]","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:31:25.754482Z","iopub.execute_input":"2024-05-17T19:31:25.754820Z","iopub.status.idle":"2024-05-17T19:31:25.818063Z","shell.execute_reply.started":"2024-05-17T19:31:25.754792Z","shell.execute_reply":"2024-05-17T19:31:25.816536Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Construct the training dataset","metadata":{}},{"cell_type":"code","source":"%%time\n\npreprocessor = Preprocessor()\npreprocessor.fit(df_train)\n\nX, y = preprocessor.transform(df_train)\n\nweek_num = X['WEEK_NUM']\nX = X.drop(columns=['WEEK_NUM'])","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:31:25.819889Z","iopub.execute_input":"2024-05-17T19:31:25.820378Z","iopub.status.idle":"2024-05-17T19:39:07.294771Z","shell.execute_reply.started":"2024-05-17T19:31:25.820348Z","shell.execute_reply":"2024-05-17T19:39:07.293372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert df_train.shape[0] == X.shape[0]\nassert week_num.shape == y.shape\nassert y.shape == (X.shape[0],)\nassert (X.index == y.index).all()\nassert (X.index == week_num.index).all()\n\nfor i in range(1, X.shape[0]):\n    assert week_num.iloc[i - 1] <= week_num.iloc[i]\n\nfor col in X.columns:\n    assert X[col].isna().sum() < X.shape[0]\n    assert X[col].unique().shape[0] >= 2\n    assert X[col].dtype.name != 'object'","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:39:07.296156Z","iopub.execute_input":"2024-05-17T19:39:07.296453Z","iopub.status.idle":"2024-05-17T19:39:32.617929Z","shell.execute_reply.started":"2024-05-17T19:39:07.296430Z","shell.execute_reply":"2024-05-17T19:39:32.616174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape, X.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:39:32.619736Z","iopub.execute_input":"2024-05-17T19:39:32.620236Z","iopub.status.idle":"2024-05-17T19:39:32.627876Z","shell.execute_reply.started":"2024-05-17T19:39:32.620193Z","shell.execute_reply":"2024-05-17T19:39:32.626393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:39:32.629398Z","iopub.execute_input":"2024-05-17T19:39:32.629829Z","iopub.status.idle":"2024-05-17T19:39:32.749721Z","shell.execute_reply.started":"2024-05-17T19:39:32.629795Z","shell.execute_reply":"2024-05-17T19:39:32.748527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv = StratifiedGroupKFold(n_splits=7, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:39:32.751331Z","iopub.execute_input":"2024-05-17T19:39:32.751705Z","iopub.status.idle":"2024-05-17T19:39:32.760105Z","shell.execute_reply.started":"2024-05-17T19:39:32.751677Z","shell.execute_reply":"2024-05-17T19:39:32.759265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fold_idx = 6\n\nfor i, (idx_train, idx_valid) in enumerate(cv.split(X, y, groups=week_num)):\n    if i != fold_idx:\n        continue\n\n    assert i == fold_idx\n    X_train, y_train = X.iloc[idx_train], y.iloc[idx_train]\n    X_valid, y_valid = X.iloc[idx_valid], y.iloc[idx_valid]\n\n    model = lgb.LGBMClassifier(\n        objective=custom_loss.compute_grad_hess,\n        boosting_type='gbdt',\n        bagging_freq=1,\n        pos_bagging_fraction=0.7,\n        neg_bagging_fraction=0.7,\n        colsample_bynode=0.9,\n        colsample_bytree=0.9,\n        device='cpu',\n        learning_rate=0.04,\n        max_depth=10,\n        metric='auc',\n        n_estimators=800,\n        num_leaves=33\n        # l1_regularization=0.01,\n        # l2_regularization=1.0\n    )\n\n    model.fit(\n        X_train, y_train,\n        eval_set=[\n            (X_train, y_train),\n            (X_valid, y_valid),\n        ],\n        eval_metric=[\n            'auc',\n        ],\n        callbacks=[\n            lgb.log_evaluation(100),\n            lgb.early_stopping(100),\n        ]\n    )\n\n    y_pred = model.predict(X_valid)\n    assert np.allclose(\n        model.booster_.predict(X_valid),\n        y_pred\n    )\n    scaled_y_pred = 1.0 / (1.0 + np.exp(-1.0 * y_pred))\n    assert (scaled_y_pred >= 0.0).all()\n    assert (scaled_y_pred <= 1.0).all()\n    print('mean(scaled_y_pred)', np.mean(scaled_y_pred))\n    print('std(scaled_y_pred)', np.std(scaled_y_pred))\n    valid_df = pd.DataFrame(data={\n        \"WEEK_NUM\": week_num.loc[y_valid.index].to_numpy(),\n        \"target\": y_valid.to_numpy(),\n        \"score\": model.booster_.predict(X_valid)\n    })\n    score = compute_gini_stability_metric(valid_df, debug=True)\n    print('ROC AUC:', roc_auc_score(y_valid, y_pred))\n    print('stability score:', score)\n\n    model.booster_.save_model(\n        f'/kaggle/working/custom_loss_lgb_model_{i}.txt'\n        # num_iteration=model.estimators[i].best_iteration\n    )","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:39:32.765061Z","iopub.execute_input":"2024-05-17T19:39:32.765349Z","iopub.status.idle":"2024-05-17T19:43:13.916933Z","shell.execute_reply.started":"2024-05-17T19:39:32.765327Z","shell.execute_reply":"2024-05-17T19:43:13.915854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.kdeplot(y_pred)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(\n    model.feature_importances_,\n    index=X.columns\n).sort_values().plot(kind='barh', figsize=(16, 90))","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:43:13.918024Z","iopub.execute_input":"2024-05-17T19:43:13.918307Z","iopub.status.idle":"2024-05-17T19:43:16.038989Z","shell.execute_reply.started":"2024-05-17T19:43:13.918283Z","shell.execute_reply":"2024-05-17T19:43:16.037896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 4))\nplt.plot(model.evals_result_['valid_1']['auc'])","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:43:16.040203Z","iopub.execute_input":"2024-05-17T19:43:16.040514Z","iopub.status.idle":"2024-05-17T19:43:16.248352Z","shell.execute_reply.started":"2024-05-17T19:43:16.040487Z","shell.execute_reply":"2024-05-17T19:43:16.247291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trees_n = 3\nfig, ax = plt.subplots(nrows=trees_n, figsize=(16, 6 * trees_n), sharex=True)\nfor i in range(trees_n):\n    lgb.plot_tree(model, tree_index=i, dpi=500, ax=ax[i])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del X\n# del y\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:43:16.256671Z","iopub.execute_input":"2024-05-17T19:43:16.256996Z","iopub.status.idle":"2024-05-17T19:43:16.270166Z","shell.execute_reply.started":"2024-05-17T19:43:16.256969Z","shell.execute_reply":"2024-05-17T19:43:16.268752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Test Files Read & Feature Engineering","metadata":{}},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_files(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n        read_files(TEST_DIR / \"test_credit_bureau_a_2_*.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:43:16.271281Z","iopub.execute_input":"2024-05-17T19:43:16.271580Z","iopub.status.idle":"2024-05-17T19:43:16.742187Z","shell.execute_reply.started":"2024-05-17T19:43:16.271553Z","shell.execute_reply":"2024-05-17T19:43:16.740750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = feature_eng(**data_store)\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:43:16.743607Z","iopub.execute_input":"2024-05-17T19:43:16.744058Z","iopub.status.idle":"2024-05-17T19:43:16.826951Z","shell.execute_reply.started":"2024-05-17T19:43:16.744019Z","shell.execute_reply":"2024-05-17T19:43:16.825112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del data_store\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:43:16.828412Z","iopub.execute_input":"2024-05-17T19:43:16.828759Z","iopub.status.idle":"2024-05-17T19:43:16.969858Z","shell.execute_reply.started":"2024-05-17T19:43:16.828733Z","shell.execute_reply":"2024-05-17T19:43:16.968713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature Elimination","metadata":{}},{"cell_type":"code","source":"df_test = df_test.select([col for col in train_cols if col != \"target\"])\nassert len(df_test.shape) == 2\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:43:16.971159Z","iopub.execute_input":"2024-05-17T19:43:16.971424Z","iopub.status.idle":"2024-05-17T19:43:16.985813Z","shell.execute_reply.started":"2024-05-17T19:43:16.971403Z","shell.execute_reply":"2024-05-17T19:43:16.984436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Pandas Conversion","metadata":{}},{"cell_type":"code","source":"df_test, _ = to_pandas(df_test, cat_cols)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:43:16.986911Z","iopub.execute_input":"2024-05-17T19:43:16.987296Z","iopub.status.idle":"2024-05-17T19:43:17.047567Z","shell.execute_reply.started":"2024-05-17T19:43:16.987261Z","shell.execute_reply":"2024-05-17T19:43:17.046686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test, _ = preprocessor.transform(df_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:43:17.049152Z","iopub.execute_input":"2024-05-17T19:43:17.049487Z","iopub.status.idle":"2024-05-17T19:43:17.265544Z","shell.execute_reply.started":"2024-05-17T19:43:17.049456Z","shell.execute_reply":"2024-05-17T19:43:17.264182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = X_test.drop(columns=['WEEK_NUM'])","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:43:17.267050Z","iopub.execute_input":"2024-05-17T19:43:17.267408Z","iopub.status.idle":"2024-05-17T19:43:17.273551Z","shell.execute_reply.started":"2024-05-17T19:43:17.267382Z","shell.execute_reply":"2024-05-17T19:43:17.272685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_pred = model.predict(X_test, prediction_type='RawFormulaVal')\ny_pred = model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:43:17.274565Z","iopub.execute_input":"2024-05-17T19:43:17.274909Z","iopub.status.idle":"2024-05-17T19:43:17.303027Z","shell.execute_reply.started":"2024-05-17T19:43:17.274886Z","shell.execute_reply":"2024-05-17T19:43:17.301911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.min(y_pred), np.max(y_pred)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:43:17.304168Z","iopub.execute_input":"2024-05-17T19:43:17.304437Z","iopub.status.idle":"2024-05-17T19:43:17.314504Z","shell.execute_reply.started":"2024-05-17T19:43:17.304415Z","shell.execute_reply":"2024-05-17T19:43:17.313397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.kdeplot(y_pred)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:43:17.315979Z","iopub.execute_input":"2024-05-17T19:43:17.316324Z","iopub.status.idle":"2024-05-17T19:43:17.603582Z","shell.execute_reply.started":"2024-05-17T19:43:17.316298Z","shell.execute_reply":"2024-05-17T19:43:17.602566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:43:17.604854Z","iopub.execute_input":"2024-05-17T19:43:17.605157Z","iopub.status.idle":"2024-05-17T19:43:17.745559Z","shell.execute_reply.started":"2024-05-17T19:43:17.605132Z","shell.execute_reply":"2024-05-17T19:43:17.744091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submission","metadata":{}},{"cell_type":"code","source":"df_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\n\ndf_subm[\"score\"] = y_pred","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:43:17.746911Z","iopub.execute_input":"2024-05-17T19:43:17.747220Z","iopub.status.idle":"2024-05-17T19:43:17.769512Z","shell.execute_reply.started":"2024-05-17T19:43:17.747195Z","shell.execute_reply":"2024-05-17T19:43:17.768321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:43:17.770502Z","iopub.execute_input":"2024-05-17T19:43:17.770767Z","iopub.status.idle":"2024-05-17T19:43:17.786728Z","shell.execute_reply.started":"2024-05-17T19:43:17.770745Z","shell.execute_reply":"2024-05-17T19:43:17.785203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-05-17T19:43:17.788123Z","iopub.execute_input":"2024-05-17T19:43:17.788507Z","iopub.status.idle":"2024-05-17T19:43:17.799364Z","shell.execute_reply.started":"2024-05-17T19:43:17.788476Z","shell.execute_reply":"2024-05-17T19:43:17.798285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}