{"metadata":{"accelerator":"GPU","colab":{"gpuType":"T4","machine_shape":"hm","provenance":[]},"kernelspec":{"display_name":"Python 3","name":"python3"},"language_info":{"name":"python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"},{"sourceId":7614092,"sourceType":"datasetVersion","datasetId":4256688},{"sourceId":7626121,"sourceType":"datasetVersion","datasetId":4422102}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#### <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:black; font-size:180%; text-align:left;padding:3.0px; background: #ccf5ff; border-bottom: 8px solid #0047b3\" > TABLE OF CONTENTS<br><div>  \n* [IMPORTS AND INSTALLATIONS](#1)\n* [INTRODUCTION](#2)\n    * [UTILITIES](#2.1)\n    * [DATASET DETAILS](#2.2)    \n    * [CONFIGURATION](#2.3)\n    * [VERSION DETAILS](#2.4)\n* [PREPROCESSING](#3)\n* [MODEL TRAINING](#4)         ","metadata":{"id":"nUVs0Nmxd0oR"}},{"cell_type":"markdown","source":"<a id=\"1\"></a>\n# <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:black; font-size:120%; text-align:left;padding:3.0px; background: #b3e0ff; border-bottom: 8px solid #1a0d00\" > PACKAGE IMPORTS AND INSTALLATIONS<br> <div>","metadata":{"id":"FNfLSqY-d79o"}},{"cell_type":"code","source":"%%time\n\n##################################################################\nfrom IPython.display import clear_output;\nfrom gc import collect;\nfrom google.colab import drive;\nprint();\ndrive.mount('/content/drive');\n\nfrom google.colab import files;\nfiles.upload();\n!ls -lha kaggle.json;\n!pip install -q kaggle --upgrade;\n!mkdir -p ~/.kaggle;\n!cp kaggle.json ~/.kaggle/\n!pwd;\n!chmod 600 ~/.kaggle/kaggle.json;\nprint();\n\n##################################################################\n!kaggle kernels output andreynesterov/lightautoml-038-dependencies -p /HomeCredit;\n!pip install -q /HomeCredit/lightautoml-0.3.8-py3-none-any.whl;\n\n!pip install -q --upgrade polars==0.20.5;\n!pip install -q joblib;\n!pip install -q colorama;\n!pip install -q optuna;\n!pip install -q ctypes;\n!pip install -q category_encoders;\nclear_output();\n\nimport lightgbm as lgb, sklearn as sk, pandas as pd, polars as pl, numpy as np;\nprint(f\"sklearn = {sk.__version__} | numpy = {np.__version__} | lightgbm = {lgb.__version__}\");\nprint(f\"polars = {pl.__version__} | pandas = {pd.__version__}\\n\\n\");\n\ncollect();","metadata":{"id":"F_swKOeVdxjA","outputId":"147ec7f8-dd5d-46de-8b50-2c731b70e9dc"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n##################################################################\n# Downloading the competition data:-\n\n!kaggle competitions download -c home-credit-credit-risk-model-stability -p /HomeCredit\n!unzip /HomeCredit/home-credit-credit-risk-model-stability.zip -d /HomeCredit;\n\nimport shutil;\ntry:\n    shutil.rmtree(f\"/HomeCredit/csv_files\");\nexcept:\n    pass;\n\n##################################################################\n\nprint();\ncollect();\nclear_output();","metadata":{"id":"TIT4wFf_dhGB","outputId":"42c4b458-e7d6-4a0c-8704-7bff5b7266bb"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nfrom gc import collect;\nfrom warnings import filterwarnings;\nfilterwarnings('ignore');\nfrom IPython.display import display_html, clear_output;\nimport ctypes;\nlibc = ctypes.CDLL(\"libc.so.6\");\n\nfrom pprint import pprint;\nfrom functools import partial;\n\nfrom copy import deepcopy;\nimport pandas as pd, numpy as np, polars as pl, os, joblib;\nimport polars.selectors as cs;\n\nfrom os import path, walk, getpid;\nfrom psutil import Process;\nimport re;\nfrom collections import Counter;\nfrom itertools import product;\nfrom glob import glob;\n\nfrom colorama import Fore, Style, init;\nfrom warnings import filterwarnings;\nfilterwarnings('ignore');\nfrom tqdm.notebook import tqdm;\n\nprint();\ncollect();","metadata":{"id":"jEzCkJZmdjzQ","outputId":"2886a166-0379-402a-86e6-7e565edf0f5c"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# Pipeline specifics:-\nfrom sklearn.model_selection import (StratifiedGroupKFold as SGKF, cross_val_score, cross_val_predict);\nfrom sklearn.pipeline import Pipeline;\nfrom sklearn.base import BaseEstimator, TransformerMixin, ClassifierMixin;\n\n# ML Model training:-\nfrom sklearn.metrics import roc_auc_score, make_scorer;\nfrom sklearn.ensemble import VotingClassifier as VC;\n\nimport torch;\nimport torch.nn as nn;\nfrom lightautoml.automl.presets.tabular_presets import TabularAutoML;\nfrom lightautoml.tasks import Task;\n\nclear_output();\n\nimport xgboost as xgb, lightgbm as lgb, catboost as cb, sklearn as sk;\nprint(f\"\\nXGBoost = {xgb.__version__} | LightGBM = {lgb.__version__} | Catboost = {cb.__version__}\");\nprint(f\"Pandas = {pd.__version__} | Sklearn = {sk.__version__}| Polars = {pl.__version__}\\n\\n\");\ncollect();","metadata":{"id":"1yoUc7f4dp0g","outputId":"a80dbe20-13ae-4fcd-ecb0-5d34cd7065f2"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# Making sklearn pipeline outputs as dataframe:-\nfrom sklearn import set_config;\nset_config(transform_output = \"pandas\");\npd.set_option('display.max_columns', 50);\npd.set_option('display.max_rows', 50);\npd.set_option('display.precision', 3);\n\n# Setting global configurations for polars:-\npl.Config.activate_decimals(True).set_tbl_hide_column_data_types(True);\npl.Config(**dict(tbl_formatting = 'ASCII_FULL_CONDENSED',\n                 tbl_hide_column_data_types = True,\n                 tbl_hide_dataframe_shape = True,\n                 fmt_float = \"mixed\",\n                 tbl_cell_alignment = 'CENTER',\n                 tbl_hide_dtype_separator = True,\n                 tbl_cols = 100,\n                 tbl_rows = 50,\n                 fmt_str_lengths = 100,\n                )\n         );","metadata":{"id":"BqljJU3Tecz0","outputId":"7130d895-976f-4fe8-b1aa-e29410b9fb10"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"2\"></a>\n# <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:black; font-size:120%; text-align:left;padding:3.0px; background: #b3e0ff; border-bottom: 8px solid #1a0d00\" > INTRODUCTION<br> <div>","metadata":{"id":"QHuPn9_1ef6k"}},{"cell_type":"code","source":"%%time\n\nclass Utility:\n    \"\"\"\n    This class serves to do the below-\n    1. Define method to print in color\n    2. Define the classifier metric, custom scorer callable and competition metrics\n    3. Define the garbage cleaning process\n    \"\"\";\n\n    def PrintColor(self,text:str, color = Fore.BLUE, style = Style.BRIGHT):\n        \"Prints color outputs using colorama using a text F-string\";\n        print(style + color + text + Style.RESET_ALL);\n\n    def ScoreMetric(self, ytrue:np.array, ypred: np.array)-> float:\n        \"\"\"\n        This method calculates the classifier metric to evaluate the base-model\n        Inputs- ytrue, ypred:- np.array - input true and predictions arrays\n        Output- float:- base classifier metric, here- GINI score\n        \"\"\";\n        return roc_auc_score(ytrue, ypred);\n\n    def StabilityMetric(self, Stb_Prf: pd.DataFrame, w_fallingrate = 88, w_resstd = -0.5)-> float:\n        \"\"\"\n        This method calculates the GINI stability metric as below-\n        1. Creates an array of GINi scores from week4-91 for in-time testing\n        2. Creates a regression fit-line using numpy.polyfit\n        3. Calculates the stability measure using the formula mentioned in the competition pinned notebook\n        \"\"\";\n\n        y     = Stb_Prf.groupby(\"WEEK_NUM\").apply(lambda x: 2 * self.ScoreMetric(x[CFG.target], x[\"score\"]) -1).values;\n        x     = range(len(y));\n        a, b  = np.polyfit(x, y, 1);\n        y_hat = a * x + b;\n\n        return np.mean(y) + w_fallingrate * min(0, a) + w_resstd * np.std(y - y_hat);\n\n    def CleanMemory(self):\n        \"This method cleans the memory off unused objects and displays the cleaned state RAM usage\";\n\n        collect();\n        libc.malloc_trim(0);\n        pid        = getpid();\n        py         = Process(pid);\n        memory_use = py.memory_info()[0] / 2. ** 30;\n        return f\"\\nRAM usage = {memory_use :.4} GB\";\n\n    def PredictBatch(self, model, X: pd.DataFrame, prd_proba_req: bool = True, batch_size: int = 1000)-> np.array:\n        \"\"\"\n        This method predicts from the model in batches instead of the complete test set to avoid OOM issues in the test set\n        Inputs:-\n        1. X:- Train/ Test set\n        2. prd_proba_req:- need predict proba/ predictions- True [predict_proba], False[predict]\n        3. batch_size:- batch size to consider in one attempt\n\n        Returns:-\n        preds:- array of predicted probabilities\n        \"\"\";\n\n        num_samples = len(X);\n        num_batches = int(np.ceil(num_samples / batch_size));\n        preds       = np.zeros((num_samples,));\n\n        for batch_idx in range(num_batches):\n            self.PrintColor(f\"---> Processing batch: {batch_idx+1}/{num_batches}\", color = Fore.CYAN);\n\n            start_idx = batch_idx * batch_size;\n            end_idx   = min((batch_idx + 1) * batch_size, num_samples);\n            X_batch   = X.iloc[start_idx : end_idx];\n\n            if prd_proba_req == True:\n                batch_probs = model.predict_proba(X_batch)[:, 1];\n            else:\n                batch_probs = model.predict(X_batch).data.squeeze();\n\n            preds[start_idx: end_idx] = batch_probs;\n            _ = self.CleanMemory();\n\n        return preds;\n\nUtils = Utility();\nprint();\n","metadata":{"id":"E6nu1Zaueqdd","outputId":"59b3741b-5d5f-465b-dbb4-befc01e0e674"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data columns**<br>\nThis is available in the original data description as below<br>\nhttps://www.kaggle.com/competitions/home-credit-credit-risk-model-stability/data <br>\n<br>**Competition details and notebook objectives**<br>\n1. This is a binary classification challenge to predict home loan credit defaulters. **GINI** is the metric for the base classifier in this challenge<br>\n2. We also have to additionally assess the stability of GINI measure across time in the evaluation period. We need to score the classifier on a weekly basis and then assess the stability of the weekly GINI score using a regression model against time. Stability measure penalizes models that wane off in prediction capabilities across time. <br>\n2. In this starter notebook, we start the assignment with a simple preprocessing, understanding the data structure of the competition data, basic feature emgineering and develop starter models to initiate the challenge. We will also incorporate other opinions and approaches as we move along the challenge.<br>\n<br>\n**Model strategy** <br>\nWe start off with simple tree based ML models and Denselight model-LAMA and a soft-voting ensemble with appropriate inference in the test set submission. <br>\n<br>**References** <br>\n1. https://www.kaggle.com/code/darynarr/home-credit-drop-date-features <br>\n2. https://www.kaggle.com/code/jetakow/home-credit-2024-starter-notebook <br>\n3. https://www.kaggle.com/code/jirkaborovec/credit-risk-lgbm-optuna-hyper-params <br>\n4. https://www.kaggle.com/code/jirkaborovec/credit-risk-eda-xgboost-depth-0-1-gpu <br>\n5. https://www.kaggle.com/code/peizhengwang/lgb-xgb-cat-ensemble-baseline <br>\n6. https://www.kaggle.com/code/andreynesterov/home-credit-baseline-training-lightautoml <br>","metadata":{"id":"kfzaZAxYeudt"}},{"cell_type":"markdown","source":"<a id=\"2.3\"></a>\n## <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:#ffffff; font-size:120%; text-align:left;padding:3.0px; background:  #008080; border-bottom: 8px solid #001a1a\" > CONFIGURATION<br><div>","metadata":{"id":"6Wd7_p60e2BS"}},{"cell_type":"code","source":"%%time\n\n# Configuration class:-\nclass CFG:\n    \"\"\"\n    Configuration class for parameters and CV strategy for tuning and training\n    \"\"\";\n\n    exp_nb             = 2;\n    version_nb         = 6;\n    test_req           = \"N\";\n    test_sample_frac   = 0.025;\n    state              = 42;\n    target             = 'target';\n    train_path         = \"/HomeCredit/parquet_files/train\";\n    test_path          = \"/HomeCredit/parquet_files/test\";\n    path               = \"/kaggle/input/home-credit-credit-risk-model-stability\";\n    model_path         = f\"/content/drive/MyDrive/HomeCreditQuality\";\n    null_cutoff        = 0.85;\n    cat_cutoff         = 200;\n    n_splits           = 3 if test_req == \"Y\" else 5;\n    n_repeats          = 1 ;\n    nbrnd_erly_stp     = 100;\n    blend_wgt          = [0.15, 0.15, 0.30, 0.20, 0.20];\n\n    # Global variables for plotting:-\n    grid_specs = {'visible': True, 'which': 'both', 'linestyle': '--',\n                  'color': 'lightgrey', 'linewidth': 0.75\n                  };\n    title_specs = {'fontsize': 9, 'fontweight': 'bold', 'color': 'tab:blue'};\n\nprint();\nUtils.PrintColor(f\"--> Configuration done!\");\n_ = Utils.PrintColor(Utils.CleanMemory());","metadata":{"id":"arG4DFTAe6By","outputId":"380a638f-52ed-497e-e6f9-85ec04d8460b"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"2.4\"></a>\n## <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:#ffffff; font-size:120%; text-align:left;padding:3.0px; background:  #008080; border-bottom: 8px solid #001a1a\" > VERSION DETAILS<br><div>","metadata":{"id":"NAQewn1YfIOR"}},{"cell_type":"markdown","source":"|Experiment <br> Number|Version|Details|Features| Models|CV score| Stability Score|Public LB score|\n|:-:|:-:|---|:-:|:-:|:-:|:-:|:-:|\n|2|1|* Used [public notebook](https://www.kaggle.com/code/batprem/home-credit-risk-mode-utility-scripts/notebook) features <br> * Designed LGBM XGB on feature subsets and batch predictions on multiple feature subsets <br> * Null cutoff = 50%|325|LGBM x 2 <br> XGB x 1|0.82743|0.63648||\n|2|2|* Used [public notebook](https://www.kaggle.com/code/batprem/home-credit-risk-mode-utility-scripts/notebook) features <br> * Designed LGBM XGB on feature subsets and batch predictions on multiple feature subsets <br> * Null cutoff = 70%|400|LGBM x 2 <br> XGB x 1|0.83245|0.63662||\n|2|3|* Used [public notebook](https://www.kaggle.com/code/batprem/home-credit-risk-mode-utility-scripts/notebook) features <br> * Designed LGBM on feature subsets and batch predictions on multiple feature subsets <br> * Null cutoff = 80%|422|LGBM x 2|0.83483|0.64798||\n|2|4|* Used [public notebook](https://www.kaggle.com/code/batprem/home-credit-risk-mode-utility-scripts/notebook) features <br> * Designed LGBM on feature subsets and batch predictions on multiple feature subsets <br> * Null cutoff = 95%|502|LGBM x 2|0.83615|0.65295|0.549|\n|2|5|* Used [public notebook](https://www.kaggle.com/code/kononenko/metric-s-trick-home-credit-baseline-inference) features <br> * Designed LGBM XGB combination <br> * Null cutoff > 95% <br> * Used post-processed CV score also|502|LGBM x 4 <br> XGB x 1 <br> LAMA x 1|0.83510<br>0.83836|0.65261<br>0.66045|0.61|\n|2|6|* Used [public notebook](https://www.kaggle.com/code/kononenko/metric-s-trick-home-credit-baseline-inference) features <br> * Null cutoff > 85%<br> * Used post-processed CV score also|440|LAMA x 1|0.82843|0.63397||\n\n","metadata":{"id":"0RF4jfQsfLli"}},{"cell_type":"markdown","source":"<a id=\"3\"></a>\n# <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:black; font-size:120%; text-align:left;padding:3.0px; background: #b3e0ff; border-bottom: 8px solid #1a0d00\" > PREPROCESSING<br> <div>","metadata":{"id":"OPIsksRgfR9C"}},{"cell_type":"code","source":"%%time\n\nclass DataXformer:\n    \"\"\"\n    This is a comprehensive preprocessing and data transformer class that does the below-\n    1. consumes the input data tables\n    2. creates secondary features\n    3. ensues memory efficient outputs for the model\n    \"\"\";\n\n    def __init__(self, null_cutoff: float, cat_cutoff: int,\n                 TrainTest: str      = \"train\",\n                 sel_cols: list      = [],\n                 cat_cols: list      = [],\n                 **kwarg\n                 ):\n\n        self.TrainTest   = TrainTest;\n        self.null_cutoff = null_cutoff;\n        self.cat_cutoff  = cat_cutoff;\n        self.target      = CFG.target;\n        self.sel_cols    = sel_cols;\n        self.cat_cols    = cat_cols;\n\n        if self.TrainTest == \"train\":\n            self.path = CFG.train_path;\n        else:\n            self.path = CFG.test_path;\n\n        Utils.PrintColor(f\"\\n{'='*10} {self.TrainTest.upper()} MODE {'='*10}\\n\", color = Fore.RED);\n\n    def _TypeCastCols(self, df: pl.DataFrame):\n        \"\"\"\n        This method casts the columns into the desired dtypes with basic date handling too\n        Input- df- pl.DataFrame:- input data table\n        Output- df:- pl.DataFrame:- dataframe with type-casting\n        \"\"\";\n\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64));\n            elif col in [\"date_decision\"] or col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date));\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64));\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String));\n\n        return df;\n\n    def _MakeDtFtre(self, df: pl.DataFrame):\n        \"\"\"\n        This method creates date features from the provided dataframe\n        Input- df- pl.DataFrame:- input data table\n        Output- df- pl.DataFrame:- dataframe with date column FE\n        \"\"\";\n\n        for col in df.columns:\n            if col.endswith(\"D\"):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"));\n                df = df.with_columns(pl.col(col).dt.total_days());\n                df = df.with_columns([pl.col(\"date_decision\").dt.month().alias(\"month_nb\").cast(pl.Int8),\n                                      pl.col(\"date_decision\").dt.weekday().alias(\"weekday_nb\").cast(pl.Int8),\n                                     ]\n                                    );\n        return df.drop(\"date_decision\", \"MONTH\");\n\n    def _MakeAgg(self, df: pl.DataFrame):\n        \"\"\"\n        This method makes a set of aggregate expressions for group by on case id for depth > 0 tables\n\n        Note:-\n        1. We make [max, min, first, last] aggregations for all columns\n        2. We make mean aggregation for columns ending with [P, A, D]\n        3. We make mode aggregations for columns ending with [M]\n\n        Input - df- pl.DataFrame:- input data table\n        Output- all_agg:- list of aggregate expressions to be used with group_by case_id\n        \"\"\";\n\n        all_agg = [];\n        df_cols = df.columns;\n\n        all_agg.extend([method(col).alias(f\"{method.__name__}_{col}\") \\\n                        for method in [pl.max, pl.min, pl.first, pl.last] \\\n                        for col in df_cols if col[-1] in (\"P\", \"A\", \"D\", \"M\", \"T\", \"L\") or \"num_group\" in col\n                        ]\n                       );\n        all_agg.extend([pl.col(col).mean().alias(f\"mean_{col}\") for col in df_cols if col.endswith((\"P\", \"A\", \"D\"))]);\n        all_agg.extend([pl.col(col).drop_nulls().mode().first().alias(f\"mode_{col}\") for col in df_cols if col.endswith(\"M\")]);\n        return df.group_by(\"case_id\").agg(all_agg);\n\n    def _PreProcessIpTbl(self, path:str, depth: int, isSingle: bool, **kwarg):\n        \"\"\"\n        This method does the below-\n        1. Creates chunks for file loads if we have multiple files (isSingle = False)\n        2. Concatenates the chunks to a single file with typecasting\n        3. Aggregating on case id for depth > 0 tables\n        \"\"\";\n\n        if isSingle == False:\n            components = [];\n            for path in glob(str(path)):\n                components.append(pl.scan_parquet(path).pipe(self._TypeCastCols));\n            df = pl.concat(components, how = \"vertical_relaxed\");\n        else:\n            df = pl.scan_parquet(path).pipe(self._TypeCastCols);\n\n        if depth > 0:\n            return df.pipe(self._MakeAgg);\n        else:\n            return df;\n\n    @staticmethod\n    def _ReduceMem(df: pd.DataFrame):\n        \"This method reduces memory for numeric columns in the dataframe\";\n\n        numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64', \"uint16\", \"uint32\", \"uint64\"];\n        start_mem = df.memory_usage().sum() / 1024**2;\n\n        for col in df.columns:\n            col_type = df[col].dtypes\n\n            if col_type in numerics:\n                c_min = df[col].min();\n                c_max = df[col].max();\n\n                if \"int\" in str(col_type):\n                    if c_min >= np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min >= np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min >= np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min >= np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n                else:\n                    if c_min >= np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    if c_min >= np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        df[col] = df[col].astype(np.float64)\n\n        end_mem = df.memory_usage().sum() / 1024**2;\n\n        Utils.PrintColor(f\"Start - end memory:- {start_mem:5.2f} - {end_mem:5.2f} Mb\");\n        return df;\n\n    def _MakeModelData(self, df_base, depth_0, depth_1, depth_2, **kwarg):\n        \"\"\"\n        This method aggregates the input tables and joins them to make a single model table for the next steps\n        It converts the final table to a pandas dataframe for the next steps, reduces the memory consumption and selects relevant columns\n        \"\"\";\n\n        for i, df in enumerate(depth_0 + depth_1 + depth_2):\n            df_base = df_base.join(df, how = \"left\", on = \"case_id\", suffix = f\"_{i}\");\n\n        df = df_base.pipe(self._MakeDtFtre).collect().to_pandas();\n        df = self._ReduceMem(df.replace([np.inf, -1*np.inf], np.NaN));\n\n        if self.TrainTest.lower() == \"train\":\n            Utils.PrintColor(f\"---> Selecting training columns by nulls and category unique values\",\n                             color = Fore.CYAN\n                             );\n\n            drop_cols = [];\n            null_cols = df.drop(columns = [self.target], errors = \"ignore\").isna().mean();\n            drop_cols.extend(null_cols.loc[null_cols > self.null_cutoff].index.to_list());\n\n            obj_cols = df.select_dtypes(include = \"object\").columns;\n            for col in obj_cols:\n                if df[col].nunique() > self.cat_cutoff or df[col].nunique() == 1:\n                    drop_cols.append(col);\n            cat_cols  = [c for c in obj_cols if c not in drop_cols];\n\n            df = df.drop(columns = drop_cols, errors = \"ignore\");\n            df[cat_cols] = df[cat_cols].astype(\"category\");\n\n        else:\n            Utils.PrintColor(f\"---> Selecting the test-set columns and aligning category columns with train data\");\n            df                = df[self.sel_cols];\n            df[self.cat_cols] = df[self.cat_cols].astype(\"category\");\n\n        return df;\n\n    def XformData(self, display_store: bool = False):\n        \"\"\"\n        This is the cynosure method that aggregates all the inputs and prepares the FE dataset for the model training/ submission\n        \"\"\";\n\n        data_store = \\\n         {\"df_base\" : self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_base.parquet\"), depth = 0, isSingle = True),\n\n          \"depth_0\" : [self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_static_cb_0.parquet\"), depth = 0, isSingle = True),\n                       self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_static_0_*.parquet\"), depth = 0, isSingle = False)\n                      ],\n\n          \"depth_1\": [self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_applprev_1_*.parquet\"), depth = 1, isSingle = False),\n                      self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_tax_registry_a_1.parquet\"), depth = 1, isSingle = True),\n                      self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_tax_registry_b_1.parquet\"), depth = 1, isSingle = True),\n                      self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_tax_registry_c_1.parquet\"), depth = 1, isSingle = True),\n                      self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_credit_bureau_b_1.parquet\"), depth = 1, isSingle = True),\n                      self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_other_1.parquet\"), depth = 1, isSingle = True),\n                      self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_person_1.parquet\"), depth = 1, isSingle = True),\n                      self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_deposit_1.parquet\"), depth = 1, isSingle = True),\n                      self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_debitcard_1.parquet\"), depth = 1, isSingle = True)\n                      ],\n\n          \"depth_2\": [self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_credit_bureau_b_2.parquet\"), depth = 2, isSingle = True)]\n          };\n\n        if display_store:\n            Utils.PrintColor(\"\\n---> Data store\\n\", color = Fore.CYAN);\n            pprint(data_store, width = 200, depth = 3, indent = 5);\n\n        df = self._MakeModelData(**data_store);\n        Utils.PrintColor(f\"\\n---> {self.TrainTest.capitalize()} set details = {df.shape} | {df.memory_usage().sum()/ 10**6 :,.2f} Mb\\n\",\n                         color = Fore.CYAN\n                         );\n        del data_store;\n\n        if self.TrainTest.lower() == \"train\":\n            cols = df.drop(columns = [CFG.target, \"WEEK_NUM\", \"case_id\"], errors = \"ignore\").columns;\n            joblib.dump(cols, os.path.join(CFG.model_path, f\"SelCols_E{CFG.exp_nb}V{CFG.version_nb}.pkl\"));\n            cat_cols = df.select_dtypes(include = \"category\").columns;\n            joblib.dump(cat_cols, os.path.join(CFG.model_path, f\"SelCatCols_E{CFG.exp_nb}V{CFG.version_nb}.pkl\"));\n\n            with np.printoptions(linewidth = 160):\n                Utils.PrintColor(\"\\n---> Train set columns\\n\");\n                pprint(np.array(cols));\n                Utils.PrintColor(\"\\n---> Train set category columns\\n\", color = Fore.CYAN);\n                pprint(np.array(cat_cols));\n        else:\n            pass;\n        return df;\n\nUtils.PrintColor(Utils.CleanMemory());","metadata":{"id":"2xZxKQGhAljN","outputId":"926448a3-848c-41f2-8a64-00cb9bf615ae"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\npp      = DataXformer(TrainTest = \"train\", null_cutoff = CFG.null_cutoff, cat_cutoff = CFG.cat_cutoff);\nXYtrain = pp.XformData(display_store = False);\nUtils.PrintColor(Utils.CleanMemory());\n\nif CFG.test_req == \"Y\":\n    XYtrain = XYtrain.groupby([\"WEEK_NUM\", CFG.target]).sample(frac = CFG.test_sample_frac).sort_index();\n    XYtrain.index = range(len(XYtrain));\n    Utils.PrintColor(f\"---> Train shape after sampling = {XYtrain.shape}\");\n\npp      = DataXformer(TrainTest   = \"test\",\n                      null_cutoff = CFG.null_cutoff,\n                      cat_cutoff  = CFG.cat_cutoff,\n                      sel_cols    = XYtrain.drop(columns = [CFG.target]).columns.tolist(),\n                      cat_cols    = XYtrain.select_dtypes(include = \"category\").columns.tolist(),\n                      );\nXtest   = pp.XformData(display_store = False);\n\nUtils.PrintColor(Utils.CleanMemory());","metadata":{"id":"UVNYRMpU00h_","outputId":"dc625f2c-90eb-4f50-ffc3-871634f3b09f"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"4\"></a>\n# <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:black; font-size:120%; text-align:left;padding:3.0px; background: #b3e0ff; border-bottom: 8px solid #1a0d00\" > MODEL TRAINING<br> <div>","metadata":{"id":"-htSH8n28ORa"}},{"cell_type":"code","source":"%%time\n\nmodel = TabularAutoML(task = Task('binary', loss = 'logloss', metric = 'auc'),\n                      timeout   = 10000,\n                      cpu_limit = 4,\n                      gpu_ids   = '0',\n                      general_params = {\"use_algos\": [[\"denselight\"]]},\n                      nn_params = {\"n_epochs\"       : 5,\n                                   \"bs\"             : 128,\n                                   \"num_workers\"    : 0,\n                                   \"path_to_save\"   : None,\n                                   \"freeze_defaults\": True,\n                                   \"cont_embedder\"  : \"cont\",\n                                   },\n                      nn_pipeline_params = {\"use_qnt\": False, \"use_te\": False},\n                      reader_params = {'n_jobs': 4,\n                                       'cv'    : CFG.n_splits,\n                                       'random_state': CFG.state,\n                                       'advanced_roles': False\n                                       },\n                      );\n\noof_preds = \\\nmodel.fit_predict(XYtrain,\n                  roles   = {'target': CFG.target,'group': \"WEEK_NUM\",'drop' : ['case_id', \"WEEK_NUM\"]},\n                  verbose = 0\n                  );\n\n","metadata":{"id":"PvSpEfVbfOFm","outputId":"ac8fda03-94aa-4806-d036-f261b2bf5d5f"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nUtils.PrintColor(f\"OOF score = {Utils.ScoreMetric(XYtrain[CFG.target], oof_preds.data):.5f}\");\nstb_metric = Utils.StabilityMetric(XYtrain[[\"WEEK_NUM\", CFG.target]].assign(score = oof_preds.data.flatten()));\nUtils.PrintColor(f\"Stability Metric = {stb_metric:.5f}\\n\\n\", color = Fore.CYAN);\n\njoblib.dump(model, os.path.join(CFG.model_path, f\"E{CFG.exp_nb}V{CFG.version_nb}_DENSELIGHT.model\"));\n\n_ = Utils.CleanMemory();","metadata":{"id":"LPb9bRbbxPKf","outputId":"8d9f26cb-31f8-4040-a764-12b54e97b831"},"execution_count":null,"outputs":[]}]}