{"metadata":{"colab":{"provenance":[],"machine_shape":"hm","gpuType":"T4"},"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"accelerator":"GPU","kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":7633217,"sourceType":"datasetVersion","datasetId":4422102},{"sourceId":7742446,"sourceType":"datasetVersion","datasetId":4256688},{"sourceId":161936385,"sourceType":"kernelVersion"},{"sourceId":162314401,"sourceType":"kernelVersion"},{"sourceId":162317063,"sourceType":"kernelVersion"},{"sourceId":162351144,"sourceType":"kernelVersion"},{"sourceId":162470947,"sourceType":"kernelVersion"},{"sourceId":162486662,"sourceType":"kernelVersion"}],"dockerImageVersionId":30648,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#### <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color: black; font-size:180%; text-align:left;padding:3.0px; background: #ffebcc; border-bottom: 8px solid black\" > TABLE OF CONTENTS<br><div>  \n* [IMPORTS AND INSTALLATIONS](#1)\n* [INTRODUCTION](#2)\n    * [UTILITIES](#2.1)\n    * [DATASET DETAILS](#2.2)    \n    * [CONFIGURATION](#2.3)\n    * [VERSION DETAILS](#2.4)\n* [PREPROCESSING](#3)\n* [MODEL INFERENCING](#4) \n    * [MY LGBM-XGB MODELS](#4.1)\n    * [LAMA DENSELIGHT MODEL](#4.2)\n    * [PUBLIC AUTOGLUON MODEL](#4.3)\n    * [FINAL PREDICTIONS](#4.4)\n* [PLANNED NEXT STEPS](#5)  ","metadata":{"id":"nUVs0Nmxd0oR"}},{"cell_type":"markdown","source":"<a id=\"1\"></a>\n# <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:white; font-size:120%; text-align:left;padding:3.0px; background: #b32d00; border-bottom: 8px solid black\" > PACKAGE IMPORTS AND INSTALLATIONS<br> <div>","metadata":{"id":"FNfLSqY-d79o"}},{"cell_type":"code","source":"%%time\n\nfrom IPython.display import clear_output;\nfrom gc import collect;\n!pip install -q \"/kaggle/input/pythonlibrarieswheelfiles/scikit_learn-1.3.2-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\";\n!pip install -q \"/kaggle/input/pythonlibrarieswheelfiles/lightgbm-4.3.0-py3-none-manylinux_2_28_x86_64.whl\";\n\nprint();\ncollect();","metadata":{"id":"F_swKOeVdxjA","outputId":"6cb6f272-5e37-4aaa-f72d-02d7f7ac13b9","execution":{"iopub.status.busy":"2024-03-23T14:00:13.133579Z","iopub.execute_input":"2024-03-23T14:00:13.134250Z","iopub.status.idle":"2024-03-23T14:01:27.673456Z","shell.execute_reply.started":"2024-03-23T14:00:13.134205Z","shell.execute_reply":"2024-03-23T14:01:27.672309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nfrom gc import collect;\nfrom warnings import filterwarnings;\nfilterwarnings('ignore');\nfrom IPython.display import display_html, clear_output;\nimport ctypes;\nlibc = ctypes.CDLL(\"libc.so.6\");\n\nfrom pprint import pprint;\nfrom functools import partial;\n\nfrom copy import deepcopy;\nimport pandas as pd, numpy as np, polars as pl, os, joblib;\nimport polars.selectors as cs;\n\nfrom os import path, walk, getpid;\nfrom psutil import Process;\nimport re;\nfrom collections import Counter;\nfrom itertools import product;\nfrom glob import glob;\n\nfrom colorama import Fore, Style, init;\nfrom warnings import filterwarnings;\nfilterwarnings('ignore');\nfrom tqdm.notebook import tqdm;\n\nprint();\ncollect();","metadata":{"id":"jEzCkJZmdjzQ","outputId":"e15adce7-9e23-4b86-fb89-9aa6ccf50cc1","execution":{"iopub.status.busy":"2024-03-23T14:01:30.713840Z","iopub.execute_input":"2024-03-23T14:01:30.714271Z","iopub.status.idle":"2024-03-23T14:01:31.562461Z","shell.execute_reply.started":"2024-03-23T14:01:30.714226Z","shell.execute_reply":"2024-03-23T14:01:31.561233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# Pipeline specifics:-\nfrom sklearn.model_selection import (StratifiedGroupKFold as SGKF, cross_val_score, cross_val_predict);\nfrom sklearn.pipeline import Pipeline;\nfrom sklearn.base import BaseEstimator, TransformerMixin, ClassifierMixin;\n\n# ML Model training:-\nfrom sklearn.metrics import roc_auc_score, make_scorer;\nfrom xgboost import DMatrix, XGBClassifier as XGBC;\nfrom lightgbm import log_evaluation, early_stopping, LGBMClassifier as LGBMC;\nfrom catboost import CatBoostClassifier as CBC, Pool;\nfrom sklearn.ensemble import VotingClassifier as VC;\nclear_output();\n\nimport lightgbm as lgb, xgboost as xgb;\nprint(f\"\\nLightGBM = {lgb.__version__}| XGBoost = {xgb.__version__}\\n\\n\");\ncollect();","metadata":{"id":"1yoUc7f4dp0g","outputId":"fa206e98-aa2e-466c-e7c9-446657f30270","execution":{"iopub.status.busy":"2024-03-23T14:01:34.656955Z","iopub.execute_input":"2024-03-23T14:01:34.658030Z","iopub.status.idle":"2024-03-23T14:01:39.739095Z","shell.execute_reply.started":"2024-03-23T14:01:34.657989Z","shell.execute_reply":"2024-03-23T14:01:39.737925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# Making sklearn pipeline outputs as dataframe:-\nfrom sklearn import set_config;\nset_config(transform_output = \"pandas\");\npd.set_option('display.max_columns', 50);\npd.set_option('display.max_rows', 50);\npd.set_option('display.precision', 3);\n\n# Setting global configurations for polars:-\npl.Config.activate_decimals(True).set_tbl_hide_column_data_types(True);\npl.Config(**dict(tbl_formatting = 'ASCII_FULL_CONDENSED',\n                 tbl_hide_column_data_types = True,\n                 tbl_hide_dataframe_shape = True,\n                 fmt_float = \"mixed\",\n                 tbl_cell_alignment = 'CENTER',\n                 tbl_hide_dtype_separator = True,\n                 tbl_cols = 100,\n                 tbl_rows = 50,\n                 fmt_str_lengths = 100,\n                )\n         );","metadata":{"id":"BqljJU3Tecz0","outputId":"ee02e66e-aefb-4e4a-9328-c81466226988","execution":{"iopub.status.busy":"2024-03-23T14:01:42.648673Z","iopub.execute_input":"2024-03-23T14:01:42.649428Z","iopub.status.idle":"2024-03-23T14:01:42.660494Z","shell.execute_reply.started":"2024-03-23T14:01:42.649391Z","shell.execute_reply":"2024-03-23T14:01:42.659241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"2\"></a>\n# <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:white; font-size:120%; text-align:left;padding:3.0px; background: #b32d00; border-bottom: 8px solid black\" > INTRODUCTION<br> <div>","metadata":{"id":"QHuPn9_1ef6k"}},{"cell_type":"markdown","source":"<a id=\"2.1\"></a>\n## <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:black; font-size:120%; text-align:left;padding:3.0px; background:  #e0ebeb; border-bottom: 8px solid #b32d00\" > UTILITIES<br><div>","metadata":{}},{"cell_type":"code","source":"%%time\n\nclass Utility:\n    \"\"\"\n    This class serves to do the below-\n    1. Define method to print in color\n    2. Define the classifier metric, custom scorer callable and competition metrics\n    3. Define the garbage cleaning process\n    4. Define the predict-in-batch method to prevent OOM issues\n    \"\"\";\n\n    def PrintColor(self,text:str, color = Fore.BLUE, style = Style.BRIGHT):\n        \"Prints color outputs using colorama using a text F-string\";\n        print(style + color + text + Style.RESET_ALL);\n\n    def ScoreMetric(self, ytrue:np.array, ypred: np.array)-> float:\n        \"\"\"\n        This method calculates the classifier metric to evaluate the base-model\n        Inputs- ytrue, ypred:- np.array - input true and predictions arrays\n        Output- float:- base classifier metric, here- GINI score\n        \"\"\";\n        return roc_auc_score(ytrue, ypred);\n\n    def StabilityMetric(self, base, w_fallingrate=88.0, w_resstd=-0.5, week_lbl = \"WEEK_NUM\"):\n        \"\"\"\n        This method defines the GINI-stability metric as required for the competition\n        Source:- https://www.kaggle.com/code/darynarr/home-credit-drop-date-features\n        \"\"\";\n\n        grp_gini = \\\n        base.loc[:, [week_lbl, \"target\", \"score\"]].\\\n        sort_values(week_lbl).\\\n        groupby(week_lbl)[[\"target\", \"score\"]].\\\n        apply(lambda x: 2*roc_auc_score(x[\"target\"], x[\"score\"])-1).tolist();\n\n        x         = np.arange(len(grp_gini));\n        y         = grp_gini;\n        a, b      = np.polyfit(x, y, 1);\n        y_hat     = a*x + b;\n        residuals = y - y_hat;\n        res_std   = np.std(residuals);\n        avg_gini  = np.mean(gini_in_time);\n\n        return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std;\n\n    def CleanMemory(self):\n        \"This method cleans the memory off unused objects and displays the cleaned state RAM usage\";\n\n        collect();\n        libc.malloc_trim(0);\n        pid        = getpid();\n        py         = Process(pid);\n        memory_use = py.memory_info()[0] / 2. ** 30;\n        return f\"\\nRAM usage = {memory_use :.4} GB\";\n    \n    def PredictBatch(self, model, X: pd.DataFrame, prd_proba_req: bool = True, batch_size: int = 1000)-> np.array:\n        \"\"\"\n        This method predicts from the model in batches instead of the complete test set to avoid OOM issues in the test set\n        Inputs:-\n        1. X:- Train/ Test set\n        2. prd_proba_req:- need predict proba/ predictions- True [predict_proba], False[predict]\n        3. batch_size:- batch size to consider in one attempt\n\n        Returns:-\n        preds:- array of predicted probabilities\n        \"\"\";\n\n        num_samples = len(X);\n        num_batches = int(np.ceil(num_samples / batch_size));\n        preds       = np.zeros((num_samples,));\n\n        for batch_idx in range(num_batches):\n            self.PrintColor(f\"---> Processing batch: {batch_idx+1}/{num_batches}\", color = Fore.CYAN);\n\n            start_idx = batch_idx * batch_size;\n            end_idx   = min((batch_idx + 1) * batch_size, num_samples);\n            X_batch   = X.iloc[start_idx : end_idx];\n\n            if prd_proba_req == True:\n                batch_probs = model.predict_proba(X_batch)[:, 1];\n            else:\n                batch_probs = model.predict(X_batch).data.squeeze();\n\n            preds[start_idx: end_idx] = batch_probs;\n            _ = self.CleanMemory();\n\n        return preds;\n    \n    def PostProcessPreds(self, week_num: np.array, sub: pd.DataFrame, shift: float = 0.025):\n        \"\"\"\n        This method hacks the metric using stability definition- use with care\n        Source- https://www.kaggle.com/code/hideyukizushi/home-aftersubmissionsopen-3-11-2024-lb-567\n        \"\"\";\n        \n        sub[\"WEEK_NUM\"] = week_num;\n        condition       = sub[\"WEEK_NUM\"] < (sub[\"WEEK_NUM\"].max() - sub[\"WEEK_NUM\"].min())/2 + sub[\"WEEK_NUM\"].min();\n        sub.loc[condition, 'score'] = (sub.loc[condition, 'score'] - shift).clip(0.0);\n        return sub.drop([\"WEEK_NUM\"], axis = 1, errors = \"ignore\");\n    \nUtils = Utility();\nprint();\n","metadata":{"id":"E6nu1Zaueqdd","outputId":"5cfb87db-db47-4d2f-f4b2-c8d26c696bd7","execution":{"iopub.status.busy":"2024-03-23T14:01:45.218433Z","iopub.execute_input":"2024-03-23T14:01:45.218813Z","iopub.status.idle":"2024-03-23T14:01:45.240106Z","shell.execute_reply.started":"2024-03-23T14:01:45.218784Z","shell.execute_reply":"2024-03-23T14:01:45.238959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"2.2\"></a>\n## <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:black; font-size:120%; text-align:left;padding:3.0px; background:  #e0ebeb; border-bottom: 8px solid #b32d00\" > FOREWORD<br><div>","metadata":{}},{"cell_type":"markdown","source":"**Data columns**<br>\nThis is available in the original data description as below<br>\nhttps://www.kaggle.com/competitions/home-credit-credit-risk-model-stability/data <br>\n<br>**Competition details and notebook objectives**<br>\n1. This is a binary classification challenge to predict home loan credit defaulters. **GINI** is the metric for the base classifier in this challenge<br>\n2. We also have to additionally assess the stability of GINI measure across time in the evaluation period. We need to score the classifier on a weekly basis and then assess the stability of the weekly GINI score using a regression model against time. Stability measure penalizes models that wane off in prediction capabilities across time. <br>\n2. In this starter notebook, we start the assignment with a simple preprocessing, understanding the data structure of the competition data, basic feature emgineering and develop starter models to initiate the challenge. We will also incorporate other opinions and approaches as we move along the challenge.<br>\n<br>\n**Model strategy** <br>\nWe start off with simple tree based ML models and a soft-voting ensemble with appropriate inference in the test set submission. \n\n<br>**Kernel and method strategy**<br>\n1. We start off with a simple data transformer class that prepares the secondary features for the challenge. Considering the large size of the data, managing these tables within the confines of a Kaggle notebook environment will be a challenge, especially for model training <br> \n2. We verify the correctness of the data processor class on the train and test datasets, keeping track of execution time and memory usage too <br>\n3. We then train ML models here using similar but slightly varying parameters and then blend them using heuristic weights. <br>\n4. To prevent a deluge of tables and model objects in the inference pipeline, we synthesize an inherited voting classifier from each model, intaking the fold level model objects as inputs. These objects will be fed into the inference kernel and a combined prediction will be created <br>\n5. Training kernel is placed [here](https://www.kaggle.com/code/ravi20076/homecredit-starter-training-v1) for perusal and possible replication <br>\n\n<br> **New beginnings** <br>\n1. This kernel is incidently my first tryst with the competition after becoming a Competitions Master late yesterday. I am highly overwhelmed and overjoyed at the collective success our team enjoyed over the past 5-6 months to ascend from competition Novices and Contributors to Master. 3 of our team-members became Masters yesterday (22/03/2024), an incredible success for the group. <br>\n2. I shall explore the new beginnings in this competition too, after the submissions reopened recently, in this kernel and beyond. I shall edit and update this kernel as we move along, adding/ removing features and models as deemed needed. <br>\n3. I shall reuse my past work as a quick starter for this challenge and append a public AutoGluon model from yester work, also a part of my previous submission herewith. Hopefully, this will help boost the score <br>\n\n<br>I extend sincere wishes to one and all for the challenge and hope to do my best herewith. Happy learning and best regards!","metadata":{"id":"kfzaZAxYeudt"}},{"cell_type":"markdown","source":"<a id=\"2.3\"></a>\n## <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:black; font-size:120%; text-align:left;padding:3.0px; background:  #e0ebeb; border-bottom: 8px solid #b32d00\" > CONFIGURATION<br><div>","metadata":{"id":"6Wd7_p60e2BS"}},{"cell_type":"code","source":"%%time\n\n# Configuration class:-\nclass CFG:\n    \"\"\"\n    Configuration class for parameters and CV strategy for tuning and training\n    \"\"\";\n    \n    exp_nb             = 1;\n    version_nb         = 2;\n    test_req           = \"N\";\n    test_sample_frac   = 0.025;\n    state              = 42;\n    target             = 'target';\n    train_path         = \"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train\";\n    test_path          = \"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/test\";\n    path               = \"/kaggle/input/home-credit-credit-risk-model-stability\";\n    model_path         = \"/kaggle/input/homecreditquality2024ancillary\";\n    null_cutoff        = 0.95;\n    cat_cutoff         = 200;\n    n_splits           = 3 if test_req == \"Y\" else 5;\n    n_repeats          = 1 ;\n    nbrnd_erly_stp     = 100;\n    all_exp_nb         = [\"E2V1\", \"E2V2\", \"E2V3\", \"E2V5\"];\n    myml_inner_wgt     = [0.50, 0.25, 0.25];\n    mymdl_wgt          = [0.10, 0.15, 0.25, 0.50];\n    blend_wgt          = [0.25, 0.10, 0.05, 0.60];\n    pp_preds           = False;\n    shift              = 0.025;\n\n    # Global variables for plotting:-\n    grid_specs = {'visible': True, 'which': 'both', 'linestyle': '--',\n                  'color': 'lightgrey', 'linewidth': 0.75\n                  };\n    title_specs = {'fontsize': 9, 'fontweight': 'bold', 'color': 'tab:blue'};\n\nprint();\nUtils.PrintColor(f\"--> Configuration done!\");\nUtils.PrintColor(f\"--> Sum of blend weights = {sum(CFG.blend_wgt):.2f}\");\n\n_ = Utils.PrintColor(Utils.CleanMemory());","metadata":{"id":"arG4DFTAe6By","outputId":"bae5bb4b-c60d-4eac-94cf-dbfe9e206302","execution":{"iopub.status.busy":"2024-03-23T14:01:49.333221Z","iopub.execute_input":"2024-03-23T14:01:49.333614Z","iopub.status.idle":"2024-03-23T14:01:49.489346Z","shell.execute_reply.started":"2024-03-23T14:01:49.333582Z","shell.execute_reply":"2024-03-23T14:01:49.488233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"2.4\"></a>\n## <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:black; font-size:120%; text-align:left;padding:3.0px; background:  #e0ebeb; border-bottom: 8px solid #b32d00\" > SUBMISSION HISTORY<br><div>","metadata":{"id":"NAQewn1YfIOR"}},{"cell_type":"markdown","source":"|Experiment <br> Number|Version|Details|Features| Models|CV score|Stability score| Public LB score|\n|:-:|:-:|---|:-:|:-:|:-:|:-:|:-:|\n|1|1|* Used my previous work including LGBM and LAMA models <br> * Aggregated experiments 2.1-2.6 |325-512|LGBM x 5|||0.561|\n|1|2|* Used my previous work including LGBM and LAMA models <br> * Added public AutoGluon model <br> * Aggregated experiments 2.1-2.6 <br> * No post-processing |325-512|LGBM x 5 <br> LAMA x 2|||0.569|\n|1|3|* Same as E1V2 above <br> * Included post-processing |325-512|LGBM x 5 <br> LAMA x 2|||0.569|","metadata":{"id":"0RF4jfQsfLli"}},{"cell_type":"markdown","source":"<a id=\"3\"></a>\n# <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:white; font-size:120%; text-align:left;padding:3.0px; background: #b32d00; border-bottom: 8px solid black\" > PREPROCESSING<br> <div>","metadata":{"id":"OPIsksRgfR9C"}},{"cell_type":"code","source":"%%time\n\nclass DataXformer:\n    \"\"\"\n    This is a comprehensive preprocessing and data transformer class that does the below-\n    1. consumes the input data tables\n    2. creates secondary features\n    3. ensues memory efficient outputs for the model\n    \"\"\";\n\n    def __init__(self, null_cutoff: float, cat_cutoff: int,\n                 TrainTest: str      = \"test\",\n                 sel_cols: list      = [],\n                 cat_cols: list      = [],\n                 **kwarg\n                 ):\n\n        self.TrainTest   = TrainTest;\n        self.null_cutoff = null_cutoff;\n        self.cat_cutoff  = cat_cutoff;\n        self.target      = CFG.target;\n        self.sel_cols    = sel_cols;\n        self.cat_cols    = cat_cols;\n\n        if self.TrainTest == \"train\":\n            self.path = CFG.train_path;\n        else:\n            self.path = CFG.test_path;\n\n        Utils.PrintColor(f\"\\n{'='*10} {self.TrainTest.upper()} MODE {'='*10}\\n\", color = Fore.RED);\n\n    def _TypeCastCols(self, df: pl.DataFrame):\n        \"\"\"\n        This method casts the columns into the desired dtypes with basic date handling too\n        Input- df- pl.DataFrame:- input data table\n        Output- df:- pl.DataFrame:- dataframe with type-casting\n        \"\"\";\n\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64));\n            elif col in [\"date_decision\"] or col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date));\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64));\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String));\n\n        return df;\n\n    def _MakeDtFtre(self, df: pl.DataFrame):\n        \"\"\"\n        This method creates date features from the provided dataframe\n        Input- df- pl.DataFrame:- input data table\n        Output- df- pl.DataFrame:- dataframe with date column FE\n        \"\"\";\n\n        for col in df.columns:\n            if col.endswith(\"D\"):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"));\n                df = df.with_columns(pl.col(col).dt.total_days());\n                df = df.with_columns([pl.col(\"date_decision\").dt.month().alias(\"month_nb\").cast(pl.Int8),\n                                      pl.col(\"date_decision\").dt.weekday().alias(\"weekday_nb\").cast(pl.Int8),\n                                     ]\n                                    );\n        return df.drop(\"date_decision\", \"MONTH\");\n\n    def _MakeAgg(self, df: pl.DataFrame):\n        \"\"\"\n        This method makes a set of aggregate expressions for group by on case id for depth > 0 tables\n\n        Note:-\n        1. We make [max, min, first, last] aggregations for all columns\n        2. We make mean aggregation for columns ending with [P, A, D]\n        3. We make mode aggregations for columns ending with [M]\n\n        Input - df- pl.DataFrame:- input data table\n        Output- all_agg:- list of aggregate expressions to be used with group_by case_id\n        \"\"\";\n\n        all_agg = [];\n        df_cols = df.columns;\n\n        all_agg.extend([method(col).alias(f\"{method.__name__}_{col}\") \\\n                        for method in [pl.max, pl.min, pl.first, pl.last] \\\n                        for col in df_cols if col[-1] in (\"P\", \"A\", \"D\", \"M\", \"T\", \"L\") or \"num_group\" in col\n                        ]\n                       );\n        all_agg.extend([pl.col(col).mean().alias(f\"mean_{col}\") for col in df_cols if col.endswith((\"P\", \"A\", \"D\"))]);\n        all_agg.extend([pl.col(col).drop_nulls().mode().first().alias(f\"mode_{col}\") for col in df_cols if col.endswith(\"M\")]);\n        return df.group_by(\"case_id\").agg(all_agg);\n\n    def _PreProcessIpTbl(self, path:str, depth: int, isSingle: bool, **kwarg):\n        \"\"\"\n        This method does the below-\n        1. Creates chunks for file loads if we have multiple files (isSingle = False)\n        2. Concatenates the chunks to a single file with typecasting\n        3. Aggregating on case id for depth > 0 tables\n        \"\"\";\n\n        if isSingle == False:\n            components = [];\n            for path in glob(str(path)):\n                components.append(pl.scan_parquet(path).pipe(self._TypeCastCols));\n            df = pl.concat(components, how = \"vertical_relaxed\");\n        else:\n            df = pl.scan_parquet(path).pipe(self._TypeCastCols);\n\n        if depth > 0:\n            return df.pipe(self._MakeAgg);\n        else:\n            return df;\n\n    @staticmethod\n    def _ReduceMem(df: pd.DataFrame):\n        \"This method reduces memory for numeric columns in the dataframe\";\n\n        numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64', \"uint16\", \"uint32\", \"uint64\"];\n        start_mem = df.memory_usage().sum() / 1024**2;\n\n        for col in df.columns:\n            col_type = df[col].dtypes\n\n            if col_type in numerics:\n                c_min = df[col].min();\n                c_max = df[col].max();\n\n                if \"int\" in str(col_type):\n                    if c_min >= np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min >= np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min >= np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min >= np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n                else:\n                    if c_min >= np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    if c_min >= np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        df[col] = df[col].astype(np.float64)\n\n        end_mem = df.memory_usage().sum() / 1024**2;\n\n        Utils.PrintColor(f\"Start - end memory:- {start_mem:5.2f} - {end_mem:5.2f} Mb\");\n        return df;\n\n    def _MakeModelData(self, df_base, depth_0, depth_1, depth_2, **kwarg):\n        \"\"\"\n        This method aggregates the input tables and joins them to make a single model table for the next steps\n        It converts the final table to a pandas dataframe for the next steps, reduces the memory consumption and selects relevant columns\n        \"\"\";\n\n        for i, df in enumerate(depth_0 + depth_1 + depth_2):\n            df_base = df_base.join(df, how = \"left\", on = \"case_id\", suffix = f\"_{i}\");\n        df = df_base.pipe(self._MakeDtFtre);\n        \n        if self.TrainTest.lower() == \"train\":\n            df = df.collect().to_pandas();\n            df = self._ReduceMem(df.replace([np.inf, -1*np.inf], np.NaN));\n            Utils.PrintColor(f\"---> Selecting training columns by nulls and category unique values\",\n                             color = Fore.CYAN\n                             );\n\n            drop_cols = [];\n            null_cols = df.drop(columns = [self.target], errors = \"ignore\").isna().mean();\n            drop_cols.extend(null_cols.loc[null_cols >= self.null_cutoff].index.to_list());\n\n            obj_cols = df.select_dtypes(include = \"object\").columns;\n            for col in obj_cols:\n                if df[col].nunique() >= self.cat_cutoff or df[col].nunique() == 1:\n                    drop_cols.append(col);\n            cat_cols  = [c for c in obj_cols if c not in drop_cols];\n\n            df = df.drop(columns = drop_cols, errors = \"ignore\");\n            df[cat_cols] = df[cat_cols].astype(\"category\");\n\n        else:\n            Utils.PrintColor(f\"---> Selecting all the test set columns to filter later\");\n        return df;\n\n    def XformData(self, display_store: bool = False):\n        \"\"\"\n        This is the cynosure method that aggregates all the inputs and prepares the FE dataset for the model training/ submission\n        \"\"\";\n\n        data_store = \\\n         {\"df_base\" : self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_base.parquet\"), depth = 0, isSingle = True),\n\n          \"depth_0\" : [self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_static_cb_0.parquet\"), depth = 0, isSingle = True),\n                       self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_static_0_*.parquet\"), depth = 0, isSingle = False)\n                      ],\n\n          \"depth_1\": [self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_applprev_1_*.parquet\"), depth = 1, isSingle = False),\n                      self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_tax_registry_a_1.parquet\"), depth = 1, isSingle = True),\n                      self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_tax_registry_b_1.parquet\"), depth = 1, isSingle = True),\n                      self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_tax_registry_c_1.parquet\"), depth = 1, isSingle = True),\n                      self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_credit_bureau_b_1.parquet\"), depth = 1, isSingle = True),\n                      self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_other_1.parquet\"), depth = 1, isSingle = True),\n                      self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_person_1.parquet\"), depth = 1, isSingle = True),\n                      self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_deposit_1.parquet\"), depth = 1, isSingle = True),\n                      self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_debitcard_1.parquet\"), depth = 1, isSingle = True)\n                      ],\n\n          \"depth_2\": [self._PreProcessIpTbl(os.path.join(self.path, f\"{self.TrainTest}_credit_bureau_b_2.parquet\"), depth = 2, isSingle = True)]\n          };\n\n        if display_store:\n            Utils.PrintColor(\"\\n---> Data store\\n\", color = Fore.CYAN);\n            pprint(data_store, width = 200, depth = 3, indent = 5);\n\n        df = self._MakeModelData(**data_store);\n        del data_store;\n\n        if self.TrainTest.lower() == \"train\":\n            Utils.PrintColor(f\"\\n---> {self.TrainTest.capitalize()} set details = {df.shape} | {df.memory_usage().sum()/ 10**6 :,.2f} Mb\\n\",\n                         color = Fore.CYAN\n                         );\n            Utils.PrintColor(\"\\n---> Train set columns\\n\");\n            with np.printoptions(linewidth = 160):\n                pprint(np.array(df.drop(columns = [self.target], errors = \"ignore\").columns));\n\n                Utils.PrintColor(\"\\n---> Train set category columns\\n\", color = Fore.CYAN);\n                pprint(np.array(df.select_dtypes(include = \"category\").columns));\n        else:\n            pass;\n        return df;\n\nUtils.PrintColor(Utils.CleanMemory());","metadata":{"id":"hucjcQeFfXER","outputId":"75109759-7856-4f82-c918-0667d8e9dbd1","execution":{"iopub.status.busy":"2024-03-23T14:01:51.799809Z","iopub.execute_input":"2024-03-23T14:01:51.800195Z","iopub.status.idle":"2024-03-23T14:01:51.996755Z","shell.execute_reply.started":"2024-03-23T14:01:51.800165Z","shell.execute_reply":"2024-03-23T14:01:51.995695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"4\"></a>\n# <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:white; font-size:120%; text-align:left;padding:3.0px; background: #b32d00; border-bottom: 8px solid black\" > MODEL INFERENCING<br> <div>","metadata":{"id":"-htSH8n28ORa"}},{"cell_type":"markdown","source":"<a id=\"4.1\"></a>\n## <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:black; font-size:120%; text-align:left;padding:3.0px; background:  #e0ebeb; border-bottom: 8px solid #b32d00\" > ML MODELS<br><div>","metadata":{}},{"cell_type":"code","source":"%%time\n\nclass VotingModelMaker(BaseEstimator, ClassifierMixin):\n    \"\"\"\n    This class prepares a voting model from the individual fold level contributions\n    Source - https://www.kaggle.com/code/greysky/home-credit-baseline\n    \"\"\";\n\n    def __init__(self, estimators: list):\n        super().__init__()\n        self.estimators = estimators;\n\n    def fit(self, X, y=None):\n        return self;\n\n    def predict(self, X):\n        y_preds = [estimator.predict(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0);\n\n    def predict_proba(self, X):\n        y_preds = [estimator.predict_proba(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0);","metadata":{"execution":{"iopub.status.busy":"2024-03-23T14:01:56.336706Z","iopub.execute_input":"2024-03-23T14:01:56.337618Z","iopub.status.idle":"2024-03-23T14:01:56.346467Z","shell.execute_reply.started":"2024-03-23T14:01:56.337583Z","shell.execute_reply":"2024-03-23T14:01:56.345392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\n# Creating the test set features:-\nsub_fl = pd.read_csv(os.path.join(CFG.path, \"sample_submission.csv\"));\n\n# Creating output dataframe for model predictions:-\nMdl_Preds = pd.DataFrame(index = range(len(sub_fl)));\n\npp = DataXformer(TrainTest   = \"test\",\n                 null_cutoff = CFG.null_cutoff,\n                 cat_cutoff  = CFG.cat_cutoff,\n                 sel_cols    = [],\n                 cat_cols    = [],\n                );\nXtest     = pp.XformData(display_store = False);\n\n# Creating a week number series to be used later:-\nmyweeknum = Xtest.select(pl.col(\"WEEK_NUM\")).collect().to_numpy().flatten();\n_ = Utils.CleanMemory();\n\nwith np.printoptions(linewidth = 160):\n    Utils.PrintColor(f\"\\n\\n---> All files in the input path\\n\");\n    for _, _, files in os.walk(CFG.model_path):\n        pprint(np.array(files));\n        \n# Creating output structure for my models:-     \nMyMdl_Preds = pd.DataFrame(index = sub_fl.index);\n\nfor exp_nb in CFG.all_exp_nb:\n    Utils.PrintColor(f\"---> Current experiment = {exp_nb}\");\n    cat_cols = joblib.load(os.path.join(CFG.model_path, f\"SelCatCols_{exp_nb}.pkl\")).to_list();\n    sel_cols = joblib.load(os.path.join(CFG.model_path, f\"SelCols_{exp_nb}.pkl\")).to_list();\n    \n    Xt = pp._ReduceMem(Xtest.select(sel_cols).\\\n                       collect().\\\n                       to_pandas().\\\n                       drop(columns = [\"WEEK_NUM\", \"case_id\"], errors = \"ignore\")\n                      );\n    Xt[cat_cols] = Xt[cat_cols].astype(\"category\");\n    \n    model_files = \\\n    sorted([f for f in files if f\"{exp_nb}\" in f and (\"LGBM\" in f or \"XGB\" in f) and f.endswith('.model')]);\n    Utils.PrintColor(f\"\\n---> {model_files} | Data shape = {Xt.shape}\", color = Fore.MAGENTA);\n    preds = pd.DataFrame(index = Mdl_Preds.index, columns = model_files);\n    \n    print();\n    for f in model_files:\n        print(f);\n        model    = joblib.load(os.path.join(CFG.model_path, f));\n        preds[f] = Utils.PredictBatch(model, Xt);\n    del Xt;\n    _ = Utils.CleanMemory();\n    \n    try:\n        MyMdl_Preds[exp_nb] = np.average(preds.values, axis=1, weights = CFG.myml_inner_wgt);\n    except:\n        MyMdl_Preds[exp_nb] = np.mean(preds.values, axis=1);\n \nMdl_Preds[\"MyLGBM\"] = np.average(MyMdl_Preds.values, axis=1, weights = CFG.mymdl_wgt);\ndel MyMdl_Preds;\n\nUtils.CleanMemory();","metadata":{"execution":{"iopub.status.busy":"2024-03-23T14:02:00.186289Z","iopub.execute_input":"2024-03-23T14:02:00.187075Z","iopub.status.idle":"2024-03-23T14:02:18.696562Z","shell.execute_reply.started":"2024-03-23T14:02:00.187036Z","shell.execute_reply":"2024-03-23T14:02:18.695433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"4.2\"></a>\n## <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:black; font-size:120%; text-align:left;padding:3.0px; background:  #e0ebeb; border-bottom: 8px solid #b32d00\" > LAMA DENSELIGHT MODEL<br><div>","metadata":{}},{"cell_type":"code","source":"%%time \n\n!pip install --no-index -Uq --find-links=/kaggle/input/lightautoml-038-dependencies lightautoml==0.3.8 -q;\nfrom lightautoml.automl.presets.tabular_presets import TabularAutoML;\nfrom lightautoml.tasks import Task;\nclear_output();\n\nmodel    = joblib.load(\"/kaggle/input/homecreditquality2024ancillary/E2V5_DENSELIGHT.model\");\nsel_cols = joblib.load(\"/kaggle/input/homecreditquality2024ancillary/SelCols_E2V5.pkl\");\ncat_cols = joblib.load(\"/kaggle/input/homecreditquality2024ancillary/SelCatCols_E2V5.pkl\");\nXtest    = pp.XformData(display_store = False);\nXtest    = pp._ReduceMem(Xtest.select(sel_cols).collect().to_pandas());\nXtest[cat_cols] = Xtest[cat_cols].astype(\"category\");\nMdl_Preds[\"LAMA1\"] = Utils.PredictBatch(model, Xtest, prd_proba_req = False);\ndel Xtest;\n_ = Utils.CleanMemory();\n\nprint(\"\\n\\n\\n\");\nmodel    = joblib.load(\"/kaggle/input/homecreditquality2024ancillary/E2V6_DENSELIGHT.model\");\nsel_cols = joblib.load(\"/kaggle/input/homecreditquality2024ancillary/SelCols_E2V6.pkl\");\ncat_cols = joblib.load(\"/kaggle/input/homecreditquality2024ancillary/SelCatCols_E2V6.pkl\");\nXtest    = pp.XformData(display_store = False);\nXtest    = pp._ReduceMem(Xtest.select(sel_cols).collect().to_pandas());\nXtest[cat_cols] = Xtest[cat_cols].astype(\"category\");\nMdl_Preds[\"LAMA2\"] = Utils.PredictBatch(model, Xtest, prd_proba_req = False);\ndel Xtest;\n_ = Utils.CleanMemory(); ","metadata":{"execution":{"iopub.status.busy":"2024-03-23T14:02:29.962691Z","iopub.execute_input":"2024-03-23T14:02:29.963541Z","iopub.status.idle":"2024-03-23T14:06:10.049108Z","shell.execute_reply.started":"2024-03-23T14:02:29.963499Z","shell.execute_reply":"2024-03-23T14:06:10.047950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"4.3\"></a>\n## <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:black; font-size:120%; text-align:left;padding:3.0px; background:  #e0ebeb; border-bottom: 8px solid #b32d00\" > PUBLIC AUTOGLUON MODEL<br><div>","metadata":{}},{"cell_type":"code","source":"%%time \n\n# Source - https://www.kaggle.com/code/batprem/home-credit-automl-inference\n!python -m pip install --no-index --find-links=/kaggle/input/autogluon-pkgs autogluon > /dev/null\n!python -m pip install --no-index --find-links=/kaggle/input/ray-pkgs --upgrade --force-reinstall -q ray==2.6.3\n!pip install --force-reinstall scikit-learn --no-index --find-links=file:///kaggle/input/scikit-learn-1-4-0/ \nclear_output();\n\n!cp /kaggle/usr/lib/home_credit_automl_inference/home_credit_automl_inference.py home_credit_automl_inference.py\n!python home_credit_automl_inference.py\n!mv submission.csv submission_4.csv\n\n_ = Utils.CleanMemory(); ","metadata":{"execution":{"iopub.status.busy":"2024-03-23T14:06:54.179654Z","iopub.execute_input":"2024-03-23T14:06:54.180499Z","iopub.status.idle":"2024-03-23T14:12:23.511687Z","shell.execute_reply.started":"2024-03-23T14:06:54.180463Z","shell.execute_reply":"2024-03-23T14:12:23.510426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"4.4\"></a>\n## <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:black; font-size:120%; text-align:left;padding:3.0px; background:  #e0ebeb; border-bottom: 8px solid #b32d00\" > FINAL PREDICTIONS<br><div>","metadata":{}},{"cell_type":"code","source":"%%time \n\nimport pandas as pd;\nprint(f\"Pandas = {pd.__version__}\");\n\nMdl_Preds[\"AutoGluon\"] = pd.read_csv(\"submission_4.csv\")[\"score\"].values;\ntry:\n    os.remove(\"/kaggle/working/submission_4.csv\");\nexcept:\n    pass;\n\nUtils.PrintColor(f\"\\n---> Model predictions across all options\\n\");\ndisplay(Mdl_Preds.head(10).style.format(precision = 4));\n\nsub_fl[\"score\"] = np.average(Mdl_Preds.values, axis=1, weights = CFG.blend_wgt);\nif CFG.pp_preds:\n    Utils.PrintColor(f\"\\n---> Post-processing predictions using metric hack\\n\", color = Fore.RED);\n    sub_fl = Utils.PostProcessPreds(week_num = myweeknum, sub = sub_fl, shift = CFG.shift);\nelse:\n    pass;\n\ndel Mdl_Preds;\n_ = Utils.CleanMemory();\n\nsub_fl = pl.DataFrame(sub_fl.reset_index());  \nUtils.PrintColor(f\"\\n\\n---> Final submission file\\n\");\ndisplay(sub_fl.head(10));\n\n# Saving the submission file for leaderboard:-\nsub_fl.select([\"case_id\", \"score\"]).write_csv(f\"submission.csv\",);     \n_ = Utils.CleanMemory(); ","metadata":{"execution":{"iopub.status.busy":"2024-03-23T14:12:28.279366Z","iopub.execute_input":"2024-03-23T14:12:28.280539Z","iopub.status.idle":"2024-03-23T14:12:29.015059Z","shell.execute_reply.started":"2024-03-23T14:12:28.280485Z","shell.execute_reply":"2024-03-23T14:12:29.013876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"5\"></a>\n# <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:white; font-size:120%; text-align:left;padding:3.0px; background: #b32d00; border-bottom: 8px solid black\" > PLANNED NEXT STEPS<br> <div>","metadata":{}},{"cell_type":"markdown","source":"<div style= \"font-family: Cambria; letter-spacing: 0px; color:#000000; font-size:110%; text-align:left;padding:3.0px; background: #f2f2f2\" >\n1. We need to understand the data structure first. Exploring through all the files and understanding the columns is key<br>\n2. The importance of a good EDA cannot be described enough in words in such a challenge <br>\n3. Developing a better set of models with better feature choices is key <br>\n4. Understanding the stability metric and incorporating it in the training and inferencing pipeline is also key <br>\n</div>","metadata":{}}]}