{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **PACKAGE INSTALLATIONS**","metadata":{}},{"cell_type":"code","source":"%%writefile req_kaggle.txt\n\nscikit-learn==1.6.1\nxgboost==3.0.1\nlightgbm==4.6.0\nnumpy==1.26.4\nscipy==1.14.1\npolars==1.29.0\npytorch_tabnet\ntabpfn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T06:33:22.467605Z","iopub.execute_input":"2025-05-01T06:33:22.468348Z","iopub.status.idle":"2025-05-01T06:33:22.480363Z","shell.execute_reply.started":"2025-05-01T06:33:22.468307Z","shell.execute_reply":"2025-05-01T06:33:22.479316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a req_colab.txt\n\nxgboost==3.0.1\ncatboost==1.2.7\nnumpy==1.26.4\nlightgbm==4.6.0\nscipy==1.14.1\npolars==1.29.0\ncolorama\ncloudpickle\noptuna\npytorch_tabnet\nscikit-lego\ntabpfn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T06:33:22.482391Z","iopub.execute_input":"2025-05-01T06:33:22.482711Z","iopub.status.idle":"2025-05-01T06:33:22.507734Z","shell.execute_reply.started":"2025-05-01T06:33:22.482687Z","shell.execute_reply":"2025-05-01T06:33:22.506667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile req_ag.txt\n\npolars==1.26.0\ncolorama\noptuna\ncatboost==1.2.7\nautogluon.tabular\nray==2.10.0\ndask","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T06:33:22.508993Z","iopub.execute_input":"2025-05-01T06:33:22.509301Z","iopub.status.idle":"2025-05-01T06:33:22.533422Z","shell.execute_reply.started":"2025-05-01T06:33:22.509275Z","shell.execute_reply":"2025-05-01T06:33:22.532395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a req_lama.txt\n\nlightautoml\ncolorama\npolars==1.26.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T06:33:22.534447Z","iopub.execute_input":"2025-05-01T06:33:22.534739Z","iopub.status.idle":"2025-05-01T06:33:22.559092Z","shell.execute_reply.started":"2025-05-01T06:33:22.534714Z","shell.execute_reply":"2025-05-01T06:33:22.557893Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **IMPORTS**","metadata":{}},{"cell_type":"code","source":"%%writefile -a myimports.py\n\nprint(f\"\\n---> Commencing imports-part1\")\n\nfrom gc import collect\nfrom warnings import filterwarnings\nfilterwarnings('ignore')\nfrom IPython.display import display_html, clear_output\nclear_output()\nimport os, sys, logging, re, joblib, ctypes, shutil, random, torch\nfrom copy import deepcopy\n\nimport xgboost as xgb, lightgbm as lgb, catboost as cb, sklearn as sk, pandas as pd\nprint(f\"---> Sklearn = {sk.__version__}| Pandas = {pd.__version__}\")\ncollect()\n\n# General library imports:-\nfrom warnings import filterwarnings\nfilterwarnings('ignore')\nfrom gc import collect\n\nfrom os import path, walk, getpid\nfrom psutil import Process\nimport re\nfrom collections import Counter\nfrom itertools import product, combinations\n\nimport ctypes\nlibc = ctypes.CDLL(\"libc.so.6\")\n\nfrom IPython.display import display_html, clear_output\nfrom pprint import pprint\nfrom functools import partial\nfrom copy import deepcopy\nimport pandas as pd, numpy as np\nfrom scipy.stats import pearsonr\nimport polars as pl\nimport polars.selectors as cs\n\nfrom warnings import filterwarnings\nfilterwarnings('ignore')\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom colorama import Fore, Style, init\nfrom warnings import filterwarnings\nfilterwarnings('ignore')\nfrom tqdm.notebook import tqdm\n\nprint(f\"---> Imports- part 1 done\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T06:33:22.56124Z","iopub.execute_input":"2025-05-01T06:33:22.561591Z","iopub.status.idle":"2025-05-01T06:33:22.584317Z","shell.execute_reply.started":"2025-05-01T06:33:22.561554Z","shell.execute_reply":"2025-05-01T06:33:22.582367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a myimports.py\n\n# Pipeline specifics:-\nfrom sklearn.preprocessing import *\n\nfrom sklearn.impute import SimpleImputer as SI\nfrom sklearn.model_selection import (RepeatedStratifiedKFold as RSKF,\n                                     StratifiedKFold as SKF,\n                                     StratifiedGroupKFold as SGKF,\n                                     LeavePGroupsOut as LPGO, \n                                     LeaveOneGroupOut as LOGO,\n                                     KFold,\n                                     GroupKFold as GKF,\n                                     RepeatedKFold as RKF,\n                                     PredefinedSplit as PDS,\n                                     cross_val_score,\n                                     cross_val_predict,\n                                    )\nfrom sklearn.inspection import permutation_importance\nfrom sklearn.feature_selection import VarianceThreshold as VT\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.pipeline import Pipeline, make_pipeline\nfrom sklearn.base import BaseEstimator, TransformerMixin, clone\nfrom sklearn.compose import ColumnTransformer, make_column_selector\n\n# ML Model training:-\nfrom sklearn.metrics import *\n\nfrom xgboost import QuantileDMatrix, XGBClassifier as XGBC, XGBRegressor as XGBR\nfrom lightgbm import log_evaluation, early_stopping, LGBMClassifier as LGBMC, LGBMRegressor as LGBMR\nfrom catboost import CatBoostClassifier as CBC, Pool, CatBoostRegressor as CBR\nfrom sklearn.ensemble import HistGradientBoostingClassifier as HGBC, RandomForestClassifier as RFC\nfrom sklearn.ensemble import HistGradientBoostingRegressor as HGBR, RandomForestRegressor as RFR\nfrom sklearn.ensemble import VotingRegressor as VR, VotingClassifier as VC\nfrom sklearn.linear_model import LogisticRegression as LRC, Ridge, Lasso\nfrom sklearn.neighbors import KNeighborsClassifier as KNNC, KNeighborsRegressor as KNNR\n\n# TabNet models\nfrom pytorch_tabnet.tab_model import (TabNetRegressor as TNR, TabNetClassifier as TNC)\n\n# TabPFN models\nfrom tabpfn import TabPFNClassifier as TPFNC\n\n# Ensemble and tuning:-\nimport optuna\nfrom optuna import Trial, trial, create_study\nfrom optuna.pruners import HyperbandPruner\nfrom optuna.samplers import TPESampler, CmaEsSampler\n\n# Setting rc parameters in seaborn for plots and graphs-\nsns.set({\"axes.facecolor\"       : \"white\",\n         \"figure.facecolor\"     : \"#ffffff\",\n         \"axes.edgecolor\"       : \"black\",\n         \"grid.color\"           : '#b0b0b0',\n         \"font.family\"          : ['Cambria'],\n         \"axes.labelcolor\"      : \"#000000\",\n         \"xtick.color\"          : \"#000000\",\n         \"ytick.color\"          : \"#000000\",\n         \"grid.linewidth\"       : 0.50,\n         \"grid.linestyle\"       : \"--\",\n         \"axes.titlecolor\"      : 'maroon',\n         'axes.titlesize'       : 9,\n         'axes.labelweight'     : \"bold\",\n         'legend.fontsize'      : 7.0,\n         'legend.title_fontsize': 7.0,\n         'font.size'            : 7.5,\n         'xtick.labelsize'      : 12.5,\n         'ytick.labelsize'      : 9.0,\n        }\n       )\n\n# Color printing\ndef PrintColor(text: str, color = Fore.BLUE, style = Style.BRIGHT):\n    \"Prints color outputs using colorama using a text F-string\"\n    print(style + color + text + Style.RESET_ALL)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T06:33:22.585417Z","iopub.execute_input":"2025-05-01T06:33:22.585692Z","iopub.status.idle":"2025-05-01T06:33:22.608931Z","shell.execute_reply.started":"2025-05-01T06:33:22.585662Z","shell.execute_reply":"2025-05-01T06:33:22.607625Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a myimports.py\n\nprint(f\"---> Commencing imports-part2\")\noptuna.logging.set_verbosity = optuna.logging.ERROR\noptuna.logging.disable_default_handler()\nprint(f\"---> XGBoost = {xgb.__version__} | LightGBM = {lgb.__version__}\")\n\n##################################################################\n# Customizing logging for LGBM\nclass MyLogger:\n    \"\"\"\n    This class helps to suppress logs in lightgbm and Optuna\n    Source - https://github.com/microsoft/LightGBM/issues/6014\n    \"\"\"\n\n    def init(self, logging_lbl: str):\n        self.logger = logging.getLogger(logging_lbl)\n        self.logger.setLevel(logging.ERROR)\n\n    def info(self, message):\n        pass\n\n    def warning(self, message):\n        pass\n\n    def error(self, message):\n        self.logger.error(message)\n\nl = MyLogger()\nl.init(logging_lbl = \"lightgbm_custom\")\nlgb.register_logger(l)\n\n##################################################################\n# Customizing logging for XGBoost\nfor handler in logging.root.handlers[:]:\n    logging.root.removeHandler(handler)\n\nlogger = logging.getLogger(__name__)\nlogger.setLevel(logging.ERROR)\nformatter = logging.Formatter('%(asctime)s | %(levelname)s | %(message)s')\n\nstdout_handler = logging.StreamHandler(sys.stdout)\nstdout_handler.setLevel(logging.INFO)\nstdout_handler.setFormatter(formatter)\n\nfile_handler = logging.FileHandler(f'xgb_optimize.log')\nfile_handler.setLevel(logging.ERROR)\nfile_handler.setFormatter(formatter)\n\nlogger.addHandler(file_handler)\nlogger.addHandler(stdout_handler)\n\nclass XGBLogging(xgb.callback.TrainingCallback):\n    \"\"\"log train logs to file\"\"\"\n\n    def __init__(self, epoch_log_interval=100):\n        self.epoch_log_interval = epoch_log_interval\n\n    def after_iteration(self, model, epoch:int,\n                        evals_log:xgb.callback.TrainingCallback.EvalsLog\n                        ):\n\n        if self.epoch_log_interval <= 0:\n            pass\n\n        elif (epoch %  self.epoch_log_interval == 0):\n            for data, metric in evals_log.items():\n                for metric_name, log in metric.items():\n                    score = log[-1][0] if isinstance(log[-1], tuple) else log[-1]\n                    logger.info(f\"XGBLogging epoch {epoch} dataset {data} {metric_name} {score}\")\n\n        return False\n\n# Making sklearn pipeline outputs as dataframe:-\nfrom sklearn import set_config\npd.set_option('display.max_columns', 1000)\npd.set_option('display.max_rows', 200)\nprint(f\"---> Imports- part 2 done\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T06:33:22.611076Z","iopub.execute_input":"2025-05-01T06:33:22.611481Z","iopub.status.idle":"2025-05-01T06:33:22.63405Z","shell.execute_reply.started":"2025-05-01T06:33:22.611447Z","shell.execute_reply":"2025-05-01T06:33:22.632936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a myimports.py\n\nprint(f\"---> Seeding everything\")\n\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\n\nseed_everything(2024)\nprint(f\"\\n---> Imports done\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T06:33:22.635095Z","iopub.execute_input":"2025-05-01T06:33:22.635387Z","iopub.status.idle":"2025-05-01T06:33:22.660351Z","shell.execute_reply.started":"2025-05-01T06:33:22.635354Z","shell.execute_reply":"2025-05-01T06:33:22.659245Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **TRAINING ELEMENTS**","metadata":{}},{"cell_type":"code","source":"%%writefile -a myutils.py\n\nclass Utils:\n    \"\"\"\n    This class creates and uses several utility methods to be used across the code\n    \"\"\";\n\n    def __init__(self):\n        pass\n\n    def ScoreMetric(self, ytrue, ypred)-> float:\n        \"\"\"\n        This method calculates the metric for the competition\n        Inputs- ytrue, ypred:- input truth and predictions\n        Output- float:- competition metric\n        \"\"\";\n\n        score, _ = pearsonr(ytrue, ypred)\n        return score\n\n    def pp_preds(self, ypreds : np.ndarray)-> np.ndarray :\n        \"Post-processes the predictions using min-max values from the training data\"\n        return np.clip( np.expm1( ypreds ), a_min = 1, a_max = 314 )\n\n    def CleanMemory(self):\n        \"This method cleans the memory off unused objects and displays the cleaned state RAM usage\"\n\n        collect();\n        libc.malloc_trim(0)\n        pid        = getpid()\n        py         = Process(pid)\n        memory_use = py.memory_info()[0] / 2. ** 30\n        return f\"\\nRAM usage = {memory_use :.4} GB\"\n\nutils = Utils()\ncollect()\nprint()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T06:33:22.661958Z","iopub.execute_input":"2025-05-01T06:33:22.662319Z","iopub.status.idle":"2025-05-01T06:33:22.688803Z","shell.execute_reply.started":"2025-05-01T06:33:22.662294Z","shell.execute_reply":"2025-05-01T06:33:22.687721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a myutils.py\n\nclass AdversarialCVMaker:\n    \"\"\"\n    This class assists in adversarial CV between the train and test data with the below steps-\n\n    1. Consider any classifier as a base model, I prefer any boosted tree model as I don't have to focus too much on preprocessing\n    2. Load the train and test set features\n    3. Make a new target column with 1 for test set occurrances and 0 for train-set\n    4. Classify to predict the test set instances with the features and new target from the step above\n    \n    If the AUC score hovers around 50% (random model), then we can be sure that the train and test set have similar distributions \n    Else, if our model is able to differentiate between the train and test data, then our model is unlikely to generalize as-is.\n    In this case, further adjustments may be necesary \n    \"\"\"\n\n    def __init__(self, n_splits: int = 5) :\n        self.model = \\\n        LGBMC(\n            n_estimators     = 200,\n            learning_rate    = 0.02,\n            max_depth        = 3, \n            colsample_bytree = 0.50,\n            objective        = \"binary\",\n            metric           = \"auc\",\n            random_state     = 42,\n            device           = \"gpu\" if torch.cuda.is_available() else \"cpu\",\n        )\n\n        self.n_splits = n_splits\n\n    @staticmethod\n    def scorer(ytrue, ypreds):\n        return roc_auc_score( ytrue, ypreds )\n\n    def make_cv(\n        self, Xtrain, Xtest, **fit_params,\n        ):\n        \"Fits the model with the auxilary target and calculates the AUC score for the CV\"\n\n        df = \\\n        pd.concat(\n            [Xtrain.assign(**{\"target\" : 0}), \n             Xtest.assign(**{\"target\" : 1}),\n            ], \n            axis=0, ignore_index = True\n        )\n\n        cv     = SKF(n_splits = self.n_splits, random_state = 42, shuffle = True)\n        scores = 0\n        \n        for train_idx, dev_idx in cv.split(df, df[\"target\"]) :\n            Xtr  = df.loc[train_idx].drop(\"target\", axis=1)\n            Xdev = df.loc[dev_idx].drop(\"target\", axis=1)\n            ytr  = df.loc[train_idx, \"target\"]\n            ydev = df.loc[dev_idx, \"target\"]\n\n            cat_cols = list( Xdev.select_dtypes(include = [\"string\", \"category\", \"object\"]).columns )\n\n            if len(cat_cols) > 0 :\n                Xtr[cat_cols]  = Xtr[cat_cols].astype(\"category\")\n                Xdev[cat_cols] = Xdev[cat_cols].astype(\"category\")\n            else:\n                pass\n                \n            model = clone(self.model)\n            model.fit(Xtr, ytr)\n            dev_preds = model.predict_proba(Xdev)[:,1]\n            score = self.scorer(ydev, dev_preds)\n            scores += score\n\n        score = scores / self.n_splits\n\n        PrintColor(\n            f\"\\n---> Overall adversarial CV score = {score :,.4f}\"\n        )\n\n        if score > 0.60 :\n            PrintColor(\n                f\"---> Check for test-train distribution shift\\n\", color = Fore.RED\n            )\n        else:\n            PrintColor(\n                f\"---> Train-test distributions are similar\\n\", color = Fore.GREEN\n            )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T06:33:22.690306Z","iopub.execute_input":"2025-05-01T06:33:22.69066Z","iopub.status.idle":"2025-05-01T06:33:22.716943Z","shell.execute_reply.started":"2025-05-01T06:33:22.690617Z","shell.execute_reply":"2025-05-01T06:33:22.71571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a training.py\n\ndef MakePermImp(\n        method, mdl, X, y, ygrp,\n        myscorer, \n        n_repeats = 2,\n        state = 42,\n        ntop: int = 15,\n        **params,\n):\n    \"\"\"\n    This function makes the permutation importance for the provided model and returns the importance scores for all features\n    \n    Note-\n    myscorer - scikit-learn -> metrics -> make_scorer object with the corresponding eval metric and relevant details\n    \"\"\"\n\n    cv        = PDS(ygrp)\n    n_splits  = ygrp.nunique()\n    drop_cols = [\"Source\", \"id\", \"Id\", \"Label\", \"fold_nb\"]\n\n    for fold_nb, (train_idx, dev_idx) in tqdm(enumerate(cv.split(X, y))):\n        Xtr  = X.iloc[train_idx].drop(drop_cols, axis=1, errors = \"ignore\")\n        Xdev = X.iloc[dev_idx].drop(drop_cols, axis=1, errors = \"ignore\")\n        ytr  = y.loc[Xtr.index]\n        ydev = y.loc[Xdev.index]\n\n        model = clone(mdl)\n        sel_cols = list(Xdev.columns)\n        model.fit(Xtr, ytr)\n\n        imp_ = permutation_importance(model,\n                                      Xdev, ydev,\n                                      scoring = myscorer,\n                                      n_repeats = n_repeats,\n                                      random_state = state,\n                                      )[\"importances_mean\"]\n        imp_ = pd.Series(index = sel_cols, data = imp_)\n\n        display(\n            imp_.\\\n            sort_values(ascending = False).\\\n            head(ntop).\\\n            to_frame().\\\n            transpose().\\\n            style.\\\n            format(formatter = '{:,.3f}').\\\n            background_gradient(\"icefire\", axis=1).\\\n            set_caption(f\"Top {ntop} features\")\n            )\n\n        return imp_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T06:33:22.719108Z","iopub.execute_input":"2025-05-01T06:33:22.719469Z","iopub.status.idle":"2025-05-01T06:33:22.7439Z","shell.execute_reply.started":"2025-05-01T06:33:22.719435Z","shell.execute_reply":"2025-05-01T06:33:22.742816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a training.py\n\nclass ModelTrainer:\n    \"This class trains the provided model on the train-test data and returns the predictions and fitted models\"\n\n    def __init__(\n        self,\n        problem_type   : str   = \"regression\", \n        es             : int   = 100,\n        target         : str   = \"\",\n        metric_lbl     : str   = \"rmse\",\n        orig_req       : bool  = False,\n        orig_all_folds : bool  = False,\n        drop_cols      : list  = [\"Source\", \"id\", \"Id\", \"Label\", \"fold_nb\"],\n        pp_preds       : bool  = False,\n    ):\n        \"\"\"\n        Key parameters-\n        es_iter  - early stopping rounds for boosted trees\n        pp_preds - do you want to post-process predictions (true/ false boolean)\n        \"\"\"\n\n        self.problem_type   = problem_type\n        self.es_iter        = es\n        self.target         = target\n        self.drop_cols      = drop_cols + [self.target]\n        self.metric_lbl     = metric_lbl\n        self.orig_req       = orig_req\n        self.orig_all_folds = orig_all_folds\n        self.pp_preds       = pp_preds\n\n        if self.metric_lbl == \"rmse\" :\n            self.ScoreMetric = lambda x,y : root_mean_squared_error( x, self.PostProcessPreds( y))\n        elif self.metric_lbl == \"mae\" :\n            self.ScoreMetric = lambda x,y : mean_absolute_error( x, self.PostProcessPreds( y))\n        elif self.metric_lbl == \"accuracy\" :\n            self.ScoreMetric = lambda x,y : accuracy_score( np.uint8( x ), np.uint8(self.PostProcessPreds( y ) ) )\n        elif self.metric_lbl == \"auc\" :\n            self.ScoreMetric = lambda x,y : roc_auc_score(  x , self.PostProcessPreds( y )  )\n        elif self.metric_lbl == \"kappa\" :\n            self.ScoreMetric = lambda x,y : cohen_kappa_score( np.uint8( x ), np.uint8(self.PostProcessPreds( y ) ), \n                                                              weights = \"quadratic\" \n                                                             )\n        elif self.metric_lbl == \"rmsle\" :\n            self.ScoreMetric = lambda x,y : root_mean_squared_log_error( x, self.PostProcessPreds( y))\n        else:\n            self.ScoreMetric = utils.ScoreMetric\n\n    def PlotFtreImp(\n        self, \n        ftreimp: pd.Series, \n        method: str,\n        ntop: int = 50,\n        title_specs: dict = {'fontsize': 12,'fontweight' : 'bold','color': '#992600'},\n        **params,\n    ):\n        \"This function plots the feature importances for the model provided\"\n\n        print()\n        \n        with sns.axes_style(\"white\"):\n            fig, ax = plt.subplots(1, 1, figsize = (25, 7.5))\n    \n            ftreimp.sort_values(ascending = False).\\\n            head(ntop).\\\n            plot.bar(ax = ax, color = \"#1285c7\")\n            ax.set_title(\n                f\"Feature Importances - {method}\", \n                **title_specs\n            )\n    \n            plt.tight_layout()\n            plt.show()\n        print()\n\n    def PostProcessPreds(self, ypred):\n        \"This method post-processes predictions optionally\"\n        return ypred\n            \n    def LoadData(\n            self, X, y, Xtest,\n            train_idx : list = [],\n            dev_idx   : list = [],\n            ):\n        \"This method loads the train and test data for the model fold using/ not using the original data\"\n\n        try:\n            mysrc = X[\"Source\"]\n        except:\n            X[\"Source\"] , Xtest[\"Source\"] = (\"Competition\", \"Competition\")\n\n        if self.orig_req == False:\n            Xtr  = X.iloc[train_idx].query(\"Source == 'Competition'\").drop(self.drop_cols, axis=1, errors = \"ignore\")\n            ytr  = y.iloc[Xtr.index]\n            Xdev = X.iloc[dev_idx].query(\"Source == 'Competition'\").drop(self.drop_cols, axis=1, errors = \"ignore\")\n            ydev = y.iloc[Xdev.index]\n\n        elif self.orig_req == True and self.orig_all_folds == True:\n            Xtr  = X.iloc[train_idx].query(\"Source == 'Competition'\").drop(self.drop_cols, axis=1, errors = \"ignore\")\n            ytr  = y.iloc[Xtr.index]\n            Xdev = X.iloc[dev_idx].query(\"Source == 'Competition'\").drop(self.drop_cols, axis=1, errors = \"ignore\")\n            ydev = y.iloc[Xdev.index]\n\n            orig_x = X.query(\"Source == 'Original'\")[Xtr.columns]\n            orig_y = y.iloc[orig_x.index]\n\n            Xtr = pd.concat([Xtr, orig_x], axis = 0, ignore_index = True)\n            ytr = pd.concat([ytr, orig_y], axis = 0, ignore_index = True)\n\n        elif self.orig_req == True and self.orig_all_folds == False:\n            Xtr  = X.iloc[train_idx].drop(self.drop_cols, axis=1, errors = \"ignore\")\n            ytr  = y.iloc[Xtr.index]\n            Xdev = X.iloc[dev_idx].query(\"Source == 'Competition'\").drop(self.drop_cols, axis=1, errors = \"ignore\")\n            ydev = y.iloc[Xdev.index]\n\n        Xt = Xtest[Xdev.columns]\n\n        print(f\"\\n---> Shapes = {Xtr.shape} {ytr.shape} -- {Xdev.shape} {ydev.shape} -- {Xt.shape}\")\n        return (Xtr, ytr, Xdev, ydev, Xt)\n    \n    def MakePreds(self, X, fitted_model):\n        \"This method creates the model predictions based on the model provided, with optional post-processing\"\n\n        if self.problem_type == \"regression\":\n            if isinstance(fitted_model, (TNC, TNR)) == True:\n                return self.PostProcessPreds(fitted_model.predict(X.to_numpy()).flatten())\n            else:\n                return self.PostProcessPreds(fitted_model.predict(X))\n                \n        elif self.problem_type == \"binary\":\n            if isinstance(fitted_model, (TNC, TNR)) == True:\n                return self.PostProcessPreds(fitted_model.predict_proba(X.to_numpy()[:,1]).flatten())\n            else:\n                return self.PostProcessPreds(fitted_model.predict_proba(X)[:, 1])\n                \n        elif self.problem_type == \"multiclass\":\n            if isinstance(fitted_model, (TNC, TNR)) == True:\n                return self.PostProcessPreds(fitted_model.predict_proba(X.to_numpy()))\n            else:\n                return self.PostProcessPreds(fitted_model.predict_proba(X))\n\n    def MakeOrigPreds(\n            self, orig: pd.DataFrame, fitted_models: list, n_splits : int, ygrp: pd.Series,\n            ):\n        \"This method creates the original data predictions separately only if required\"\n\n        if self.orig_req == False:\n            orig_preds = 0\n\n        elif self.orig_req == True and self.orig_all_folds == True:\n            orig_preds = 0\n            df = orig.drop(self.drop_cols, axis = 1, errors = \"ignore\")\n\n            for fitted_model in fitted_models:\n                orig_preds = orig_preds + (self.MakePreds(df, fitted_model) / n_splits)\n\n        elif self.orig_req == True and self.orig_all_folds == False:\n            len_orig   = orig.shape[0]\n            orig.index = range(len_orig)\n            orig_ygrp  = ygrp[-1 * len_orig:]\n            orig_ygrp.index = range(len_orig)\n            \n            orig_preds = np.zeros(len_orig)\n            for fold_nb, fitted_model in enumerate(fitted_models):\n                df = \\\n                orig.iloc[orig_ygrp.loc[orig_ygrp == fold_nb].index].\\\n                drop(self.drop_cols, axis=1, errors = \"ignore\")\n                \n                orig_preds[df.index] = self.MakePreds(df, fitted_model)\n                del df\n        return orig_preds\n\n    def MakeOfflineModel(\n        self, X, y, ygrp, Xtest, mdl, method,\n        test_preds_req   : bool = True,\n        ftreimp_plot_req : bool = True,\n        ntop             : int  = 50,\n        **params,\n    ):\n        \"\"\"\n        This method trains the provided model on the dataset and cross-validates appropriately\n\n        Inputs-\n        X, y, ygrp       - training data components (Xtrain, ytrain, fold_nb)\n        Xtest            - test data (optional)\n        model            - model object for training\n        method           - model method label\n        test_preds_req   - boolean flag to extract test set predictions\n        ftreimp_plot_req - boolean flag to plot tree feature importances\n        ntop             - top n features for feature importances plot\n\n        Returns-\n        oof_preds, test_preds - prediction arrays\n        fitted_models         - fitted model list for test set\n        ftreimp               - feature importances across selected features\n        mdl_best_iter         - model average best iteration across folds\n        \"\"\"\n\n        oof_preds     = np.zeros(len(X.loc[X.Source == \"Competition\"]))\n        orig_preds    = np.zeros(len(X.loc[X.Source == \"Original\"]))\n        test_preds    = []\n        mdl_best_iter = []\n        ftreimp       = 0\n\n        scores, tr_scores, fitted_models = [], [], []\n\n        if self.orig_req == True:\n            cv = PDS(ygrp)\n        elif self.orig_req == False:\n            X  = X.loc[X.Source == \"Competition\"]\n            y  = y.iloc[X.index]\n            cv = PDS(ygrp.iloc[0 : len(X)])\n\n        n_splits = ygrp.nunique()\n\n        for fold_nb, (train_idx, dev_idx) in tqdm(enumerate(cv.split(X, y))):\n            Xtr, ytr, Xdev, ydev, Xt = \\\n            self.LoadData(X, y, Xtest, train_idx, dev_idx)\n\n            model = clone(mdl)\n\n            if \"CB\" in method and self.es_iter > 0:\n                model.fit(Xtr, ytr,\n                          eval_set = [(Xdev, ydev)],\n                          verbose = 0,\n                          early_stopping_rounds = self.es_iter,\n                          )\n                best_iter = model.get_best_iteration()\n\n            elif \"LGB\" in method and self.es_iter > 0:\n                model.fit(Xtr, ytr,\n                          eval_set = [(Xdev, ydev)],\n                          callbacks = [log_evaluation(0),\n                                       early_stopping(stopping_rounds = self.es_iter, verbose = False,),\n                                       ],\n                          )\n                best_iter = model.best_iteration_\n\n            elif \"XGB\" in method and self.es_iter > 0:\n                model.fit(Xtr, ytr,\n                          eval_set = [(Xdev, ydev)],\n                          verbose  = 0,\n                          )\n                best_iter = model.best_iteration\n\n            else:\n                model.fit(Xtr, ytr)\n                best_iter = -1\n\n            fitted_models.append(model)\n\n            try:\n                ftreimp += model.feature_importances_\n            except:\n                try:\n                    ftreimp += model.coef_.flatten()\n                except:\n                    pass\n\n            try:\n                ftreimp += model[\"M\"].feature_importances_\n            except:\n                try:\n                    ftreimp += model[\"M\"].coef_.flatten()\n                except:\n                    pass               \n            \n            dev_preds = self.MakePreds(Xdev, model)\n            oof_preds[Xdev.index] = dev_preds\n\n            train_preds  = self.MakePreds(Xtr, model)\n            tr_score     = self.ScoreMetric(ytr.values.flatten(), train_preds)\n            score        = self.ScoreMetric(ydev.values.flatten(), dev_preds)\n\n            scores.append(score)\n            tr_scores.append(tr_score)\n\n            nspace = 15 - len(method) - 2 if fold_nb <= 9 else 15 - len(method) - 1\n\n            if self.es_iter > 0 :\n                PrintColor(f\"{method} Fold{fold_nb} {' ' * nspace} OOF = {score:.6f} | Train = {tr_score:.6f} | Iter = {best_iter:,.0f} \")\n            else:\n                PrintColor(f\"{method} Fold{fold_nb} {' ' * nspace} OOF = {score:.6f} | Train = {tr_score:.6f} \")\n                \n            mdl_best_iter.append(best_iter)\n\n            if test_preds_req:\n                test_preds.append(self.MakePreds(Xt, model))\n            else:\n                pass\n\n        test_preds    = np.mean(np.stack(test_preds, axis = 1), axis=1)\n        ftreimp       = pd.Series(ftreimp, index = Xdev.columns)\n        mdl_best_iter = np.uint16(np.amax(mdl_best_iter))\n\n        if ftreimp_plot_req :\n            print()\n            self.PlotFtreImp(ftreimp, method = method, ntop = ntop,)\n        else:\n            pass\n\n        PrintColor(f\"\\n---> {np.mean(scores):.6f} +- {np.std(scores):.6f} | OOF\", color = Fore.RED)\n        PrintColor(f\"---> {np.mean(tr_scores):.6f} +- {np.std(tr_scores):.6f} | Train\", color = Fore.RED)\n\n        if self.es_iter <= 0 :\n            pass\n        else:\n            PrintColor(\n                f\"---> Max best iteration = {mdl_best_iter :,.0f}\",\n                color = Fore.RED\n            )\n\n        if self.orig_req:\n            print(f\"---> Collecting original predictions\")\n            orig_preds = self.MakeOrigPreds(X.loc[X.Source == \"Original\"],\n                                            fitted_models,\n                                            n_splits,\n                                            ygrp,\n                                            )\n            oof_preds = np.concatenate([oof_preds, orig_preds], axis= 0)\n        else:\n            pass\n        return (fitted_models, oof_preds, test_preds, ftreimp, mdl_best_iter)\n\n    def MakeOnlineModel(\n        self, X, y, Xtest, model, method,\n        test_preds_req : bool = False,\n    ):\n        \"This method refits the model on the complete train data and returns the model fitted object and predictions\"\n\n        try:\n            model.early_stopping_rounds = None\n        except:\n            pass\n\n        if \"TN\" in method:\n            model.fit(\n                X.to_numpy(), y.to_numpy().reshape(-1,1),\n                max_epochs  = 100,\n                batch_size  = 128,\n                virtual_batch_size = 64,\n                )\n        else:\n            try:\n                model.fit(X, y, verbose = 0)\n            except:\n                model.fit(X, y,)\n\n        oof_preds  = self.MakePreds(X, model)\n        if test_preds_req:\n            test_preds = self.MakePreds(Xtest[X.columns], model)\n        else:\n            test_preds = 0\n            \n        return (model, oof_preds, test_preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T06:33:22.745134Z","iopub.execute_input":"2025-05-01T06:33:22.745491Z","iopub.status.idle":"2025-05-01T06:33:22.777849Z","shell.execute_reply.started":"2025-05-01T06:33:22.745463Z","shell.execute_reply":"2025-05-01T06:33:22.77678Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **HILL CLIMBER**","metadata":{}},{"cell_type":"code","source":"%%writefile -a training.py\n\nclass HillClimber:\n    \"This class develops the Hill Climber algorithm for the provided datasets\"\n\n    def __init__(self):\n        self.ScoreMetric = utils.ScoreMetric\n\n    def DoHillClimb(\n        self, \n        target:str,\n        direction:str,\n        cutoff:float,\n        neg_wgt:str,\n        OOF_Preds: pd.DataFrame,\n        Mdl_Preds: pd.DataFrame,\n        y: pd.Series,\n        **kwargs\n    ):\n        \"\"\"\n        This method performs hill-climbing on the OOF and Test predictions dataset and returns the below-\n        1. OOF ensemble predictions\n        2. Test set predictions\n        3. Score dataframe (with scores in sort-order)\n        \"\"\"\n\n        oof_df     = OOF_Preds\n        test_preds = Mdl_Preds\n    \n        # Scoring the individual models:-\n        Scores = pd.DataFrame(index = oof_df.columns, columns = ['Score'])\n    \n        for col in oof_df.columns:\n            Scores.at[col, 'Score'] = self.ScoreMetric(y, oof_df[col].values.flatten())\n    \n        # Sorting scores\n        Scores.sort_values(\n            by= 'Score',\n            ascending = [True if direction == 'minimize' else False],\n            inplace = True,\n        )\n    \n        PrintColor(f\"\\n----- Data preparation: ------ \\n\");\n        display(\n            Scores.\n            transpose().\n            style.\n            format(precision = 5)\n            )\n    \n        PrintColor(f\"\\n ----- Initiating hill-climb ----- \\n\");\n        STOP = False\n        current_best_ensemble   = oof_df.iloc[:,0]\n        current_best_test_preds = test_preds.iloc[:,0]\n        MODELS                  = oof_df.iloc[:,1:]\n    \n        if neg_wgt == \"Y\":\n            weight_range = np.arange(-0.5,0.51,0.01);\n        else:\n            weight_range = np.arange(0.01,0.51,0.01);\n    \n        history = [self.ScoreMetric(y, current_best_ensemble)]\n    \n        i=0\n    \n        # Hill climbing algorithm:-\n        while not STOP:\n            i+=1\n    \n            potential_new_best_cv_score = self.ScoreMetric(y, current_best_ensemble)\n            k_best, wgt_best = None, None\n    \n            for k in MODELS:\n                for wgt in weight_range:\n                    potential_ensemble = (1- wgt) * current_best_ensemble + wgt * MODELS[k]\n                    cv_score = self.ScoreMetric(y, potential_ensemble)\n    \n                    if direction == 'minimize':\n                        if cv_score < potential_new_best_cv_score:\n                            potential_new_best_cv_score, k_best, wgt_best = cv_score, k, wgt\n    \n                    if direction == 'maximize':\n                        if cv_score > potential_new_best_cv_score:\n                            potential_new_best_cv_score, k_best, wgt_best = cv_score, k, wgt\n    \n            if k_best is not None:\n                current_best_ensemble   = (1- wgt_best) * current_best_ensemble + wgt_best * MODELS[k_best]\n                current_best_test_preds = (1- wgt_best) * current_best_test_preds + wgt_best * test_preds[k_best]\n                MODELS.drop(k_best, axis=1, inplace=True)\n    \n                if MODELS.shape[1]==0:  STOP = True\n    \n                num_space = 50 - len(k_best) if i <= 9 else 49 - len(k_best)\n                PrintColor(f\" {i}.{k_best} {' ' * num_space} Weight = {wgt_best: .4f} {' ' * 5} Score = {potential_new_best_cv_score:.6f}\",\n                           color = Fore.CYAN\n                          )\n                del num_space\n    \n                history.append(potential_new_best_cv_score)\n    \n            else:\n                STOP = True\n    \n        return (current_best_ensemble, current_best_test_preds, Scores)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T06:33:22.780102Z","iopub.execute_input":"2025-05-01T06:33:22.780517Z","iopub.status.idle":"2025-05-01T06:33:22.803517Z","shell.execute_reply.started":"2025-05-01T06:33:22.780494Z","shell.execute_reply":"2025-05-01T06:33:22.802249Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **LAMA IMPORTS**","metadata":{}},{"cell_type":"code","source":"%%writefile -a myimports_lama.py\n\nfrom warnings import filterwarnings\nfilterwarnings('ignore')\n\nimport os, re, joblib, tempfile, ctypes\nfrom os import path, walk, getpid\nfrom psutil import Process\nfrom collections import Counter\nfrom itertools import product\nfrom gc import collect\n\nlibc = ctypes.CDLL(\"libc.so.6\")\n\nfrom IPython.display import display_html, clear_output\nfrom pprint import pprint\nfrom functools import partial\nfrom copy import deepcopy\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom colorama import Fore, Style, init\nfrom tqdm.notebook import tqdm\n\n# Essential DS libraries\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport polars.selectors as cs\nfrom sklearn.metrics import *\n\nfrom sklearn.model_selection import PredefinedSplit as PDS\nfrom sklearn.preprocessing import RobustScaler\nimport torch\nimport torch.nn as nn\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\n\n# LightAutoML presets, task and report generation\nfrom lightautoml.automl.presets.tabular_presets import TabularAutoML\nfrom lightautoml.tasks import Task\n\n# Color printing\ndef PrintColor(text: str, color = Fore.BLUE, style = Style.BRIGHT):\n    \"Prints color outputs using colorama using a text F-string\"\n    print(style + color + text + Style.RESET_ALL)\n\nprint(f\"---> CUDA available = {torch.cuda.is_available()}\\n\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T06:33:22.804338Z","iopub.execute_input":"2025-05-01T06:33:22.804714Z","iopub.status.idle":"2025-05-01T06:33:22.834016Z","shell.execute_reply.started":"2025-05-01T06:33:22.804684Z","shell.execute_reply":"2025-05-01T06:33:22.832856Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **AUTOGLUON IMPORTS**","metadata":{}},{"cell_type":"code","source":"%%writefile -a myimports_ag.py\n\nimport numpy as np, pandas as pd\nimport polars as pl\nimport polars.selectors as cs\nimport re, os, joblib, logging\nfrom gc import collect\n\nfrom IPython.display import display_html, clear_output\nfrom pprint import pprint\nfrom tqdm.notebook import tqdm\nfrom colorama import Fore, Back, Style\nfrom os import path, walk, getpid\nfrom psutil import Process\nimport ctypes\nlibc = ctypes.CDLL(\"libc.so.6\")\n\nfrom warnings import filterwarnings\nfilterwarnings(\"ignore\")\n\nfrom sklearn.model_selection import StratifiedKFold as SKF, GroupKFold as GKF\nfrom sklearn.metrics import *\nfrom autogluon.tabular import TabularPredictor, TabularDataset\nfrom autogluon.core.metrics import make_scorer as ag_make_scorer\n\n# Color printing\ndef PrintColor(text: str, color = Fore.BLUE, style = Style.BRIGHT):\n    \"Prints color outputs using colorama using a text F-string\"\n    print(style + color + text + Style.RESET_ALL)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T06:33:22.835329Z","iopub.execute_input":"2025-05-01T06:33:22.835694Z","iopub.status.idle":"2025-05-01T06:33:22.860439Z","shell.execute_reply.started":"2025-05-01T06:33:22.835664Z","shell.execute_reply":"2025-05-01T06:33:22.859203Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **PREPROCESSOR**","metadata":{}},{"cell_type":"code","source":"%%writefile -a mypp.py\n\nclass Preprocessor():\n    \"\"\"\n    This class aims to do the below-\n    1. Read the datasets\n    2. In this case, we need to process the original data target column to be compatible with the competition dataset\n    3. Check information and description\n    4. Check unique values and nulls\n    5. Collate starting features \n    \"\"\"\n    \n    def __init__(self):\n\n        self.train             = pd.read_csv(os.path.join(CFG.ip_path,\"train.csv\"), index_col = 'id') \n        self.test              = pd.read_csv(os.path.join(CFG.ip_path ,\"test.csv\"), index_col = 'id')\n        self.target            = CFG.target \n        \n        self.conjoin_orig_data = True if CFG.nb_orig > 0 else False\n        self.dtl_preproc_req   = CFG.dtl_preproc_req\n        self.test_req          = CFG.test_req\n        self.cv                = cv_selector[CFG.mdlcv_mthd]\n         \n        self.original            = pd.read_csv(CFG.orig_path , index_col = 'User_ID').drop_duplicates()\n        self.original.index      = range(len(self.original))\n        self.original.index.name = \"id\"    \n        self.original            = self.original.rename(columns = {\"Gender\":\"Sex\"})\n        self.original            = self.original[self.train.columns]\n\n        self.sub_fl = pd.read_csv(os.path.join(CFG.ip_path, \"sample_submission.csv\"))\n        PrintColor(f\"Data shapes - train-test-original | {self.train.shape} {self.test.shape} {self.original.shape}\")\n        \n        for tbl in [self.train, self.original, self.test]:\n            obj_cols      = tbl.select_dtypes(include = [\"object\", \"category\"]).columns\n            tbl.columns   = tbl.columns.str.replace(r\"\\(|\\)|\\.|\\?|/|\\s+\",\"\", regex = True)\n            \n    def _VisualizeDF(self):\n        \"This method visualizes the heads for the train, test and original data\"\n        \n        PrintColor(f\"\\nTrain set head\", color = Fore.CYAN)\n        display(self.train.head(5).style.format(precision = 3))\n        \n        PrintColor(f\"\\nTest set head\", color = Fore.CYAN)\n        display(self.test.head(5).style.format(precision = 3))\n        \n        PrintColor(f\"\\nOriginal set head\", color = Fore.CYAN)\n        display(self.original.head(5).style.format(precision = 3))\n              \n    def _AddSourceCol(self):\n        self.train['Source']    = \"Competition\"\n        self.test['Source']     = \"Competition\"\n        self.original['Source'] = 'Original'\n        \n        self.strt_ftre = self.test.columns\n        return self\n          \n    def _CollateInfoDesc(self):\n        if self.dtl_preproc_req :\n            PrintColor(f\"\\n{'-' * 20} Information and description {'-' * 20}\\n\", color = Fore.MAGENTA);\n\n            # Creating dataset information and description:\n            for lbl, df in {'Train': self.train, 'Test': self.test, 'Original': self.original}.items():\n                PrintColor(f\"\\n{lbl} description\\n\");\n                display(df.describe(percentiles= [0.05, 0.25, 0.50, 0.75, 0.9, 0.95, 0.99]).\\\n                        transpose().\\\n                        drop(columns = ['count'], errors = 'ignore').\\\n                        drop([self.target], axis=0, errors = 'ignore').\\\n                        style.format(formatter = '{:,.2f}').\\\n                        background_gradient(cmap = 'Blues')\n                       );\n\n                PrintColor(f\"\\n{lbl} information\\n\");\n                display(df.info());\n                collect();\n        return self;\n    \n    def _CollateUnqNull(self):\n        \n        if self.dtl_preproc_req :\n            PrintColor(f\"\\nUnique and null values\\n\")\n            _ = pd.concat([self.train[self.strt_ftre].nunique(), \n                           self.test[self.strt_ftre].nunique(), \n                           self.original[self.strt_ftre].nunique(),\n                           self.train[self.strt_ftre].isna().sum(axis=0),\n                           self.test[self.strt_ftre].isna().sum(axis=0),\n                           self.original[self.strt_ftre].isna().sum(axis=0)\n                          ], \n                          axis=1)\n            _.columns = ['Train_Nunq', 'Test_Nunq', 'Original_Nunq', \n                         'Train_Nulls', 'Test_Nulls', 'Original_Nulls'\n                        ]\n            display(_.T.style.background_gradient(cmap = 'Blues', axis=1).\\\n                    format(formatter = '{:,.0f}')\n                   )\n            \n        return self\n       \n    def _ConjoinTrainOrig(self):\n        if self.conjoin_orig_data :\n            PrintColor(f\"\\n\\nTrain shape before conjoining with original = {self.train.shape}\")\n            train = pd.concat([self.train] + [self.original] * CFG.nb_orig, \n                              axis=0, \n                              ignore_index = True\n                             )\n            PrintColor(f\"Train shape after conjoining with original= {train.shape}\")\n\n            train.index = range(len(train))\n            train.index.name = 'id'\n\n        else:\n            PrintColor(f\"\\nWe are using the competition training data only\")\n            train = self.train\n        return train\n       \n    def DoPreprocessing(self):\n        self._VisualizeDF()\n        self._AddSourceCol()\n        self._CollateInfoDesc()\n        self._CollateUnqNull()\n        self.train = self._ConjoinTrainOrig()\n\n        self.train = self.train.dropna(subset = [self.target])\n        self.train.index = range(len(self.train))\n        \n        self.cat_cols  = \\\n        list(\n            self.test.drop(\"Source\", axis=1).select_dtypes([\"object\", \"string\", \"category\"]).columns\n        )\n        self.cont_cols = \\\n        [c for c in self.strt_ftre if c not in self.cat_cols + ['Source']]\n        \n        return self \n            ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T06:39:09.217068Z","iopub.execute_input":"2025-05-01T06:39:09.21738Z","iopub.status.idle":"2025-05-01T06:39:09.226184Z","shell.execute_reply.started":"2025-05-01T06:39:09.217359Z","shell.execute_reply":"2025-05-01T06:39:09.225041Z"}},"outputs":[],"execution_count":null}]}