{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":197529256,"sourceType":"kernelVersion"}],"dockerImageVersionId":30761,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **IMPORTS**","metadata":{}},{"cell_type":"code","source":"%%capture \n\n!pip install -q lightgbm==4.5.0 --no-index --find-links=/kaggle/input/cmi2024-packages-v1/MLPackages\n!pip install -q polars==1.7.1 --no-index --find-links=/kaggle/input/cmi2024-packages-v1/Polars171","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-22T14:02:02.796063Z","iopub.execute_input":"2024-09-22T14:02:02.796441Z","iopub.status.idle":"2024-09-22T14:02:32.506762Z","shell.execute_reply.started":"2024-09-22T14:02:02.796404Z","shell.execute_reply":"2024-09-22T14:02:32.505020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a myimports.py\n\nprint(f\"\\n---> Commencing imports\")\n\nfrom gc import collect\nfrom warnings import filterwarnings\nfilterwarnings('ignore')\nfrom IPython.display import display_html, clear_output\nclear_output()\nimport os, sys, logging, re, joblib, ctypes, shutil\nfrom copy import deepcopy\n\n# General library imports:-\nfrom os import path, walk, getpid\nfrom psutil import Process\nfrom collections import Counter\nfrom itertools import product\nimport ctypes\nlibc = ctypes.CDLL(\"libc.so.6\")\n\nfrom IPython.display import display_html, clear_output\nfrom pprint import pprint\nfrom functools import partial\nfrom copy import deepcopy\nimport pandas as pd, numpy as np\nfrom scipy.optimize import minimize\nfrom numpy.typing import ArrayLike, NDArray\nimport polars as pl\nimport polars.selectors as cs\nfrom polars.testing import assert_frame_equal\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom colorama import Fore, Style, init\nfrom tqdm.notebook import tqdm","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-22T14:02:32.508590Z","iopub.execute_input":"2024-09-22T14:02:32.509026Z","iopub.status.idle":"2024-09-22T14:02:32.519276Z","shell.execute_reply.started":"2024-09-22T14:02:32.508974Z","shell.execute_reply":"2024-09-22T14:02:32.517260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a myimports.py\n\n# Importing model and pipeline specifics:-\nfrom category_encoders import OrdinalEncoder, OneHotEncoder\n\n# Pipeline specifics:-\nfrom sklearn.preprocessing import (RobustScaler,\n                                   MinMaxScaler,\n                                   StandardScaler,\n                                   FunctionTransformer as FT,\n                                   PowerTransformer,\n                                  )\nfrom sklearn.impute import SimpleImputer as SI\nfrom sklearn.model_selection import (RepeatedStratifiedKFold as RSKF,\n                                     StratifiedKFold as SKF,\n                                     StratifiedGroupKFold as SGKF,\n                                     KFold,\n                                     GroupKFold as GKF,\n                                     RepeatedKFold as RKF,\n                                     PredefinedSplit as PDS,\n                                     cross_val_score,\n                                     cross_val_predict,\n                                    )\nfrom sklearn.inspection import permutation_importance\nfrom sklearn.feature_selection import VarianceThreshold as VT\nfrom sklearn.pipeline import Pipeline, make_pipeline\nfrom sklearn.base import (BaseEstimator, TransformerMixin, RegressorMixin, clone)\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.linear_model import Ridge\nfrom sklearn.metrics import (mean_squared_error as mse, \n                             cohen_kappa_score,\n                             ConfusionMatrixDisplay,\n                             confusion_matrix,\n                            )\n\n# Importing model packages\nimport xgboost as xgb, lightgbm as lgb\nfrom xgboost import QuantileDMatrix, XGBRegressor as XGBR\nfrom lightgbm import log_evaluation, early_stopping, LGBMRegressor as LGBMR\nfrom catboost import CatBoostRegressor as CBR, Pool\n\n# Importing Ensemble and tuning packages\nimport optuna\nfrom optuna import Trial, trial, create_study\nfrom optuna.pruners import HyperbandPruner\nfrom optuna.samplers import TPESampler, CmaEsSampler\noptuna.logging.disable_default_handler()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-22T14:02:32.522102Z","iopub.execute_input":"2024-09-22T14:02:32.522516Z","iopub.status.idle":"2024-09-22T14:02:32.540748Z","shell.execute_reply.started":"2024-09-22T14:02:32.522477Z","shell.execute_reply":"2024-09-22T14:02:32.539469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a myimports.py\n\n# Setting rc parameters in seaborn for plots and graphs-\nsns.set({\"axes.facecolor\"       : \"#ffffff\",\n         \"figure.facecolor\"     : \"#ffffff\",\n         \"axes.edgecolor\"       : \"#000000\",\n         \"grid.color\"           : \"#ffffff\",\n         \"font.family\"          : ['Cambria'],\n         \"axes.labelcolor\"      : \"#000000\",\n         \"xtick.color\"          : \"#000000\",\n         \"ytick.color\"          : \"#000000\",\n         \"grid.linewidth\"       : 0.75,\n         \"grid.linestyle\"       : \"--\",\n         \"axes.titlecolor\"      : '#0099e6',\n         'axes.titlesize'       : 8.5,\n         'axes.labelweight'     : \"bold\",\n         'legend.fontsize'      : 7.0,\n         'legend.title_fontsize': 7.0,\n         'font.size'            : 7.5,\n         'xtick.labelsize'      : 7.5,\n         'ytick.labelsize'      : 7.5,\n        }\n       )\n\n# Color printing\ndef PrintColor(text: str, color = Fore.BLUE, style = Style.BRIGHT):\n    \"Prints color outputs using colorama using a text F-string\"\n    print(style + color + text + Style.RESET_ALL)\n \n# Checking package versions\nimport xgboost as xgb, lightgbm as lgb, catboost as cb, sklearn as sk\nprint(f\"---> XGBoost = {xgb.__version__} | LightGBM = {lgb.__version__} | Catboost = {cb.__version__}\")\nprint(f\"---> Sklearn = {sk.__version__}| Pandas = {pd.__version__} | Polars = {pl.__version__}\")\ncollect()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-22T14:02:32.542226Z","iopub.execute_input":"2024-09-22T14:02:32.542645Z","iopub.status.idle":"2024-09-22T14:02:32.559409Z","shell.execute_reply.started":"2024-09-22T14:02:32.542606Z","shell.execute_reply":"2024-09-22T14:02:32.558006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a myimports.py\n\nclass MyLogger:\n    \"\"\"\n    This class helps to suppress logs in lightgbm and Optuna\n    Source - https://github.com/microsoft/LightGBM/issues/6014\n    \"\"\"\n\n    def init(self, logging_lbl: str):\n        self.logger = logging.getLogger(logging_lbl)\n        self.logger.setLevel(logging.ERROR)\n\n    def info(self, message):\n        pass\n\n    def warning(self, message):\n        pass\n\n    def error(self, message):\n        self.logger.error(message)\n        \n        \n# Customizing logging for XGBoost\nfor handler in logging.root.handlers[:]:\n    logging.root.removeHandler(handler)\n\nlogger = logging.getLogger(__name__)\nlogger.setLevel(logging.ERROR)\nformatter = logging.Formatter('%(asctime)s | %(levelname)s | %(message)s')\n\nstdout_handler = logging.StreamHandler(sys.stdout)\nstdout_handler.setLevel(logging.INFO)\nstdout_handler.setFormatter(formatter)\n\nfile_handler = logging.FileHandler(f'xgb_optimize.log')\nfile_handler.setLevel(logging.ERROR)\nfile_handler.setFormatter(formatter)\n\nlogger.addHandler(file_handler)\nlogger.addHandler(stdout_handler)\n\nclass XGBLogging(xgb.callback.TrainingCallback):\n    \"\"\"\n    This class designs the custom logging for XGboost \n    This is to be used inside the XGboost callback\n    \"\"\"\n\n    def __init__(self, epoch_log_interval=100):\n        self.epoch_log_interval = epoch_log_interval\n\n    def after_iteration(\n        self, model, epoch:int, evals_log:xgb.callback.TrainingCallback.EvalsLog\n    ):\n        if self.epoch_log_interval <= 0:\n            pass\n        \n        elif (epoch %  self.epoch_log_interval == 0):\n            for data, metric in evals_log.items():\n                for metric_name, log in metric.items():\n                    score = log[-1][0] if isinstance(log[-1], tuple) else log[-1]\n                    logger.info(f\"XGBLogging epoch {epoch} dataset {data} {metric_name} {score}\")\n\n        return False","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-22T14:02:32.561242Z","iopub.execute_input":"2024-09-22T14:02:32.561777Z","iopub.status.idle":"2024-09-22T14:02:32.577386Z","shell.execute_reply.started":"2024-09-22T14:02:32.561722Z","shell.execute_reply":"2024-09-22T14:02:32.576162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nexec(open('myimports.py','r').read())\nprint()","metadata":{"execution":{"iopub.status.busy":"2024-09-22T14:02:32.579067Z","iopub.execute_input":"2024-09-22T14:02:32.579580Z","iopub.status.idle":"2024-09-22T14:02:36.756094Z","shell.execute_reply.started":"2024-09-22T14:02:32.579529Z","shell.execute_reply":"2024-09-22T14:02:36.754753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **APPROACH DETAILS**","metadata":{}},{"cell_type":"markdown","source":"This is a starter notebook for the **Child Mind Institute — Problematic Internet Use** code competition. This is a time series tabular assignment to detect and classify the impact of excessive internet usage on child sleep patterns. We use **Quadratic Weighted Kappa Score** as evaluation metric. This needs to be maximized <br>\n\nThis kernel is a continuation of my [version1 baseline](https://www.kaggle.com/code/ravi20076/cmi2024-baseline-v1) where I illustrate the usage of proxy targets for the regression model. My approach here is based on the reference [here](https://www.kaggle.com/code/abdmental01/cmi-single-lgbm/notebook) <br>\n\nWe are provided a set of 3 datasets- <br> \n1. Parquet files for accelerometer train data, with separate tables for each id <br> \n2. Parquet files for accelerometer test data, with separate tables for each id <br>\n3. Metadata for remaining information and a separate table for data dictionary <br> \n\nIn this kernel, we start with the metadata and a few aggregate columns from the parquet file only and develop simple ML models as below- <br>\n1. I directly aim to predict the **sii true target** using the features provided. I ignore the PCIAT_* proxy targets in this version <br>\n2. I use lightgbm and xgboost models here and use a **custom objective** (Cohen Kappa loss) and the quadratic kappa eval metric for early stopping <br>\n3. I use threshold tuning to convert continuous predictions into labels. I shall experiment with scipy.minimize/ Optuna tuner for the same <br>\n\n### **MY NEW MODEL TRAINING CLASS** <br>\nI am happy to present my new model trainer class here. I used to use a trainer class in my yester work, example is [here](https://www.kaggle.com/code/ravi20076/playgrounds4e09-baseline-v1) <br>\nThis code became a bit clunky and needed revision, hence I thought of shortening the code, making it more readable and introducing a few new parameters. <br>\nThis class supports single and multiple model entries and trains one model at a time. It also allows the user to choose an ensemble method of his/ her choice separate from the training process <br>\n\n","metadata":{}},{"cell_type":"markdown","source":"# **CONFIGURATION**","metadata":{}},{"cell_type":"code","source":"%%writefile -a training.py\n\n# Configuration class:-\nclass CFG:\n    \"\"\"\n    Configuration class for parameters and CV strategy for tuning and training\n    Some parameters may be unused here as this is a general configuration class\n    \"\"\";\n\n    # Data preparation:-\n    version_nb  = 2\n    model_id    = \"V2_2\"\n    model_label = \"ML\"\n\n    test_req           = False\n    test_sample_frac   = 1000\n\n    gpu_switch         = \"OFF\"\n    state              = 42\n    target             = f\"sii\"\n    grouper            = f\"sii\"\n\n    ip_path            = f\"/kaggle/input/child-mind-institute-problematic-internet-use\"\n    op_path            = f\"/kaggle/working\"\n\n    # Model Training:-\n    pstprcs_oof        = False\n    pstprcs_train      = False\n    pstprcs_test       = False\n    ML                 = True\n    test_preds_req     = True\n\n    pseudo_lbl_req     = False\n    pseudolbl_up       = 0.975\n    pseudolbl_low      = 0.00\n\n    n_splits           = 3 if test_req == \"Y\" else 5\n    n_repeats          = 1\n    nbrnd_erly_stp     = 40\n    mdlcv_mthd         = 'RSKF'\n\n    # Ensemble:-\n    ensemble_req       = True\n    optuna_req         = True\n    metric_obj         = 'maximize'\n    ntrials            = 10 if test_req == \"Y\" else 300\n\n    # Global variables for plotting:-\n    grid_specs = {'visible'  : True,\n                  'which'    : 'both',\n                  'linestyle': '--',\n                  'color'    : 'lightgrey',\n                  'linewidth': 0.75\n                 }\n\n    title_specs = {'fontsize'   : 9,\n                   'fontweight' : 'bold',\n                   'color'      : '#992600',\n                  }\n    \ncv_selector = \\\n{\n \"RKF\"   : RKF(n_splits = CFG.n_splits, n_repeats= CFG.n_repeats, random_state= CFG.state),\n \"RSKF\"  : RSKF(n_splits = CFG.n_splits, n_repeats= CFG.n_repeats, random_state= CFG.state),\n \"SKF\"   : SKF(n_splits = CFG.n_splits, shuffle = True, random_state= CFG.state),\n \"KF\"    : KFold(n_splits = CFG.n_splits, shuffle = True, random_state= CFG.state),\n}\n\nPrintColor(f\"\\n---> Configuration done!\\n\")\ncollect()","metadata":{"execution":{"iopub.status.busy":"2024-09-22T14:02:36.757770Z","iopub.execute_input":"2024-09-22T14:02:36.758365Z","iopub.status.idle":"2024-09-22T14:02:36.765638Z","shell.execute_reply.started":"2024-09-22T14:02:36.758307Z","shell.execute_reply":"2024-09-22T14:02:36.764511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"|Configuration parameter| Explanation| Data type| Sample values |  \n| ---------------------- | ------------------------------- | --------------------- | --------------- |\n| version_nb    | Version Number | int | 1 | \n| model_id      | Model ID    | string | V1_1 | \n| model_label   | Model Label | string | ML | \n| test_req      | Test Required| bool | True / False | \n| test_sample_frac| Test sampled fraction | int | 1000 |\n| gpu_switch      | Do we need GPU support | bool | True / False |\n| state           | Random state | int | 42 |\n| target          | Target column | str | sii |\n| grouper         | CV grouper column | str | sii |\n| ip_path, op_path | Data paths  | str | |\n| pstprcs_* | Do we need post-processing  | bool |True / False |\n| ML| Do we need machine learning models  | bool |True / False |\n| test_preds_req| Do we need test set predictions (training in inference kernel)  | bool |True / False |\n| pseudo_lbl_req| Pseudo label required?  | bool |True / False |\n| pseudo_lbl_* | Pseudo label cutoff | float | |\n| n_splits/ n_repeats | N-splits and repeats for CV scheme | int | 3/5/10|\n| nbrnd_erly_stp | Early stopping rounds | int | 40|\n| mdlcv_mthd | Model CV method | str | RSKF|\n| ensemble_req | Do we need ensemble | bool | True / False |\n| optuna_req   | Do we need optuna | bool | True / False |\n| metric_obj   | Metric direction | str | minimize/ maximize |\n| ntrials      | Trials | int | 300 |","metadata":{}},{"cell_type":"code","source":"%%writefile -a training.py\n\nclass Utils:\n    \"\"\"\n    This class creates and uses several utility methods to be used across the code\n    \"\"\";\n\n    def __init__(self):\n        pass\n\n    def ScoreMetric(self, ytrue, ypred)-> float:\n        \"\"\"\n        This method calculates the metric for the competition\n        Inputs- ytrue, ypred:- input truth and predictions\n        Output- float:- competition metric\n        \"\"\"\n        myscore = \\\n        cohen_kappa_score(\n            np.uint8(np.round(ytrue)), \n            np.uint8(np.round(ypred)), \n            weights = \"quadratic\"\n        )\n        return myscore\n\n    def CleanMemory(self):\n        \"This method cleans the memory off unused objects and displays the cleaned state RAM usage\"\n\n        collect();\n        libc.malloc_trim(0)\n        pid        = getpid()\n        py         = Process(pid)\n        memory_use = py.memory_info()[0] / 2. ** 30\n        return f\"\\nRAM usage = {memory_use :.4} GB\"\n\n    def DisplayAdjTbl(self, *args):\n        \"\"\"\n        This function displays pandas tables in an adjacent manner, sourced from the below link-\n        https://stackoverflow.com/questions/38783027/jupyter-notebook-display-two-pandas-tables-side-by-side\n        \"\"\"\n\n        html_str = ''\n        for df in args:\n            html_str += df.to_html()\n        display_html(html_str.replace('table','table style=\"display:inline\"'),raw=True)\n        collect()\n\n    def DisplayScores(\n        self, Scores: pd.DataFrame, TrainScores: pd.DataFrame, methods: list\n    ):\n        \"This method displays the scores and their means\"\n\n        args = \\\n        [Scores.style.format(precision = 5).\\\n         background_gradient(cmap = \"Blues\", subset = methods + [\"Ensemble\"]).\\\n         set_caption(f\"\\nOOF scores across methods and folds\\n\"),\n\n         TrainScores.style.format(precision = 5).\\\n         background_gradient(cmap = \"Pastel2\", subset = methods).\\\n         set_caption(f\"\\nTrain scores across methods and folds\\n\")\n        ];\n\n        PrintColor(f\"\\n\\n\\n---> OOF score across all methods and folds\\n\",\n                   color = Fore.LIGHTMAGENTA_EX\n                   )\n        self.DisplayAdjTbl(*args)\n\n        print('\\n')\n        display(Scores.mean().to_frame().\\\n                transpose().\\\n                style.format(precision = 5).\\\n                background_gradient(cmap = \"mako\", axis=1,\n                                    subset = Scores.columns\n                                   ).\\\n                set_caption(f\"\\nOOF mean scores across methods and folds\\n\")\n               )\n\n\nutils = Utils()\ncollect()\nprint()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-22T14:04:59.744282Z","iopub.execute_input":"2024-09-22T14:04:59.744730Z","iopub.status.idle":"2024-09-22T14:04:59.753582Z","shell.execute_reply.started":"2024-09-22T14:04:59.744690Z","shell.execute_reply":"2024-09-22T14:04:59.752551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **PREPROCESSING**","metadata":{}},{"cell_type":"code","source":"%%writefile -a training.py\n\nclass Preprocessor:\n    \"This class organizes the preprocessing steps for the train-test data into a single code block\"\n    \n    def __init__(\n        self, cat_imp_val : str= \"missing\", ip_path: str = CFG.ip_path\n    ):\n        self.cat_imp_val = cat_imp_val\n        self.ip_path     = ip_path\n        \n    def make_pqfile_cols(\n        self, verbose: bool = False, label: str = \"Train\"\n    )->pl.DataFrame:\n        \"This method collates the id level parquet files and creates the aggregation columns in a polars dataframe\"\n\n        cols = [\"X\", \"Y\", \"Z\", \"enmo\", \"anglez\", \"light\", \"battery_voltage\"]\n        \n        ip_path   = os.path.join(self.ip_path, f\"series_{label.lower()}.parquet\")\n        all_files = os.listdir(ip_path)\n\n        for file_nb, file in tqdm(enumerate(all_files)):\n            df = \\\n            pl.scan_parquet(\n                os.path.join(ip_path, file, f\"part-0.parquet\")\n            ).select(pl.col(cols)).\\\n            collect().\\\n            describe(\n                percentiles = np.arange(0.05, 0.95, 0.10)\n            ).\\\n            filter(~pl.col(\"statistic\").is_in([\"count\", \"null_count\"])).\\\n            unpivot(index = \"statistic\").\\\n            with_columns(\n                pl.concat_str([pl.col(\"variable\"), pl.col(\"statistic\")],separator = \"_\",).alias(\"myvar\")\n            ).\\\n            with_columns(pl.col(\"myvar\").str.replace(r\"\\%\", \"\")).\\\n            select([\"myvar\", \"value\"]).\\\n            transpose(column_names = \"myvar\").\\\n            select(pl.all().shrink_dtype()).\\\n            with_columns(\n                pl.Series(\"id\", np.array(re.sub(\"id=\", \"\", file)))\n            )\n\n            if file_nb == 0:\n                op_df = df.clone()\n            elif file_nb > 0:\n                op_df = pl.concat([op_df, df], how = \"vertical_relaxed\")\n\n                if verbose:\n                    print(f\"---> Shapes = {op_df.shape}\")\n                else:\n                    pass\n            del df\n\n        PrintColor(f\"---> {label} - shape = {op_df.shape}\", color = Fore.CYAN)\n        return op_df\n\n    def pp_data(\n        self, df: pl.DataFrame, label: str = \"Train\", cat_cols: list = [], \n    ):\n        \"This method preprocesses the train-test data with requisite steps\"\n        \n        PrintColor(f\"\\n --- Data Processing - {label} --- \\n\")\n        PrintColor(f\"---> Shape = {df.shape} - memory usage {df.estimated_size('mb') :.3f} MB\", \n                   color = Fore.CYAN\n                  )\n        \n        if label == \"Train\":\n            cat_cols = df.select(cs.string().exclude(\"id\")).columns\n        else:\n            pass\n        \n        df    = df.with_columns(pl.col(cat_cols).fill_null(self.cat_imp_val).cast(pl.Categorical))\n        op_df = self.make_pqfile_cols(label = label)\n        df    = df.join(op_df, how = \"left\", on = \"id\")\n        df    = df.select(pl.all().shrink_dtype())\n        del op_df\n        \n        PrintColor(f\"---> Shape = {df.shape} - memory usage {df.estimated_size('mb') :.3f} MB\", \n                   color = Fore.CYAN\n                  )\n        return df, cat_cols\n        ","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-22T14:05:03.688373Z","iopub.execute_input":"2024-09-22T14:05:03.688793Z","iopub.status.idle":"2024-09-22T14:05:03.697204Z","shell.execute_reply.started":"2024-09-22T14:05:03.688751Z","shell.execute_reply":"2024-09-22T14:05:03.695953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\nexec(open('training.py','r').read())\ntrain  = pl.read_csv(os.path.join(CFG.ip_path, \"train.csv\")).drop(\"PCIAT-Season\", strict = False)\ntest   = pl.read_csv(os.path.join(CFG.ip_path, \"test.csv\"))\nsub_fl = pl.read_csv(os.path.join(CFG.ip_path, \"sample_submission.csv\"))\n\npp    = Preprocessor()\ntrain, cat_cols = pp.pp_data(train, \"Train\")\ntest, _  = pp.pp_data(test, \"Test\", cat_cols)\nsel_cols = test.drop(\"id\", strict = False).columns\n\nprint()\n_ = utils.CleanMemory()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-22T14:05:07.854369Z","iopub.execute_input":"2024-09-22T14:05:07.854778Z","iopub.status.idle":"2024-09-22T14:07:21.531845Z","shell.execute_reply.started":"2024-09-22T14:05:07.854740Z","shell.execute_reply":"2024-09-22T14:07:21.530725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **MODEL TRAINING**","metadata":{}},{"cell_type":"code","source":"%%writefile -a training.py\n\nclass ModelTrainer:\n    \"This class trains the provided model on the train-test data and returns the predictions and fitted models\"\n\n    def __init__(\n        self,\n        es_req         : bool  = False,\n        es             : int   = 100,\n        target         : str   = CFG.target,\n        metric_lbl     : str   = \"kappa\",\n        drop_cols      : list  = [\"Source\", \"id\", \"Id\", \"Label\", CFG.target, \"fold_nb\"],\n    ):\n        \"\"\"\n        Key parameters-\n        es_iter - early stopping rounds for boosted trees\n        \"\"\"\n        \n        drop_cols = list(set(drop_cols + [target]))\n        \n        self.es_req         = es_req\n        self.es_iter        = es\n        self.target         = target\n        self.drop_cols      = drop_cols\n        self.metric_lbl     = metric_lbl\n        \n    def ScoreMetric(self, ytrue, ypred)->float:\n        \"\"\"\n        This is the metric function for the competition scoring\n        \"\"\"\n        if self.metric_lbl == \"rmse\":\n            return mse(ytrue, ypred, squared = False)\n        \n        elif self.metric_lbl == \"kappa\":\n            myscore = \\\n            cohen_kappa_score(\n                np.uint8(np.around(ytrue,0)),\n                np.uint8(np.around(ypred,0)),\n                weights = \"quadratic\",\n            )\n            return myscore\n\n    def PlotFtreImp(\n        self, ftreimp: pd.Series, method: str,\n        ntop: int = 50,\n        title_specs: dict = CFG.title_specs,\n        **params,\n    ):\n        \"This function plots the feature importances for the model provided\"\n\n        print()\n        fig, ax = plt.subplots(1, 1, figsize = (25, 7.5))\n\n        ftreimp.sort_values(ascending = False).\\\n        head(ntop).\\\n        plot.bar(ax = ax, color = \"blue\")\n        ax.set_title(f\"Feature Importances - {method}\", **title_specs)\n\n        plt.tight_layout()\n        plt.show()\n        print()\n\n    def PostProcessPreds(self, ypred):\n        \"This method post-processes predictions optionally\"\n        return np.clip(ypred, a_min = 0, a_max = np.inf)\n\n    def MakeOfflineModel(\n        self, X, y, ygrp, Xtest, mdl, method,\n        test_preds_req   : bool = True,\n        ftreimp_plot_req : bool = True,\n        ntop             : int  = 50,\n        **params,\n    ):\n        \"\"\"\n        This function trains the provided model on the dataset and cross-validates appropriately\n\n        Inputs-\n        X, y, ygrp       - training data components\n        Xtest            - test data\n        model            - model object for training\n        method           - model method label\n        test_preds_req   - boolean flag to extract test set predictions\n        ftreimp_plot_req - boolean flag to plot tree feature importances\n        ntop             - top n features for feature importances plot\n\n        Returns-\n        oof_preds, test_preds - prediction arrays\n        fitted_models         - fitted model list for test set\n        ftreimp               - feature importances across selected features\n        mdl_best_iter         - model average best iteration across folds\n        \"\"\"\n\n        oof_preds     = np.zeros(len(X))\n        test_preds    = []\n        mdl_best_iter = []\n        ftreimp       = 0\n        \n        scores, tr_scores, fitted_models = [], [], []\n        cv = PDS(ygrp)\n        n_splits = ygrp.nunique()\n\n        for fold_nb, (train_idx, dev_idx) in tqdm(enumerate(cv.split(X, y))):\n            Xtr   = X.iloc[train_idx]\n            Xdev  = X.iloc[dev_idx]\n            ytr   = y.iloc[train_idx]\n            ydev  = y.iloc[dev_idx]\n            model = clone(mdl)\n\n            if \"CB\" in method and self.es_req == True:\n                model.fit(Xtr, ytr,\n                          eval_set = [(Xdev, ydev)],\n                          verbose = 0,\n                          early_stopping_rounds = self.es_iter,\n                          )\n                best_iter = model.get_best_iteration()\n\n            elif \"LGB\" in method and self.es_req == True:\n                model.fit(Xtr, ytr,\n                          eval_set = [(Xdev, ydev)],\n                          callbacks = [log_evaluation(0),\n                                       early_stopping(stopping_rounds = self.es_iter, verbose = False,),\n                                       ],\n                          eval_metric = MakeEvalMetric,\n                          )\n                best_iter = model.best_iteration_\n\n            elif \"XGB\" in method and self.es_req == True:\n                model.fit(Xtr, ytr,\n                          eval_set = [(Xdev, ydev)],\n                          verbose  = 0,\n                          )\n                best_iter = model.best_iteration\n\n            else:\n                model.fit(Xtr, ytr)\n                best_iter = -1\n\n            fitted_models.append(model)\n\n            try:\n                ftreimp += model.feature_importances_\n            except:\n                pass\n\n            dev_preds = self.PostProcessPreds(model.predict(Xdev))\n            oof_preds[Xdev.index] = dev_preds\n\n            train_preds  = self.PostProcessPreds(model.predict(Xtr))\n            tr_score     = self.ScoreMetric(ytr.values.flatten(), train_preds)\n            score        = self.ScoreMetric(ydev.values.flatten(), dev_preds)\n\n            scores.append(score)\n            tr_scores.append(tr_score)\n\n            nspace = 15 - len(method) - 2 if fold_nb <= 9 else 15 - len(method) - 1\n            \n            if self.es_req:\n                PrintColor(f\"{method} Fold{fold_nb} {' ' * nspace} OOF = {score:.6f} | Train = {tr_score:.6f} | Iter = {best_iter:,.0f} \")\n            else:\n                PrintColor(f\"{method} Fold{fold_nb} {' ' * nspace} OOF = {score:.6f} | Train = {tr_score:.6f}\")\n                \n            mdl_best_iter.append(best_iter)\n\n            if test_preds_req:\n                test_preds.append(\n                    self.PostProcessPreds(\n                        model.predict(Xtest)\n                    )\n                )\n            else:\n                pass\n\n        test_preds    = np.mean(np.stack(test_preds, axis = 1), axis=1)\n        ftreimp       = pd.Series(ftreimp, index = Xdev.columns)\n        mdl_best_iter = np.uint16(np.amax(mdl_best_iter))\n\n        if ftreimp_plot_req :\n            print()\n            self.PlotFtreImp(ftreimp, method = method, ntop = ntop,)\n        else:\n            pass\n\n        PrintColor(f\"\\n---> {np.mean(scores):.6f} +- {np.std(scores):.6f} | OOF\", color = Fore.RED)\n        PrintColor(f\"---> {np.mean(tr_scores):.6f} +- {np.std(tr_scores):.6f} | Train\", color = Fore.RED)\n\n        if self.es_req == False:\n            pass\n        else:\n            PrintColor(f\"---> Max best iteration = {mdl_best_iter :,.0f}\", color = Fore.RED)\n\n        return (fitted_models, oof_preds, test_preds, ftreimp, mdl_best_iter)\n\n    def MakeOnlineModel(\n        self, X, y, Xtest, model, method,\n        test_preds_req : bool = False,\n    ):\n        \"This method refits the model on the complete train data and returns the model fitted object and predictions\"\n\n        try:\n            model.early_stopping_rounds = None\n        except:\n            pass\n\n        try:\n            model.fit(X, y, verbose = 0)\n        except:\n            model.fit(X, y,)\n\n        oof_preds  = model.predict(X)\n        if test_preds_req:\n            test_preds = model.predict(Xtest[X.columns])\n        else:\n            test_preds = 0\n        return (model, oof_preds, test_preds)\n\n    def MakeOfflinePreds(self, X, fitted_models):\n        \"This method creates test-set predictions for the offline model provided\"\n\n        test_preds = 0\n        n_splits   = len(fitted_models)\n        PrintColor(f\"---> Number of splits = {n_splits}\")\n\n        for model in fitted_models:\n            test_preds = test_preds + (model.predict(X) / n_splits)\n\n        return test_preds","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-22T14:08:14.566557Z","iopub.execute_input":"2024-09-22T14:08:14.566961Z","iopub.status.idle":"2024-09-22T14:08:14.578004Z","shell.execute_reply.started":"2024-09-22T14:08:14.566922Z","shell.execute_reply":"2024-09-22T14:08:14.576693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a training.py\n\nclass OptunaEnsembler:\n    \"\"\"\n    This is the Optuna ensemble class-\n    Source- https://www.kaggle.com/code/arunklenin/ps3e26-cirrhosis-survial-prediction-multiclass\n    \"\"\";\n\n    def __init__(\n        self, state: int = 42, ntrials: int = 300, \n        metric_obj: str  = \"maximize\", \n        metric_lbl: str  = \"kappa\",\n        **params\n    ):\n        self.study        = None\n        self.weights      = None\n        self.random_state = state\n        self.n_trials     = ntrials\n        self.direction    = metric_obj\n        self.metric_lbl   = metric_lbl\n\n    def ScoreMetric(self, ytrue, ypred)->float:\n        \"\"\"\n        This is the metric function for the competition\n        \"\"\"\n        \n        if self.metric_lbl == \"rmse\":\n            return mse(ytrue, ypred, squared = False)\n        else:\n            myscore = \\\n            cohen_kappa_score(\n                np.uint8(np.around(ytrue,0)),\n                np.uint8(np.around(ypred,0)), \n                weights = \"quadratic\"\n            )\n            return myscore\n\n    def _objective(\n        self, trial, y_true, y_preds\n    ):\n        \"\"\"\n        This method defines the objective function for the ensemble\n        \"\"\";\n\n        if isinstance(y_preds, pd.DataFrame) or isinstance(y_preds, np.ndarray):\n            weights = [trial.suggest_float(f\"weight{n}\", 0.001, 0.999)\n                       for n in range(y_preds.shape[-1])\n                      ]\n            axis = 1\n\n        elif isinstance(y_preds, list):\n            weights = [trial.suggest_float(f\"weight{n}\", 0.001, 0.999)\n                       for n in range(len(y_preds))\n                      ]\n            axis = 0\n\n        # Calculating the weighted prediction:-\n        weighted_pred  = np.average(np.array(y_preds), axis = axis, weights = weights)\n        score          = self.ScoreMetric(y_true, weighted_pred)\n        return score\n\n    def fit(self, y_true, y_preds):\n        \"This method fits the Optuna objective on the fold level data\";\n\n        optuna.logging.set_verbosity = optuna.logging.ERROR\n\n        self.study = \\\n        optuna.create_study(sampler    = TPESampler(seed = self.random_state),\n                            pruner     = HyperbandPruner(),\n                            study_name = \"Ensemble\",\n                            direction  = self.direction,\n                           )\n\n        obj = partial(self._objective, y_true = y_true, y_preds = y_preds)\n        self.study.optimize(obj, n_trials = self.n_trials)\n\n        if isinstance(y_preds, list):\n            self.weights = [self.study.best_params[f\"weight{n}\"] for n in range(len(y_preds))]\n\n        else:\n            self.weights = [self.study.best_params[f\"weight{n}\"] for n in range(y_preds.shape[-1])]\n\n    def predict(self, y_preds):\n        \"This method predicts using the fitted Optuna objective\";\n\n        assert self.weights is not None, 'OptunaWeights error, must be fitted before predict';\n\n        if isinstance(y_preds, list):\n            weighted_pred = np.average(np.array(y_preds), axis=0, weights = self.weights)\n\n        else:\n            weighted_pred = np.average(np.array(y_preds), axis=1, weights = self.weights)\n\n        return weighted_pred\n\n    def fit_predict(self, y_true, y_preds):\n        \"\"\"\n        This method fits the Optuna objective on the fold data, then predicts the test set\n        \"\"\";\n        self.fit(y_true, y_preds)\n        return self.predict(y_preds)\n\n    def weights(self):\n        \"This method returns the non-normalized weights for all models in a fold\"\n        return self.weights\n\nprint()\ncollect();","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-22T14:08:20.431645Z","iopub.execute_input":"2024-09-22T14:08:20.432047Z","iopub.status.idle":"2024-09-22T14:08:20.440934Z","shell.execute_reply.started":"2024-09-22T14:08:20.432012Z","shell.execute_reply":"2024-09-22T14:08:20.439700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a training.py\n\ndef NormWeights(weights: dict, methods: list):\n    \"This function normalizes the weights and returns a dataframe of normalized weights across folds and models\"\n\n    weights = pd.DataFrame.from_dict(weights).T\n    weights[\"row_sum\"] = weights.sum(axis=1)\n\n    for col in weights.columns:\n        weights[col] = weights[col] / weights[\"row_sum\"]\n\n    weights.drop(\"row_sum\", axis = 1, inplace = True, errors = \"ignore\")\n    weights.columns    = methods\n    weights.index.name = \"Fold_Nb\"\n    return weights\n\ndef MakeEnsemble(\n    target: str, metric_obj: str, metric_lbl: str = \"kappa\"\n):\n    \"This function implements the Optuna ensemble on the OOF and test prediction datasets\"\n\n    global OOF_Preds, Mdl_Preds\n\n    PrintColor(f\"\\n{'=' * 20} ENSEMBLE {'=' * 20}\\n\")\n    \n    ygrp       = OOF_Preds[\"fold_nb\"]\n    cv         = PDS(ygrp)\n    oof_preds  = np.zeros(len(OOF_Preds))\n    test_preds = []\n    scores     = []\n    weights    = {}\n    drop_cols  = [\"fold_nb\", target, \"Ensemble\"]\n    n_splits   = ygrp.nunique()\n\n    for fold_nb, (_, dev_idx) in tqdm(enumerate(cv.split(OOF_Preds, OOF_Preds[target]))):\n        Xdev = OOF_Preds.iloc[dev_idx].drop(drop_cols, axis=1, errors = \"ignore\")\n        ydev = OOF_Preds.loc[dev_idx, target]\n\n        ens = OptunaEnsembler(\n            ntrials = CFG.ntrials, metric_lbl = metric_lbl, metric_obj = metric_obj\n        )\n        ens.fit(ydev, Xdev,)\n\n        dev_preds = ens.predict(Xdev)\n        score     = ens.ScoreMetric(ydev.values, dev_preds)\n        oof_preds[dev_idx] = dev_preds\n        test_preds.append(\n            ens.predict(Mdl_Preds.drop(drop_cols, axis=1, errors = \"ignore\"))\n        )\n\n        PrintColor(f\"---> {score: .6f} | Fold {fold_nb}\", color = Fore.CYAN)\n        scores.append(score)\n\n        weights[f\"Fold{fold_nb}\"] = ens.weights\n\n    PrintColor(f\"\\n---> OOF = {np.mean(scores): .6f} +- {np.std(scores): .6f} | Ensemble\",\n               color = Fore.RED\n              )\n\n    test_preds = np.mean(np.stack(test_preds, axis=1), axis=1,)\n\n    OOF_Preds[\"Ensemble\"] = oof_preds\n    Mdl_Preds[\"Ensemble\"] = test_preds\n\n    weights = \\\n    NormWeights(\n        weights,\n        methods = Mdl_Preds.drop(drop_cols, axis=1, errors = \"ignore\").columns\n    )\n\n    print(\"\\n\\n\\n\")\n    display(\n        weights.\\\n        style.\\\n        set_caption(\"Normalized weights\").\\\n        format(precision = 6).\\\n        set_properties(\n            props = \"color:red; background-color:white; font-weight: bold; border: maroon dashed 1.6px\"\n        )\n    )\n    \n    print()\n    display(\n        weights.mean().to_frame().transpose().\\\n        style.\\\n        format(precision = 6).\\\n        set_caption(\"Normalized Mean weights\").\\\n        set_properties(\n            props = \"color:red; background-color:white; font-weight: bold; border: maroon dashed 1.6px\"\n        )        \n    )\n\n    return weights\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-22T14:08:23.967613Z","iopub.execute_input":"2024-09-22T14:08:23.968094Z","iopub.status.idle":"2024-09-22T14:08:23.976185Z","shell.execute_reply.started":"2024-09-22T14:08:23.968050Z","shell.execute_reply":"2024-09-22T14:08:23.974807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a training.py\n\nclass OptimizedRounder:\n    \"\"\"\n    Source - https://www.kaggle.com/code/tubotubo/starter-notebook-multi-target-prediction\n    \"\"\"\n\n    def __init__(\n        self, n_classes: int, n_trials: int = 100, direction : str = \"maximize\"\n    ):\n        self.n_classes  = n_classes\n        self.labels     = np.arange(n_classes)\n        self.n_trials   = n_trials\n        self.metric     = partial(cohen_kappa_score, weights=\"quadratic\")\n        self.direction  = direction\n        \n    def _objective(\n        self, trial: optuna.Trial, y_true: NDArray[np.int_], y_pred: NDArray[np.float_],\n    ) -> float:\n        \n        thresholds = []\n        for i in range(self.n_classes - 1):\n            low  = max(thresholds) if i > 0 else min(self.labels)\n            high = max(self.labels)\n            th   = trial.suggest_float(f\"threshold_{i}\", low, high)\n            thresholds.append(th)\n            \n        try:\n            y_pred_rounded = np.digitize(y_pred, thresholds)\n        except ValueError:\n            return -100\n        return self.metric(y_true, y_pred_rounded)\n\n    def fit(\n        self, y_pred: NDArray[np.float_], y_true: NDArray[np.int_]\n    ) -> None:\n        y_pred = self._normalize(y_pred)\n        study  = optuna.create_study(direction = self.direction)\n        obj    = partial(self._objective, y_true = y_true, y_pred = y_pred)\n        \n        study.optimize(obj, n_trials = self.n_trials)\n        self.thresholds = [study.best_params[f\"threshold_{i}\"] for i in range(self.n_classes - 1)]\n\n    def predict(self, y_pred: NDArray[np.float_]) -> NDArray[np.int_]:\n        assert hasattr(self, \"thresholds\"), \"fit() must be called before predict()\"\n        y_pred = self._normalize(y_pred)\n        return np.digitize(y_pred, self.thresholds)\n\n    def _normalize(self, y: NDArray[np.float_]) -> NDArray[np.float_]:\n        return (y - y.min()) / (y.max() - y.min()) * (self.n_classes - 1)\n    \n    def thresholds(self):\n        return self.thresholds\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-22T14:08:27.381816Z","iopub.execute_input":"2024-09-22T14:08:27.382736Z","iopub.status.idle":"2024-09-22T14:08:27.390152Z","shell.execute_reply.started":"2024-09-22T14:08:27.382693Z","shell.execute_reply":"2024-09-22T14:08:27.388993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **MODEL I-O**","metadata":{}},{"cell_type":"code","source":"%%writefile -a training.py\n\na = 2.998\nb = 1.092\n\ndef MakeObj(y_true, y_pred):\n    \"This function is the common custom objective for LGBM and XGB\"\n    \n    labels = y_true + a\n    preds  = y_pred + a\n    preds  = preds.clip(0, np.inf)\n    f      = 1/2*np.sum((preds-labels)**2)\n    g      = 1/2*np.sum((preds-a)**2+b)\n    df     = preds - labels\n    dg     = preds - a\n    grad   = (df/g - f*dg/g**2)*len(labels)\n    hess   = np.ones(len(labels))\n    \n    return grad, hess\n\ndef MakeEvalMetric(y_true, y_pred):\n    \n    if isinstance(y_pred, xgb.QuantileDMatrix):\n        y_true, y_pred = y_pred, y_true\n        y_true = (y_true.get_label() + a).round()\n        y_pred = (y_pred + a).clip(0, np.inf).round()\n        \n        qwk = cohen_kappa_score(np.int8(y_true), np.int8(y_pred), weights=\"quadratic\")\n        return 'QWK', qwk\n\n    else:\n        y_true = y_true + a\n        y_pred = (y_pred + a).clip(0, np.inf).round()\n        \n        qwk = cohen_kappa_score(np.int8(y_true), np.int8(y_pred), weights=\"quadratic\")\n        return 'QWK', qwk, True","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-22T14:08:31.269177Z","iopub.execute_input":"2024-09-22T14:08:31.269607Z","iopub.status.idle":"2024-09-22T14:08:31.277145Z","shell.execute_reply.started":"2024-09-22T14:08:31.269568Z","shell.execute_reply":"2024-09-22T14:08:31.275921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n\nexec(open('training.py','r').read())\n\ntry:\n    l = MyLogger()\n    l.init(logging_lbl = \"lightgbm_custom\")\n    lgb.register_logger(l)\nexcept:\n    pass\n\n# Initializing CV scheme\ncv = cv_selector[CFG.mdlcv_mthd]\n\n# Initializing model parameters\nMdl_Master = \\\n{\n f'LGBM1R' : LGBMR(**{\"objective\"           : MakeObj,\n                      \"metrics\"             : \"None\",\n                      'device'              : \"gpu\" if CFG.gpu_switch == \"ON\" else \"cpu\",\n                      'learning_rate'       : 0.025, \n                      'n_estimators'        : 270,\n                      'max_depth'           : 6, \n                      'num_leaves'          : 85, \n                      'min_data_in_leaf'    : 12,\n                      'feature_fraction'    : 0.70, \n                      'bagging_fraction'    : 0.88, \n                      'bagging_freq'        : 6, \n                      'lambda_l1'           : 9.92, \n                      'lambda_l2'           : 4.35,\n                      'verbosity'           : -1,\n                      'random_state'        : CFG.state,\n                     }\n                  ),\n    \n f'XGB1R' : XGBR(**{  \"objective\"           : MakeObj,\n                      'device'              : \"cuda\" if CFG.gpu_switch == \"ON\" else \"cpu\",\n                      'learning_rate'       : 0.03, \n                      'n_estimators'        : 200,\n                      'max_depth'           : 4, \n                      'colsample_bytree'    : 0.55, \n                      'colsample_bynode'    : 0.60,\n                      'colsample_bylevel'   : 0.70,                     \n                      'reg_alpha'           : 2.50, \n                      'reg_lambda'          : 7.50,\n                      'verbose'             : 0,\n                      'random_state'        : CFG.state,\n                      'enable_categorical'  : True,\n                      'callbacks'           : [XGBLogging(epoch_log_interval= 0)],\n                     }\n                  ),\n}\n\n# Initializing model outputs\nOOF_Preds    = {}\nMdl_Preds    = {}\nFittedModels = {}\nFtreImp      = {}\nSelMdlCols   = {}","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-22T14:27:31.183777Z","iopub.execute_input":"2024-09-22T14:27:31.184751Z","iopub.status.idle":"2024-09-22T14:27:32.774658Z","shell.execute_reply.started":"2024-09-22T14:27:31.184707Z","shell.execute_reply":"2024-09-22T14:27:32.773431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **MODEL CV ANALYSIS**","metadata":{}},{"cell_type":"code","source":"%%time \n\nmytrain  = train.to_pandas().dropna(subset = [CFG.target])\nmytest   = test.to_pandas()[sel_cols]\nmytarget = CFG.target\n\nmytrain.index = range(len(mytrain))\nPrintColor(f\"---> Shapes = {mytrain.shape} {mytest.shape}\", color = Fore.CYAN)\n\n# Initializing CV folds across the training data:-\nfolds = np.zeros(len(mytrain))\nfor fold_nb, (train_idx, dev_idx) in enumerate(cv.split(mytrain, mytrain[mytarget])):\n    folds[dev_idx] = fold_nb\nmytrain[\"fold_nb\"] = folds\ndel folds\n\nPrintColor(f\"---> Shapes = {mytrain.shape} {mytest.shape}\\n\\n\", color = Fore.CYAN)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T14:27:33.998589Z","iopub.execute_input":"2024-09-22T14:27:33.999590Z","iopub.status.idle":"2024-09-22T14:27:34.033277Z","shell.execute_reply.started":"2024-09-22T14:27:33.999545Z","shell.execute_reply":"2024-09-22T14:27:34.032213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n      \n# Creating CV scheme:-\nmd = ModelTrainer(es = CFG.nbrnd_erly_stp, target = mytarget)\n\nfor method, mdl in tqdm(Mdl_Master.items()):\n    PrintColor(f\"\\n{'-' * 10} {method} MODEL TRAINING - {mytarget} {'-' * 10}\\n\", \n               color = Fore.MAGENTA\n              )\n\n    fitted_models, oof_preds, test_preds, ftreimp, mdl_best_iter =  \\\n    md.MakeOfflineModel(\n        mytrain[sel_cols],\n        mytrain[mytarget],\n        mytrain[\"fold_nb\"],\n        mytest,\n        clone(mdl),\n        method,\n        test_preds_req   = True,\n        ftreimp_plot_req = True,\n        ntop = 50,\n    ) \n\n    # Integrating data    \n    OOF_Preds[f\"{method}\"]    = oof_preds\n    Mdl_Preds[f\"{method}\"]    = test_preds\n    FtreImp[f\"{method}\"]      = ftreimp\n    FittedModels[f\"{method}\"] = fitted_models\n\n    del fitted_models, oof_preds, test_preds, ftreimp, mdl_best_iter\n    _ = utils.CleanMemory()\n    \nPrintColor(utils.CleanMemory())    ","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-22T14:27:35.949724Z","iopub.execute_input":"2024-09-22T14:27:35.950510Z","iopub.status.idle":"2024-09-22T14:28:03.856235Z","shell.execute_reply.started":"2024-09-22T14:27:35.950464Z","shell.execute_reply":"2024-09-22T14:28:03.855191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **ENSEMBLE**","metadata":{}},{"cell_type":"code","source":"%%time \n\nOOF_Preds = pd.DataFrame.from_dict(OOF_Preds, orient = \"columns\")\nMdl_Preds = pd.DataFrame.from_dict(Mdl_Preds, orient = \"columns\")\n\nOOF_Preds[mytarget]  = mytrain[mytarget].values\nOOF_Preds[\"fold_nb\"] = mytrain[\"fold_nb\"].values\n\nweights = \\\nMakeEnsemble(target = mytarget, metric_obj = CFG.metric_obj, metric_lbl = 'kappa')\n\nprint()\nPrintColor(utils.CleanMemory())","metadata":{"execution":{"iopub.status.busy":"2024-09-22T14:28:30.207608Z","iopub.execute_input":"2024-09-22T14:28:30.208031Z","iopub.status.idle":"2024-09-22T14:28:30.217665Z","shell.execute_reply.started":"2024-09-22T14:28:30.207993Z","shell.execute_reply":"2024-09-22T14:28:30.216523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **THRESHOLD TUNING**","metadata":{}},{"cell_type":"markdown","source":"In this section, we try and establish the best CV based cutoffs to convert the continuous predictions to labels for the target prediction <br>\nHere, we use the predicted values for the true target and unearth the best values for **sii** labels","metadata":{}},{"cell_type":"code","source":"%%time \n\nmytuner = OptimizedRounder(n_classes = 4, n_trials = CFG.ntrials)\nytrain  = np.uint8(mytrain[CFG.target])\n\nmytuner.fit(OOF_Preds[\"LGBM1R\"], ytrain)\nens_preds  = mytuner.predict(OOF_Preds[\"LGBM1R\"])\ntest_preds = mytuner.predict(Mdl_Preds[\"LGBM1R\"])\n\nPrintColor(f\"---> Thresholds for labels\")\nwith np.printoptions(linewidth = 100, precision = 5):\n    pprint(np.array(mytuner.thresholds))\n\n# Displaying the confusion matrix \nscore = utils.ScoreMetric(ytrain, ens_preds)\nPrintColor(f\"\\n---> Final ensemble OOF score = {score :.6f}\\n\\n\")\n\nfig, ax = plt.subplots(1,1, figsize = (5,5))\ndisp = \\\nConfusionMatrixDisplay(\n    confusion_matrix = confusion_matrix(ytrain, ens_preds),  \n    display_labels = list(range(4))\n)\n\ndisp.plot(\n    cmap = \"Blues\", \n    ax = ax, \n    colorbar = False, \n    xticks_rotation = 0,\n    text_kw = {\"fontweight\": \"bold\",  \n               \"fontsize\"  : 16,\n              }\n)\nax.set_title(\n    f\"Confusion matrix - CV = {score :.6f}\", **CFG.title_specs\n)\nax.grid(**CFG.grid_specs)\nax.set(ylabel = f\"True {CFG.target}\", \n       xlabel = f\"Predicted {CFG.target}\", \n      )\nplt.show()\n\n_ = utils.CleanMemory()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-22T14:30:45.377742Z","iopub.execute_input":"2024-09-22T14:30:45.378187Z","iopub.status.idle":"2024-09-22T14:30:51.832827Z","shell.execute_reply.started":"2024-09-22T14:30:45.378142Z","shell.execute_reply":"2024-09-22T14:30:51.831704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **SUBMISSION**","metadata":{}},{"cell_type":"code","source":"%%time \n\ntry:\n    print()\n    display(\n        OOF_Preds.head().style.format(precision = 3).set_caption(\"OOF Predictions\")\n    )\n\n    print()\n    display(\n        Mdl_Preds.head().style.format(precision = 3).set_caption(\"Model Predictions\")\n    )\nexcept:\n    pass\n\nsub_fl.with_columns(\n    pl.Series(CFG.target, test_preds.flatten(), pl.UInt8)\n).write_csv(\"submission.csv\")\n\nprint()\n!ls \nprint(f\"\\n\\n---> Submission file\\n\\n\")\n!head submission.csv\n\nPrintColor(utils.CleanMemory())","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-22T14:30:58.992028Z","iopub.execute_input":"2024-09-22T14:30:58.992588Z","iopub.status.idle":"2024-09-22T14:31:01.732415Z","shell.execute_reply.started":"2024-09-22T14:30:58.992544Z","shell.execute_reply":"2024-09-22T14:31:01.731123Z"},"trusted":true},"execution_count":null,"outputs":[]}]}