{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":197529256,"sourceType":"kernelVersion"}],"dockerImageVersionId":30761,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **IMPORTS**","metadata":{}},{"cell_type":"code","source":"%%capture \n\n!pip install -q lightgbm==4.5.0 --no-index --find-links=/kaggle/input/cmi2024-packages-v1/MLPackages\n!pip install -q polars==1.7.1 --no-index --find-links=/kaggle/input/cmi2024-packages-v1/Polars171","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-21T19:57:36.363393Z","iopub.execute_input":"2024-09-21T19:57:36.364003Z","iopub.status.idle":"2024-09-21T19:58:12.351965Z","shell.execute_reply.started":"2024-09-21T19:57:36.363939Z","shell.execute_reply":"2024-09-21T19:58:12.350447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a myimports.py\n\nprint(f\"\\n---> Commencing imports\")\n\nfrom gc import collect\nfrom warnings import filterwarnings\nfilterwarnings('ignore')\nfrom IPython.display import display_html, clear_output\nclear_output()\nimport os, sys, logging, re, joblib, ctypes, shutil\nfrom copy import deepcopy\n\n# General library imports:-\nfrom os import path, walk, getpid\nfrom psutil import Process\nfrom collections import Counter\nfrom itertools import product\nimport ctypes\nlibc = ctypes.CDLL(\"libc.so.6\")\n\nfrom IPython.display import display_html, clear_output\nfrom pprint import pprint\nfrom functools import partial\nfrom copy import deepcopy\nimport pandas as pd, numpy as np\nfrom numpy.typing import ArrayLike, NDArray\nimport polars as pl\nimport polars.selectors as cs\nfrom polars.testing import assert_frame_equal\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom colorama import Fore, Style, init\nfrom tqdm.notebook import tqdm","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-21T19:58:12.354369Z","iopub.execute_input":"2024-09-21T19:58:12.354778Z","iopub.status.idle":"2024-09-21T19:58:12.363994Z","shell.execute_reply.started":"2024-09-21T19:58:12.354736Z","shell.execute_reply":"2024-09-21T19:58:12.362472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a myimports.py\n\n# Importing model and pipeline specifics:-\nfrom category_encoders import OrdinalEncoder, OneHotEncoder\n\n# Pipeline specifics:-\nfrom sklearn.preprocessing import (RobustScaler,\n                                   MinMaxScaler,\n                                   StandardScaler,\n                                   FunctionTransformer as FT,\n                                   PowerTransformer,\n                                  )\nfrom sklearn.impute import SimpleImputer as SI\nfrom sklearn.model_selection import (RepeatedStratifiedKFold as RSKF,\n                                     StratifiedKFold as SKF,\n                                     StratifiedGroupKFold as SGKF,\n                                     KFold,\n                                     GroupKFold as GKF,\n                                     RepeatedKFold as RKF,\n                                     PredefinedSplit as PDS,\n                                     cross_val_score,\n                                     cross_val_predict,\n                                    )\nfrom sklearn.inspection import permutation_importance\nfrom sklearn.feature_selection import VarianceThreshold as VT\nfrom sklearn.pipeline import Pipeline, make_pipeline\nfrom sklearn.base import (BaseEstimator, TransformerMixin, RegressorMixin, clone)\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.linear_model import Ridge\nfrom sklearn.metrics import (mean_squared_error as mse, \n                             cohen_kappa_score,\n                             ConfusionMatrixDisplay,\n                             confusion_matrix,\n                            )\n\n# Importing model packages\nimport xgboost as xgb, lightgbm as lgb\nfrom xgboost import QuantileDMatrix, XGBRegressor as XGBR\nfrom lightgbm import log_evaluation, early_stopping, LGBMRegressor as LGBMR\nfrom catboost import CatBoostRegressor as CBR, Pool\n\n# Importing Ensemble and tuning packages\nimport optuna\nfrom optuna import Trial, trial, create_study\nfrom optuna.pruners import HyperbandPruner\nfrom optuna.samplers import TPESampler, CmaEsSampler\noptuna.logging.disable_default_handler()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-21T19:58:12.365946Z","iopub.execute_input":"2024-09-21T19:58:12.366480Z","iopub.status.idle":"2024-09-21T19:58:12.382305Z","shell.execute_reply.started":"2024-09-21T19:58:12.366424Z","shell.execute_reply":"2024-09-21T19:58:12.380972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a myimports.py\n\n# Setting rc parameters in seaborn for plots and graphs-\nsns.set({\"axes.facecolor\"       : \"#ffffff\",\n         \"figure.facecolor\"     : \"#ffffff\",\n         \"axes.edgecolor\"       : \"#000000\",\n         \"grid.color\"           : \"#ffffff\",\n         \"font.family\"          : ['Cambria'],\n         \"axes.labelcolor\"      : \"#000000\",\n         \"xtick.color\"          : \"#000000\",\n         \"ytick.color\"          : \"#000000\",\n         \"grid.linewidth\"       : 0.75,\n         \"grid.linestyle\"       : \"--\",\n         \"axes.titlecolor\"      : '#0099e6',\n         'axes.titlesize'       : 8.5,\n         'axes.labelweight'     : \"bold\",\n         'legend.fontsize'      : 7.0,\n         'legend.title_fontsize': 7.0,\n         'font.size'            : 7.5,\n         'xtick.labelsize'      : 7.5,\n         'ytick.labelsize'      : 7.5,\n        }\n       )\n\n# Color printing\ndef PrintColor(text: str, color = Fore.BLUE, style = Style.BRIGHT):\n    \"Prints color outputs using colorama using a text F-string\"\n    print(style + color + text + Style.RESET_ALL)\n \n# Checking package versions\nimport xgboost as xgb, lightgbm as lgb, catboost as cb, sklearn as sk\nprint(f\"---> XGBoost = {xgb.__version__} | LightGBM = {lgb.__version__} | Catboost = {cb.__version__}\")\nprint(f\"---> Sklearn = {sk.__version__}| Pandas = {pd.__version__} | Polars = {pl.__version__}\")\ncollect()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-21T19:58:12.385727Z","iopub.execute_input":"2024-09-21T19:58:12.386510Z","iopub.status.idle":"2024-09-21T19:58:12.396008Z","shell.execute_reply.started":"2024-09-21T19:58:12.386467Z","shell.execute_reply":"2024-09-21T19:58:12.394830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nexec(open('myimports.py','r').read())\nprint()","metadata":{"execution":{"iopub.status.busy":"2024-09-21T19:58:12.397305Z","iopub.execute_input":"2024-09-21T19:58:12.397695Z","iopub.status.idle":"2024-09-21T19:58:16.974011Z","shell.execute_reply.started":"2024-09-21T19:58:12.397656Z","shell.execute_reply":"2024-09-21T19:58:16.972746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **APPROACH DETAILS**","metadata":{}},{"cell_type":"markdown","source":"This is a starter notebook for the **Child Mind Institute — Problematic Internet Use** code competition. This is a time series tabular assignment to detect and classify the impact of excessive internet usage on child sleep patterns. We use **Quadratic Weighted Kappa Score** as evaluation metric. This needs to be maximized <br>\n\nWe are provided a set of 3 datasets- <br> \n1. Parquet files for accelerometer train data, with separate tables for each id <br> \n2. Parquet files for accelerometer test data, with separate tables for each id <br>\n3. Metadata for remaining information and a separate table for data dictionary <br> \n\nIn this kernel, we start with the metadata and a few aggregate columns from the parquet file only and develop simple ML models as below- <br>\n1. The data has targets in 2 steps. In the first step, we make a regression model to predict a proxy target- **PCIAT_Total**. We have 20 target columns from PCIAT_1 - PCIAT_20 and the total column - **PCIAT_Total** <br>\n2. We predict each target using a separate model. So we have 21 first level models, one for each of the **PCIAT_1-20** columns and one for **PCIAT_Total**. We calculate PCIAT_Total as a sum of the calculated 20 targets and then blend this with the predicted PCIAT_Total to yield some additional signals in the data <br>\n3. In the second step, we split the **PCIAT_Total** column into 4 classes and estimate the actual target for the assignment <br> ","metadata":{}},{"cell_type":"markdown","source":"# **CONFIGURATION**","metadata":{}},{"cell_type":"code","source":"%%writefile -a training.py\n\n# Configuration class:-\nclass CFG:\n    \"\"\"\n    Configuration class for parameters and CV strategy for tuning and training\n    Some parameters may be unused here as this is a general configuration class\n    \"\"\";\n\n    # Data preparation:-\n    version_nb  = 1\n    model_id    = \"V1_3\"\n    model_label = \"ML\"\n\n    test_req           = False\n    test_sample_frac   = 1000\n\n    gpu_switch         = \"OFF\"\n    state              = 42\n    target             = f\"sii\"\n    grouper            = f\"sii\"\n\n    ip_path            = f\"/kaggle/input/child-mind-institute-problematic-internet-use\"\n    op_path            = f\"/kaggle/working\"\n\n    # Model Training:-\n    pstprcs_oof        = False\n    pstprcs_train      = False\n    pstprcs_test       = False\n    ML                 = True\n    test_preds_req     = True\n\n    pseudo_lbl_req     = False\n    pseudolbl_up       = 0.975\n    pseudolbl_low      = 0.00\n\n    n_splits           = 3 if test_req == \"Y\" else 5\n    n_repeats          = 1\n    nbrnd_erly_stp     = 40\n    mdlcv_mthd         = 'RSKF'\n\n    # Ensemble:-\n    ensemble_req       = True\n    optuna_req         = True\n    metric_obj         = 'maximize'\n    ntrials            = 10 if test_req == \"Y\" else 300\n\n    # Global variables for plotting:-\n    grid_specs = {'visible'  : True,\n                  'which'    : 'both',\n                  'linestyle': '--',\n                  'color'    : 'lightgrey',\n                  'linewidth': 0.75\n                 }\n\n    title_specs = {'fontsize'   : 9,\n                   'fontweight' : 'bold',\n                   'color'      : '#992600',\n                  }\n    \ncv_selector = \\\n{\n \"RKF\"   : RKF(n_splits = CFG.n_splits, n_repeats= CFG.n_repeats, random_state= CFG.state),\n \"RSKF\"  : RSKF(n_splits = CFG.n_splits, n_repeats= CFG.n_repeats, random_state= CFG.state),\n \"SKF\"   : SKF(n_splits = CFG.n_splits, shuffle = True, random_state= CFG.state),\n \"KF\"    : KFold(n_splits = CFG.n_splits, shuffle = True, random_state= CFG.state),\n}\n\nPrintColor(f\"\\n---> Configuration done!\\n\")\ncollect()","metadata":{"execution":{"iopub.status.busy":"2024-09-21T19:58:16.975380Z","iopub.execute_input":"2024-09-21T19:58:16.975973Z","iopub.status.idle":"2024-09-21T19:58:16.985087Z","shell.execute_reply.started":"2024-09-21T19:58:16.975931Z","shell.execute_reply":"2024-09-21T19:58:16.983767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"|Configuration parameter| Explanation| Data type| Sample values |  \n| ---------------------- | ------------------------------- | --------------------- | --------------- |\n| version_nb    | Version Number | int | 1 | \n| model_id      | Model ID    | string | V1_1 | \n| model_label   | Model Label | string | ML | \n| test_req      | Test Required| bool | True / False | \n| test_sample_frac| Test sampled fraction | int | 1000 |\n| gpu_switch      | Do we need GPU support | bool | True / False |\n| state           | Random state | int | 42 |\n| target          | Target column | str | sii |\n| grouper         | CV grouper column | str | sii |\n| ip_path, op_path | Data paths  | str | |\n| pstprcs_* | Do we need post-processing  | bool |True / False |\n| ML| Do we need machine learning models  | bool |True / False |\n| test_preds_req| Do we need test set predictions (training in inference kernel)  | bool |True / False |\n| pseudo_lbl_req| Pseudo label required?  | bool |True / False |\n| pseudo_lbl_* | Pseudo label cutoff | float | |\n| n_splits/ n_repeats | N-splits and repeats for CV scheme | int | 3/5/10|\n| nbrnd_erly_stp | Early stopping rounds | int | 40|\n| mdlcv_mthd | Model CV method | str | RSKF|\n| ensemble_req | Do we need ensemble | bool | True / False |\n| optuna_req   | Do we need optuna | bool | True / False |\n| metric_obj   | Metric direction | str | minimize/ maximize |\n| ntrials      | Trials | int | 300 |","metadata":{}},{"cell_type":"code","source":"%%writefile -a training.py\n\nclass Utils:\n    \"\"\"\n    This class creates and uses several utility methods to be used across the code\n    \"\"\";\n\n    def __init__(self):\n        pass\n\n    def ScoreMetric(self, ytrue, ypred)-> float:\n        \"\"\"\n        This method calculates the metric for the competition\n        Inputs- ytrue, ypred:- input truth and predictions\n        Output- float:- competition metric\n        \"\"\"\n        return cohen_kappa_score(ytrue, ypred, weights = \"quadratic\")\n\n    def CleanMemory(self):\n        \"This method cleans the memory off unused objects and displays the cleaned state RAM usage\"\n\n        collect();\n        libc.malloc_trim(0)\n        pid        = getpid()\n        py         = Process(pid)\n        memory_use = py.memory_info()[0] / 2. ** 30\n        return f\"\\nRAM usage = {memory_use :.4} GB\"\n\n    def DisplayAdjTbl(self, *args):\n        \"\"\"\n        This function displays pandas tables in an adjacent manner, sourced from the below link-\n        https://stackoverflow.com/questions/38783027/jupyter-notebook-display-two-pandas-tables-side-by-side\n        \"\"\"\n\n        html_str = ''\n        for df in args:\n            html_str += df.to_html()\n        display_html(html_str.replace('table','table style=\"display:inline\"'),raw=True)\n        collect()\n\n    def DisplayScores(\n        self, Scores: pd.DataFrame, TrainScores: pd.DataFrame, methods: list\n    ):\n        \"This method displays the scores and their means\"\n\n        args = \\\n        [Scores.style.format(precision = 5).\\\n         background_gradient(cmap = \"Blues\", subset = methods + [\"Ensemble\"]).\\\n         set_caption(f\"\\nOOF scores across methods and folds\\n\"),\n\n         TrainScores.style.format(precision = 5).\\\n         background_gradient(cmap = \"Pastel2\", subset = methods).\\\n         set_caption(f\"\\nTrain scores across methods and folds\\n\")\n        ];\n\n        PrintColor(f\"\\n\\n\\n---> OOF score across all methods and folds\\n\",\n                   color = Fore.LIGHTMAGENTA_EX\n                   )\n        self.DisplayAdjTbl(*args)\n\n        print('\\n')\n        display(Scores.mean().to_frame().\\\n                transpose().\\\n                style.format(precision = 5).\\\n                background_gradient(cmap = \"mako\", axis=1,\n                                    subset = Scores.columns\n                                   ).\\\n                set_caption(f\"\\nOOF mean scores across methods and folds\\n\")\n               )\n\n\nutils = Utils()\ncollect()\nprint()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-21T19:58:16.987005Z","iopub.execute_input":"2024-09-21T19:58:16.987379Z","iopub.status.idle":"2024-09-21T19:58:17.024001Z","shell.execute_reply.started":"2024-09-21T19:58:16.987341Z","shell.execute_reply":"2024-09-21T19:58:17.022781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **PREPROCESSING**","metadata":{}},{"cell_type":"code","source":"%%writefile -a training.py\n\ndef MakeAggs(\n    all_files: list, verbose: bool = False, ip_path: str = CFG.ip_path,\n)->pl.DataFrame:\n    \"This function collates the id level parquet files and creates the aggregation columns in a polars dataframe\"\n    \n    cols = [\"X\", \"Y\", \"Z\", \"enmo\", \"anglez\", \"light\", \"battery_voltage\"]\n    \n    for file_nb, file in enumerate(all_files):\n        df = \\\n        pl.scan_parquet(\n            os.path.join(ip_path, file)\n        ).select(pl.col(cols)).\\\n        collect().\\\n        describe(\n            percentiles = np.arange(0.05, 0.95, 0.10)\n        ).\\\n        filter(~pl.col(\"statistic\").is_in([\"count\", \"null_count\"])).\\\n        unpivot(index = \"statistic\").\\\n        with_columns(\n            pl.concat_str([pl.col(\"variable\"), pl.col(\"statistic\")],separator = \"_\",).alias(\"myvar\")\n        ).\\\n        with_columns(pl.col(\"myvar\").str.replace(r\"\\%\", \"\")).\\\n        select([\"myvar\", \"value\"]).\\\n        transpose(column_names = \"myvar\").\\\n        select(pl.all().shrink_dtype()).\\\n        with_columns(\n            pl.Series(\"id\", np.array(re.sub(\"id=\", \"\", file)))\n        )\n\n        if file_nb == 0:\n            op_df = df.clone()\n        elif file_nb > 0:\n            op_df = pl.concat([op_df, df], how = \"vertical_relaxed\")\n            \n            if verbose:\n                print(f\"---> Shapes = {op_df.shape}\")\n            else:\n                pass\n        del df\n    \n    print(f\"---> Shape = {op_df.shape}\")\n    return op_df\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-21T19:58:17.025517Z","iopub.execute_input":"2024-09-21T19:58:17.026000Z","iopub.status.idle":"2024-09-21T19:58:17.039618Z","shell.execute_reply.started":"2024-09-21T19:58:17.025944Z","shell.execute_reply":"2024-09-21T19:58:17.038255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\nexec(open('training.py','r').read())\ntrain  = pl.read_csv(os.path.join(CFG.ip_path, \"train.csv\")).drop(\"PCIAT-Season\")\ntest   = pl.read_csv(os.path.join(CFG.ip_path, \"test.csv\"))\nsub_fl = pl.read_csv(os.path.join(CFG.ip_path, \"sample_submission.csv\"))\n\nPrintColor(f\"---> Data Processing\")\nprint(f\"---> Shapes = {train.shape} {test.shape}\")\n\ncat_cols = train.select(cs.string().exclude(\"id\")).columns\ntrain = train.with_columns(pl.col(cat_cols).fill_null(\"missing\").cast(pl.Categorical))\ntest  = test.with_columns(pl.col(cat_cols).fill_null(\"missing\").cast(pl.Categorical))\n\ntargets  = \\\nsorted(\n    list(\n        set(train.columns).difference(set(test.columns)).difference(set([CFG.target]))\n    )\n)\n\n# Adding parquet file columns:- \nall_files = os.listdir(os.path.join(CFG.ip_path, f\"series_train.parquet\"))\nop_df = MakeAggs(all_files, verbose = False, \n                 ip_path = os.path.join(CFG.ip_path, f\"series_train.parquet\")\n                )\ntrain = train.join(op_df, how = \"left\", on = \"id\")\ndel op_df\n\nall_files = os.listdir(os.path.join(CFG.ip_path, f\"series_test.parquet\"))\nop_df = MakeAggs(all_files, verbose = False, \n                 ip_path = os.path.join(CFG.ip_path, f\"series_test.parquet\")\n                )\ntest  = test.join(op_df, how = \"left\", on = \"id\")\ndel op_df\n\nsel_cols = test.drop(\"id\").columns\ntrain    = train.select(pl.all().shrink_dtype())\ntest     = test.select(pl.all().shrink_dtype())\nPrintColor(f\"---> Shapes = {train.shape} {test.shape}\\n\")\n\nwith np.printoptions(linewidth = 100):\n    PrintColor(f\"\\n---> All targets\")\n    pprint(np.array(targets))\n    PrintColor(f\"\\n---> All category columns\")\n    pprint(np.array(cat_cols))    \n    \nPrintColor(utils.CleanMemory())","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-21T19:58:17.041344Z","iopub.execute_input":"2024-09-21T19:58:17.042069Z","iopub.status.idle":"2024-09-21T20:00:41.556850Z","shell.execute_reply.started":"2024-09-21T19:58:17.042013Z","shell.execute_reply":"2024-09-21T20:00:41.555536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **MODEL TRAINING**","metadata":{}},{"cell_type":"code","source":"%%writefile -a training.py\n\nclass ModelTrainer:\n    \"This class trains the provided model on the train-test data and returns the predictions and fitted models\"\n\n    def __init__(\n        self,\n        es             : int   = 100,\n        target         : str   = CFG.target,\n        metric_lbl     : str   = \"rmse\",\n        drop_cols      : list  = [\"Source\", \"id\", \"Id\", \"Label\", CFG.target, \"fold_nb\"],\n    ):\n        \"\"\"\n        Key parameters-\n        es_iter - early stopping rounds for boosted trees\n        \"\"\"\n        self.es_iter        = es\n        self.target         = target\n        self.drop_cols      = drop_cols\n        self.metric_lbl     = metric_lbl\n\n\n    def ScoreMetric(self, ytrue, ypred)->float:\n        \"\"\"\n        This is the metric function for the competition scoring\n        \"\"\"\n        if self.metric_lbl == \"rmse\":\n            return mse(ytrue, ypred, squared = False)\n        \n        elif self.metric_lbl == \"kappa\":\n            myscore = \\\n            cohen_kappa_score(\n                np.uint8(np.around(ytrue,0)),\n                np.uint8(np.around(ypred,0)),\n                weights = \"quadratic\",\n            )\n            return myscore\n\n    def PlotFtreImp(\n        self, ftreimp: pd.Series, method: str,\n        ntop: int = 50,\n        title_specs: dict = CFG.title_specs,\n        **params,\n    ):\n        \"This function plots the feature importances for the model provided\"\n\n        print()\n        fig, ax = plt.subplots(1, 1, figsize = (25, 7.5))\n\n        ftreimp.sort_values(ascending = False).\\\n        head(ntop).\\\n        plot.bar(ax = ax, color = \"blue\")\n        ax.set_title(f\"Feature Importances - {method}\", **title_specs)\n\n        plt.tight_layout()\n        plt.show()\n        print()\n\n    def PostProcessPreds(self, ypred):\n        \"This method post-processes predictions optionally\"\n        return ypred\n\n    def MakeOfflineModel(\n        self, X, y, ygrp, Xtest, mdl, method,\n        test_preds_req   : bool = True,\n        ftreimp_plot_req : bool = True,\n        ntop             : int  = 50,\n        **params,\n    ):\n        \"\"\"\n        This function trains the provided model on the dataset and cross-validates appropriately\n\n        Inputs-\n        X, y, ygrp       - training data components\n        Xtest            - test data\n        model            - model object for training\n        method           - model method label\n        test_preds_req   - boolean flag to extract test set predictions\n        ftreimp_plot_req - boolean flag to plot tree feature importances\n        ntop             - top n features for feature importances plot\n\n        Returns-\n        oof_preds, test_preds - prediction arrays\n        fitted_models         - fitted model list for test set\n        ftreimp               - feature importances across selected features\n        mdl_best_iter         - model average best iteration across folds\n        \"\"\"\n\n        oof_preds     = np.zeros(len(X))\n        test_preds    = []\n        mdl_best_iter = []\n        ftreimp       = 0\n        \n        scores, tr_scores, fitted_models = [], [], []\n        cv = PDS(ygrp)\n        n_splits = ygrp.nunique()\n\n        for fold_nb, (train_idx, dev_idx) in tqdm(enumerate(cv.split(X, y))):\n            Xtr   = X.iloc[train_idx]\n            Xdev  = X.iloc[dev_idx]\n            ytr   = y.iloc[train_idx]\n            ydev  = y.iloc[dev_idx]\n            model = clone(mdl)\n\n            if \"CB\" in method:\n                model.fit(Xtr, ytr,\n                          eval_set = [(Xdev, ydev)],\n                          verbose = 0,\n                          early_stopping_rounds = self.es_iter,\n                          )\n                best_iter = model.get_best_iteration()\n\n            elif \"LGB\" in method:\n                model.fit(Xtr, ytr,\n                          eval_set = [(Xdev, ydev)],\n                          callbacks = [log_evaluation(0),\n                                       early_stopping(stopping_rounds = self.es_iter, verbose = False,),\n                                       ],\n                          )\n                best_iter = model.best_iteration_\n\n            elif \"XGB\" in method:\n                model.fit(Xtr, ytr,\n                          eval_set = [(Xdev, ydev)],\n                          verbose  = 0,\n                          )\n                best_iter = model.best_iteration\n\n            else:\n                model.fit(Xtr, ytr)\n                best_iter = -1\n\n            fitted_models.append(model)\n\n            try:\n                ftreimp += model.feature_importances_\n            except:\n                pass\n\n            dev_preds = self.PostProcessPreds(model.predict(Xdev))\n            oof_preds[Xdev.index] = dev_preds\n\n            train_preds  = self.PostProcessPreds(model.predict(Xtr))\n            tr_score     = self.ScoreMetric(ytr.values.flatten(), train_preds)\n            score        = self.ScoreMetric(ydev.values.flatten(), dev_preds)\n\n            scores.append(score)\n            tr_scores.append(tr_score)\n\n            nspace = 15 - len(method) - 2 if fold_nb <= 9 else 15 - len(method) - 1\n            PrintColor(f\"{method} Fold{fold_nb} {' ' * nspace} OOF = {score:.6f} | Train = {tr_score:.6f} | Iter = {best_iter:,.0f} \")\n            mdl_best_iter.append(best_iter)\n\n            if test_preds_req:\n                test_preds.append(\n                    self.PostProcessPreds(\n                        model.predict(Xtest)\n                    )\n                )\n            else:\n                pass\n\n        test_preds    = np.mean(np.stack(test_preds, axis = 1), axis=1)\n        ftreimp       = pd.Series(ftreimp, index = Xdev.columns)\n        mdl_best_iter = np.uint16(np.amax(mdl_best_iter))\n\n        if ftreimp_plot_req :\n            print()\n            self.PlotFtreImp(ftreimp, method = method, ntop = ntop,)\n        else:\n            pass\n\n        PrintColor(f\"\\n---> {np.mean(scores):.6f} +- {np.std(scores):.6f} | OOF\", color = Fore.RED)\n        PrintColor(f\"---> {np.mean(tr_scores):.6f} +- {np.std(tr_scores):.6f} | Train\", color = Fore.RED)\n\n        if mdl_best_iter < 0:\n            pass\n        else:\n            PrintColor(f\"---> Max best iteration = {mdl_best_iter :,.0f}\", color = Fore.RED)\n\n        return (fitted_models, oof_preds, test_preds, ftreimp, mdl_best_iter)\n\n    def MakeOnlineModel(\n        self, X, y, Xtest, model, method,\n        test_preds_req : bool = False,\n    ):\n        \"This method refits the model on the complete train data and returns the model fitted object and predictions\"\n\n        try:\n            model.early_stopping_rounds = None\n        except:\n            pass\n\n        try:\n            model.fit(X, y, verbose = 0)\n        except:\n            model.fit(X, y,)\n\n        oof_preds  = model.predict(X)\n        if test_preds_req:\n            test_preds = model.predict(Xtest[X.columns])\n        else:\n            test_preds = 0\n        return (model, oof_preds, test_preds)\n\n    def MakeOfflinePreds(self, X, fitted_models):\n        \"This method creates test-set predictions for the offline model provided\"\n\n        test_preds = 0\n        n_splits   = len(fitted_models)\n        PrintColor(f\"---> Number of splits = {n_splits}\")\n\n        for model in fitted_models:\n            test_preds = test_preds + (model.predict(X) / n_splits)\n\n        return test_preds","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-21T20:00:41.560999Z","iopub.execute_input":"2024-09-21T20:00:41.561424Z","iopub.status.idle":"2024-09-21T20:00:41.575183Z","shell.execute_reply.started":"2024-09-21T20:00:41.561374Z","shell.execute_reply":"2024-09-21T20:00:41.573743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a training.py\n\nclass OptunaEnsembler:\n    \"\"\"\n    This is the Optuna ensemble class-\n    Source- https://www.kaggle.com/code/arunklenin/ps3e26-cirrhosis-survial-prediction-multiclass\n    \"\"\";\n\n    def __init__(\n        self, state: int = 42, ntrials: int = 300, \n        metric_obj: str = \"maximize\", \n        metric_lbl: str = \"rmse\",\n        **params\n    ):\n        self.study        = None\n        self.weights      = None\n        self.random_state = state\n        self.n_trials     = ntrials\n        self.direction    = metric_obj\n        self.metric_lbl   = metric_lbl\n\n    def ScoreMetric(self, ytrue, ypred)->float:\n        \"\"\"\n        This is the metric function for the competition\n        \"\"\"\n        \n        if self.metric_lbl == \"rmse\":\n            return mse(ytrue, ypred, squared = False)\n        else:\n            myscore = \\\n            cohen_kappa_score(\n                np.uint8(np.around(ytrue,0)),\n                np.uint8(np.around(ypred,0)), \n                weights = \"quadratic\"\n            )\n            return myscore\n\n    def _objective(\n        self, trial, y_true, y_preds\n    ):\n        \"\"\"\n        This method defines the objective function for the ensemble\n        \"\"\";\n\n        if isinstance(y_preds, pd.DataFrame) or isinstance(y_preds, np.ndarray):\n            weights = [trial.suggest_float(f\"weight{n}\", 0.001, 0.999)\n                       for n in range(y_preds.shape[-1])\n                      ]\n            axis = 1\n\n        elif isinstance(y_preds, list):\n            weights = [trial.suggest_float(f\"weight{n}\", 0.001, 0.999)\n                       for n in range(len(y_preds))\n                      ]\n            axis = 0\n\n        # Calculating the weighted prediction:-\n        weighted_pred  = np.average(np.array(y_preds), axis = axis, weights = weights)\n        score          = self.ScoreMetric(y_true, weighted_pred)\n        return score\n\n    def fit(self, y_true, y_preds):\n        \"This method fits the Optuna objective on the fold level data\";\n\n        optuna.logging.set_verbosity = optuna.logging.ERROR\n\n        self.study = \\\n        optuna.create_study(sampler    = TPESampler(seed = self.random_state),\n                            pruner     = HyperbandPruner(),\n                            study_name = \"Ensemble\",\n                            direction  = self.direction,\n                           )\n\n        obj = partial(self._objective, y_true = y_true, y_preds = y_preds)\n        self.study.optimize(obj, n_trials = self.n_trials)\n\n        if isinstance(y_preds, list):\n            self.weights = [self.study.best_params[f\"weight{n}\"] for n in range(len(y_preds))]\n\n        else:\n            self.weights = [self.study.best_params[f\"weight{n}\"] for n in range(y_preds.shape[-1])]\n\n    def predict(self, y_preds):\n        \"This method predicts using the fitted Optuna objective\";\n\n        assert self.weights is not None, 'OptunaWeights error, must be fitted before predict';\n\n        if isinstance(y_preds, list):\n            weighted_pred = np.average(np.array(y_preds), axis=0, weights = self.weights)\n\n        else:\n            weighted_pred = np.average(np.array(y_preds), axis=1, weights = self.weights)\n\n        return weighted_pred\n\n    def fit_predict(self, y_true, y_preds):\n        \"\"\"\n        This method fits the Optuna objective on the fold data, then predicts the test set\n        \"\"\";\n        self.fit(y_true, y_preds)\n        return self.predict(y_preds)\n\n    def weights(self):\n        \"This method returns the non-normalized weights for all models in a fold\"\n        return self.weights\n\nprint()\ncollect();","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-21T20:00:41.576887Z","iopub.execute_input":"2024-09-21T20:00:41.577378Z","iopub.status.idle":"2024-09-21T20:00:41.597584Z","shell.execute_reply.started":"2024-09-21T20:00:41.577321Z","shell.execute_reply":"2024-09-21T20:00:41.596259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a training.py\n\ndef NormWeights(weights: dict, methods: list):\n    \"This function normalizes the weights and returns a dataframe of normalized weights across folds and models\"\n\n    weights = pd.DataFrame.from_dict(weights).T\n    weights[\"row_sum\"] = weights.sum(axis=1)\n\n    for col in weights.columns:\n        weights[col] = weights[col] / weights[\"row_sum\"]\n\n    weights.drop(\"row_sum\", axis = 1, inplace = True, errors = \"ignore\")\n    weights.columns    = methods\n    weights.index.name = \"Fold_Nb\"\n    return weights\n\ndef MakeEnsemble(\n    target: str, metric_obj: str, metric_lbl: str = \"rmse\"\n):\n    \"This function implements the Optuna ensemble on the OOF and test prediction datasets\"\n\n    global OOF_Preds, Mdl_Preds\n\n    PrintColor(f\"\\n{'=' * 20} ENSEMBLE {'=' * 20}\\n\")\n    \n    ygrp       = OOF_Preds[\"fold_nb\"]\n    cv         = PDS(ygrp)\n    oof_preds  = np.zeros(len(OOF_Preds))\n    test_preds = []\n    scores     = []\n    weights    = {}\n    drop_cols  = [\"fold_nb\", target, \"Ensemble\"]\n    n_splits   = ygrp.nunique()\n\n    for fold_nb, (_, dev_idx) in tqdm(enumerate(cv.split(OOF_Preds, OOF_Preds[target]))):\n        Xdev = OOF_Preds.iloc[dev_idx].drop(drop_cols, axis=1, errors = \"ignore\")\n        ydev = OOF_Preds.loc[dev_idx, target]\n\n        ens = OptunaEnsembler(\n            ntrials = CFG.ntrials, metric_lbl = metric_lbl, metric_obj = metric_obj\n        )\n        ens.fit(ydev, Xdev,)\n\n        dev_preds = ens.predict(Xdev)\n        score     = ens.ScoreMetric(ydev.values, dev_preds)\n        oof_preds[dev_idx] = dev_preds\n        test_preds.append(\n            ens.predict(Mdl_Preds.drop(drop_cols, axis=1, errors = \"ignore\"))\n        )\n\n        PrintColor(f\"---> {score: .6f} | Fold {fold_nb}\", color = Fore.CYAN)\n        scores.append(score)\n\n        weights[f\"Fold{fold_nb}\"] = ens.weights\n\n    PrintColor(f\"\\n---> OOF = {np.mean(scores): .6f} +- {np.std(scores): .6f} | Ensemble\",\n               color = Fore.RED\n              )\n\n    test_preds = np.mean(np.stack(test_preds, axis=1), axis=1,)\n\n    OOF_Preds[\"Ensemble\"] = oof_preds\n    Mdl_Preds[\"Ensemble\"] = test_preds\n\n    weights = \\\n    NormWeights(\n        weights,\n        methods = Mdl_Preds.drop(drop_cols, axis=1, errors = \"ignore\").columns\n    )\n\n    print(\"\\n\\n\\n\")\n    display(\n        weights.\\\n        style.\\\n        set_caption(\"Normalized weights\").\\\n        format(precision = 6).\\\n        set_properties(\n            props = \"color:red; background-color:white; font-weight: bold; border: maroon dashed 1.6px\"\n        )\n    )\n    \n    print()\n    display(\n        weights.mean().to_frame().transpose().\\\n        style.\\\n        format(precision = 6).\\\n        set_caption(\"Normalized Mean weights\").\\\n        set_properties(\n            props = \"color:red; background-color:white; font-weight: bold; border: maroon dashed 1.6px\"\n        )        \n    )\n\n    return weights\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-21T20:00:41.599384Z","iopub.execute_input":"2024-09-21T20:00:41.599861Z","iopub.status.idle":"2024-09-21T20:00:41.617783Z","shell.execute_reply.started":"2024-09-21T20:00:41.599819Z","shell.execute_reply":"2024-09-21T20:00:41.616477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a training.py\n\nclass OptimizedRounder:\n    \"\"\"\n    Source - https://www.kaggle.com/code/tubotubo/starter-notebook-multi-target-prediction\n    \"\"\"\n\n    def __init__(\n        self, n_classes: int, n_trials: int = 100, direction : str = \"maximize\"\n    ):\n        self.n_classes  = n_classes\n        self.labels     = np.arange(n_classes)\n        self.n_trials   = n_trials\n        self.metric     = partial(cohen_kappa_score, weights=\"quadratic\")\n        self.direction  = direction\n        \n    def _objective(\n        self, trial: optuna.Trial, y_true: NDArray[np.int_], y_pred: NDArray[np.float_],\n    ) -> float:\n        \n        thresholds = []\n        for i in range(self.n_classes - 1):\n            low  = max(thresholds) if i > 0 else min(self.labels)\n            high = max(self.labels)\n            th   = trial.suggest_float(f\"threshold_{i}\", low, high)\n            thresholds.append(th)\n            \n        try:\n            y_pred_rounded = np.digitize(y_pred, thresholds)\n        except ValueError:\n            return -100\n        return self.metric(y_true, y_pred_rounded)\n\n    def fit(\n        self, y_pred: NDArray[np.float_], y_true: NDArray[np.int_]\n    ) -> None:\n        y_pred = self._normalize(y_pred)\n        study  = optuna.create_study(direction = self.direction)\n        obj    = partial(self._objective, y_true = y_true, y_pred = y_pred)\n        \n        study.optimize(obj, n_trials = self.n_trials)\n        self.thresholds = [study.best_params[f\"threshold_{i}\"] for i in range(self.n_classes - 1)]\n\n    def predict(self, y_pred: NDArray[np.float_]) -> NDArray[np.int_]:\n        assert hasattr(self, \"thresholds\"), \"fit() must be called before predict()\"\n        y_pred = self._normalize(y_pred)\n        return np.digitize(y_pred, self.thresholds)\n\n    def _normalize(self, y: NDArray[np.float_]) -> NDArray[np.float_]:\n        return (y - y.min()) / (y.max() - y.min()) * (self.n_classes - 1)\n    \n    def thresholds(self):\n        return self.thresholds\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-21T20:00:41.619419Z","iopub.execute_input":"2024-09-21T20:00:41.619932Z","iopub.status.idle":"2024-09-21T20:00:41.635813Z","shell.execute_reply.started":"2024-09-21T20:00:41.619880Z","shell.execute_reply":"2024-09-21T20:00:41.634594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **MODEL I-O**","metadata":{}},{"cell_type":"code","source":"%%capture\n\nexec(open('training.py','r').read())\n\n# Initializing CV scheme\ncv = cv_selector[CFG.mdlcv_mthd]\n\n# Initializing model parameters\nMdl_Master = \\\n{\n f'CB1R': CBR(**{'task_type'           : \"GPU\" if CFG.gpu_switch == \"ON\" else \"CPU\",\n                 'loss_function'       : 'RMSE',\n                 'eval_metric'         : \"RMSE\",\n                 'bagging_temperature' : 0.25,\n                 'colsample_bylevel'   : 0.60,\n                 'iterations'          : 5_000,\n                 'learning_rate'       : 0.028,\n                 'max_depth'           : 7,\n                 'l2_leaf_reg'         : 0.25,\n                 'min_data_in_leaf'    : 13,\n                 'random_strength'     : 0.25,\n                 'verbose'             : 0,\n                 'use_best_model'      : True,\n                 'cat_features'        : cat_cols,\n                }\n             ),\n}\n\n# Initializing model outputs\nOOF_Preds    = {}\nMdl_Preds    = {}\nFittedModels = {}\nFtreImp      = {}\nSelMdlCols   = {}","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-21T20:00:41.637420Z","iopub.execute_input":"2024-09-21T20:00:41.637912Z","iopub.status.idle":"2024-09-21T20:00:42.027745Z","shell.execute_reply.started":"2024-09-21T20:00:41.637859Z","shell.execute_reply":"2024-09-21T20:00:42.026464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **MODEL CV ANALYSIS**","metadata":{}},{"cell_type":"code","source":"mytrain       = train.to_pandas().dropna(subset = [CFG.target] + targets)\nmytrain.index = range(len(mytrain))\nmytest        = test.to_pandas()[sel_cols]\nmytarget      = \"PCIAT-PCIAT_Total\"\n\nPrintColor(f\"---> Shapes = {mytrain.shape} {mytest.shape}\", color = Fore.CYAN)\n\n# Initializing CV folds across the training data:-\nfolds = np.zeros(len(mytrain))\nfor fold_nb, (train_idx, dev_idx) in enumerate(cv.split(mytrain, mytrain[mytarget])):\n    folds[dev_idx] = fold_nb\nmytrain[\"fold_nb\"] = folds\ndel folds","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-21T20:00:42.029830Z","iopub.execute_input":"2024-09-21T20:00:42.030275Z","iopub.status.idle":"2024-09-21T20:00:42.104060Z","shell.execute_reply.started":"2024-09-21T20:00:42.030232Z","shell.execute_reply":"2024-09-21T20:00:42.102778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\nfor tgt_nb, mytarget in tqdm(enumerate(targets)):\n      \n    # Creating CV scheme:-\n    md = ModelTrainer(es = CFG.nbrnd_erly_stp, target = mytarget)\n\n    for method, mdl in tqdm(Mdl_Master.items()):\n        PrintColor(f\"\\n{'-' * 10} {method} MODEL TRAINING - {mytarget} {'-' * 10}\\n\", \n                   color = Fore.MAGENTA\n                  )\n        \n        fitted_models, oof_preds, test_preds, ftreimp, mdl_best_iter =  \\\n        md.MakeOfflineModel(\n            mytrain[sel_cols],\n            mytrain[mytarget],\n            mytrain[\"fold_nb\"],\n            mytest,\n            clone(mdl),\n            method,\n            test_preds_req   = True,\n            ftreimp_plot_req = True,\n            ntop = 50,\n        ) \n    \n        # Integrating data    \n        OOF_Preds[f\"{method}_{mytarget}\"]    = oof_preds\n        Mdl_Preds[f\"{method}_{mytarget}\"]    = test_preds\n        FtreImp[f\"{method}_{mytarget}\"]      = ftreimp\n        FittedModels[f\"{method}_{mytarget}\"] = fitted_models\n    \n        del fitted_models, oof_preds, test_preds, ftreimp, mdl_best_iter\n        _ = utils.CleanMemory()\n    \nPrintColor(utils.CleanMemory())    ","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-21T20:03:10.959779Z","iopub.execute_input":"2024-09-21T20:03:10.960294Z","iopub.status.idle":"2024-09-21T20:19:11.066040Z","shell.execute_reply.started":"2024-09-21T20:03:10.960250Z","shell.execute_reply":"2024-09-21T20:19:11.064883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **THRESHOLD TUNING**","metadata":{}},{"cell_type":"markdown","source":"In this section, we try and establish the best CV based cutoffs to convert the continuous predictions to labels for the target prediction <br>\nHere, we use the predicted values for the proxy target **PCIAT-PCIAT_Total** as a base column and unearth the best values for **sii**, our actual target","metadata":{}},{"cell_type":"code","source":"%%time \n\nOOF_Preds = pd.DataFrame.from_dict(OOF_Preds, orient = \"columns\")\nMdl_Preds = pd.DataFrame.from_dict(Mdl_Preds, orient = \"columns\")\n\n# Creating the mean PCIAT_Total from component models and direct predictions:-\noof_preds = \\\nnp.mean(\n    np.stack(\n        [OOF_Preds.iloc[:, 0: -1].sum(axis=1).values, \n         OOF_Preds.iloc[:, -1].values\n        ],axis=1\n    ), axis=1\n)\n\nmdl_preds = \\\nnp.mean(\n    np.stack(\n        [Mdl_Preds.iloc[:, 0: -1].sum(axis=1).values, \n         Mdl_Preds.iloc[:, -1].values\n        ],axis=1\n    ), axis=1\n)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-21T20:19:11.068206Z","iopub.execute_input":"2024-09-21T20:19:11.069451Z","iopub.status.idle":"2024-09-21T20:19:11.093698Z","shell.execute_reply.started":"2024-09-21T20:19:11.069401Z","shell.execute_reply":"2024-09-21T20:19:11.092460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\nmytuner = OptimizedRounder(n_classes = 4, n_trials = CFG.ntrials)\nytrain  = np.uint8(mytrain[CFG.target])\n\nmytuner.fit(oof_preds, ytrain)\nens_preds  = mytuner.predict(oof_preds)\ntest_preds = mytuner.predict(mdl_preds)\n\n# Displaying the confusion matrix \nscore = utils.ScoreMetric(ytrain, ens_preds)\nPrintColor(f\"\\n---> Final ensemble OOF score = {score :.6f}\\n\\n\")\n\nfig, ax = plt.subplots(1,1, figsize = (5,5))\ndisp = \\\nConfusionMatrixDisplay(\n    confusion_matrix = confusion_matrix(ytrain, ens_preds),  \n    display_labels = list(range(4))\n)\n\ndisp.plot(\n    cmap = \"Blues\", \n    ax = ax, \n    colorbar = False, \n    xticks_rotation = 0,\n    text_kw = {\"fontweight\": \"bold\", \n               \"fontfamily\": \"Cambria\", \n               \"fontsize\"  : 16,\n              }\n)\nax.set_title(\n    f\"Confusion matrix - CV = {score :.6f}\", **CFG.title_specs\n)\nax.grid(**CFG.grid_specs)\nax.set(ylabel = f\"True {CFG.target}\", \n       xlabel = f\"Predicted {CFG.target}\", \n      )\nplt.show()\n\n_ = utils.CleanMemory()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-21T20:19:11.095165Z","iopub.execute_input":"2024-09-21T20:19:11.095590Z","iopub.status.idle":"2024-09-21T20:19:19.761796Z","shell.execute_reply.started":"2024-09-21T20:19:11.095515Z","shell.execute_reply":"2024-09-21T20:19:19.760574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **SUBMISSION**","metadata":{}},{"cell_type":"code","source":"%%time \n\nprint()\ndisplay(\n    OOF_Preds.head().style.format(precision = 3).set_caption(\"OOF Predictions\")\n)\n\nprint()\ndisplay(\n    Mdl_Preds.head().style.format(precision = 3).set_caption(\"Model Predictions\")\n)\n\nsub_fl.with_columns(\n    pl.Series(CFG.target, test_preds.flatten(), pl.UInt8)\n).write_csv(\"submission.csv\")\n\nprint()\n!ls \nprint(f\"\\n\\n---> Submission file\\n\\n\")\n!head submission.csv\n\nPrintColor(utils.CleanMemory())","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-21T20:19:19.764373Z","iopub.execute_input":"2024-09-21T20:19:19.764910Z","iopub.status.idle":"2024-09-21T20:19:22.433839Z","shell.execute_reply.started":"2024-09-21T20:19:19.764852Z","shell.execute_reply":"2024-09-21T20:19:22.432272Z"},"trusted":true},"execution_count":null,"outputs":[]}]}