{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.11"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":102335,"databundleVersionId":12518947,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":178.729812,"end_time":"2025-05-30T15:48:48.643233","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-05-30T15:45:49.913421","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.008042,"end_time":"2025-05-30T15:45:55.214338","exception":false,"start_time":"2025-05-30T15:45:55.206296","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **PACKAGE INSTALLATIONS**","metadata":{"papermill":{"duration":0.005364,"end_time":"2025-05-30T15:45:55.226486","exception":false,"start_time":"2025-05-30T15:45:55.221122","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%writefile req_kaggle.txt\n\nscikit-learn==1.6.1\nxgboost==3.0.1\nlightgbm==4.6.0\nnumpy==1.26.4\nscipy==1.14.1\npolars==1.29.0\npytorch_tabnet\ntabpfn","metadata":{"execution":{"iopub.execute_input":"2025-05-30T15:45:55.237874Z","iopub.status.busy":"2025-05-30T15:45:55.237540Z","iopub.status.idle":"2025-05-30T15:45:55.246688Z","shell.execute_reply":"2025-05-30T15:45:55.245642Z"},"papermill":{"duration":0.015856,"end_time":"2025-05-30T15:45:55.248157","exception":false,"start_time":"2025-05-30T15:45:55.232301","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a req_colab.txt\n\nxgboost==3.0.1\ncatboost==1.2.7\nnumpy==1.26.4\nlightgbm==4.6.0\nscipy==1.14.1\npolars==1.29.0\ncolorama\ncloudpickle\noptuna\npytorch_tabnet\nscikit-lego\ntabpfn","metadata":{"execution":{"iopub.execute_input":"2025-05-30T15:45:55.258420Z","iopub.status.busy":"2025-05-30T15:45:55.258137Z","iopub.status.idle":"2025-05-30T15:45:55.263778Z","shell.execute_reply":"2025-05-30T15:45:55.262917Z"},"papermill":{"duration":0.012426,"end_time":"2025-05-30T15:45:55.265173","exception":false,"start_time":"2025-05-30T15:45:55.252747","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile req_ag.txt\n\npolars==1.26.0\ncolorama\noptuna\ncatboost==1.2.7\nautogluon.tabular\nray==2.10.0\ndask","metadata":{"execution":{"iopub.execute_input":"2025-05-30T15:45:55.275642Z","iopub.status.busy":"2025-05-30T15:45:55.275338Z","iopub.status.idle":"2025-05-30T15:45:55.281474Z","shell.execute_reply":"2025-05-30T15:45:55.280613Z"},"papermill":{"duration":0.012946,"end_time":"2025-05-30T15:45:55.282798","exception":false,"start_time":"2025-05-30T15:45:55.269852","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a req_lama.txt\n\nlightautoml\ncolorama\npolars==1.26.0","metadata":{"execution":{"iopub.execute_input":"2025-05-30T15:45:55.293416Z","iopub.status.busy":"2025-05-30T15:45:55.293149Z","iopub.status.idle":"2025-05-30T15:45:55.298414Z","shell.execute_reply":"2025-05-30T15:45:55.297506Z"},"papermill":{"duration":0.01232,"end_time":"2025-05-30T15:45:55.299847","exception":false,"start_time":"2025-05-30T15:45:55.287527","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **PACKAGE DOWNLOADS**","metadata":{"papermill":{"duration":0.004334,"end_time":"2025-05-30T15:45:55.308839","exception":false,"start_time":"2025-05-30T15:45:55.304505","status":"completed"},"tags":[]}},{"cell_type":"code","source":"!pip download -q -r req_kaggle.txt -d /kaggle/working/packages\n!pip download -q -r req_ag.txt -d /kaggle/working/ag\n!pip download -q -r req_lama.txt -d /kaggle/working/lama","metadata":{"execution":{"iopub.execute_input":"2025-05-30T15:45:55.318750Z","iopub.status.busy":"2025-05-30T15:45:55.318473Z","iopub.status.idle":"2025-05-30T15:48:45.587707Z","shell.execute_reply":"2025-05-30T15:48:45.586483Z"},"papermill":{"duration":170.276845,"end_time":"2025-05-30T15:48:45.590047","exception":false,"start_time":"2025-05-30T15:45:55.313202","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **IMPORTS**","metadata":{"papermill":{"duration":0.115585,"end_time":"2025-05-30T15:48:45.889832","exception":false,"start_time":"2025-05-30T15:48:45.774247","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%writefile -a myimports.py\n\nprint(f\"\\n---> Commencing imports-part1\")\n\nfrom gc import collect\nfrom warnings import filterwarnings\nfilterwarnings('ignore')\nfrom IPython.display import display_html, clear_output\nclear_output()\nimport os, sys, logging, re, joblib, ctypes, shutil, random, torch\nfrom copy import deepcopy\n\nimport xgboost as xgb, lightgbm as lgb, catboost as cb, sklearn as sk, pandas as pd\nprint(f\"---> Sklearn = {sk.__version__}| Pandas = {pd.__version__}\")\ncollect()\n\n# General library imports:-\nfrom warnings import filterwarnings\nfilterwarnings('ignore')\nfrom gc import collect\n\nfrom os import path, walk, getpid\nfrom psutil import Process\nimport re\nfrom collections import Counter\nfrom itertools import product, combinations\n\nimport ctypes\nlibc = ctypes.CDLL(\"libc.so.6\")\n\nfrom IPython.display import display_html, clear_output\nfrom pprint import pprint\nfrom functools import partial\nfrom copy import deepcopy\nimport pandas as pd, numpy as np\nfrom scipy.stats import pearsonr\nimport polars as pl\nimport polars.selectors as cs\n\nfrom warnings import filterwarnings\nfilterwarnings('ignore')\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom colorama import Fore, Style, init\nfrom warnings import filterwarnings\nfilterwarnings('ignore')\nfrom tqdm.notebook import tqdm\n\nprint(f\"---> Imports- part 1 done\")","metadata":{"execution":{"iopub.execute_input":"2025-05-30T15:48:46.121312Z","iopub.status.busy":"2025-05-30T15:48:46.120105Z","iopub.status.idle":"2025-05-30T15:48:46.128926Z","shell.execute_reply":"2025-05-30T15:48:46.128031Z"},"papermill":{"duration":0.125245,"end_time":"2025-05-30T15:48:46.130555","exception":false,"start_time":"2025-05-30T15:48:46.005310","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a myimports.py\n\n# Pipeline specifics:-\nfrom sklearn.preprocessing import *\n\nfrom sklearn.impute import SimpleImputer as SI\nfrom sklearn.model_selection import *\nfrom sklearn.inspection import permutation_importance\nfrom sklearn.feature_selection import VarianceThreshold as VT\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.pipeline import Pipeline, make_pipeline\nfrom sklearn.base import BaseEstimator, TransformerMixin, clone\nfrom sklearn.compose import ColumnTransformer, make_column_selector\n\n# ML Model training:-\nfrom sklearn.metrics import *\n\nfrom xgboost import QuantileDMatrix, XGBClassifier as XGBC, XGBRegressor as XGBR\nfrom lightgbm import log_evaluation, early_stopping, LGBMClassifier as LGBMC, LGBMRegressor as LGBMR\nfrom catboost import CatBoostClassifier as CBC, Pool, CatBoostRegressor as CBR\nfrom sklearn.ensemble import HistGradientBoostingClassifier as HGBC, RandomForestClassifier as RFC\nfrom sklearn.ensemble import HistGradientBoostingRegressor as HGBR, RandomForestRegressor as RFR\nfrom sklearn.ensemble import VotingRegressor as VR, VotingClassifier as VC\nfrom sklearn.linear_model import LogisticRegression as LRC, Ridge, Lasso\nfrom sklearn.neighbors import KNeighborsClassifier as KNNC, KNeighborsRegressor as KNNR\n\n# TabNet models\nfrom pytorch_tabnet.tab_model import (TabNetRegressor as TNR, TabNetClassifier as TNC)\n\n# TabPFN models\nfrom tabpfn import TabPFNClassifier as TPFNC\n\n# Ensemble and tuning:-\nimport optuna\nfrom optuna import Trial, trial, create_study\nfrom optuna.pruners import HyperbandPruner\nfrom optuna.samplers import TPESampler, CmaEsSampler\n\n# Setting rc parameters in seaborn for plots and graphs-\nsns.set({\"axes.facecolor\"       : \"white\",\n         \"figure.facecolor\"     : \"#ffffff\",\n         \"axes.edgecolor\"       : \"black\",\n         \"grid.color\"           : '#b0b0b0',\n         \"font.family\"          : ['Cambria'],\n         \"axes.labelcolor\"      : \"#000000\",\n         \"xtick.color\"          : \"#000000\",\n         \"ytick.color\"          : \"#000000\",\n         \"grid.linewidth\"       : 0.50,\n         \"grid.linestyle\"       : \"--\",\n         \"axes.titlecolor\"      : 'maroon',\n         'axes.titlesize'       : 9,\n         'axes.labelweight'     : \"bold\",\n         'legend.fontsize'      : 7.0,\n         'legend.title_fontsize': 7.0,\n         'font.size'            : 7.5,\n         'xtick.labelsize'      : 12.5,\n         'ytick.labelsize'      : 9.0,\n        }\n       )\n\n# Color printing\ndef PrintColor(text: str, color = Fore.BLUE, style = Style.BRIGHT):\n    \"Prints color outputs using colorama using a text F-string\"\n    print(style + color + text + Style.RESET_ALL)\n","metadata":{"execution":{"iopub.execute_input":"2025-05-30T15:48:46.356710Z","iopub.status.busy":"2025-05-30T15:48:46.356350Z","iopub.status.idle":"2025-05-30T15:48:46.363996Z","shell.execute_reply":"2025-05-30T15:48:46.363248Z"},"papermill":{"duration":0.141035,"end_time":"2025-05-30T15:48:46.366068","exception":false,"start_time":"2025-05-30T15:48:46.225033","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a myimports.py\n\nprint(f\"---> Commencing imports-part2\")\noptuna.logging.set_verbosity = optuna.logging.ERROR\noptuna.logging.disable_default_handler()\nprint(f\"---> XGBoost = {xgb.__version__} | LightGBM = {lgb.__version__}\")\n\n##################################################################\n# Customizing logging for LGBM\nclass MyLogger:\n    \"\"\"\n    This class helps to suppress logs in lightgbm and Optuna\n    Source - https://github.com/microsoft/LightGBM/issues/6014\n    \"\"\"\n\n    def init(self, logging_lbl: str):\n        self.logger = logging.getLogger(logging_lbl)\n        self.logger.setLevel(logging.ERROR)\n\n    def info(self, message):\n        pass\n\n    def warning(self, message):\n        pass\n\n    def error(self, message):\n        self.logger.error(message)\n\nl = MyLogger()\nl.init(logging_lbl = \"lightgbm_custom\")\nlgb.register_logger(l)\n\n##################################################################\n# Customizing logging for XGBoost\nfor handler in logging.root.handlers[:]:\n    logging.root.removeHandler(handler)\n\nlogger = logging.getLogger(__name__)\nlogger.setLevel(logging.ERROR)\nformatter = logging.Formatter('%(asctime)s | %(levelname)s | %(message)s')\n\nstdout_handler = logging.StreamHandler(sys.stdout)\nstdout_handler.setLevel(logging.INFO)\nstdout_handler.setFormatter(formatter)\n\nfile_handler = logging.FileHandler(f'xgb_optimize.log')\nfile_handler.setLevel(logging.ERROR)\nfile_handler.setFormatter(formatter)\n\nlogger.addHandler(file_handler)\nlogger.addHandler(stdout_handler)\n\nclass XGBLogging(xgb.callback.TrainingCallback):\n    \"\"\"log train logs to file\"\"\"\n\n    def __init__(self, epoch_log_interval=100):\n        self.epoch_log_interval = epoch_log_interval\n\n    def after_iteration(self, model, epoch:int,\n                        evals_log:xgb.callback.TrainingCallback.EvalsLog\n                        ):\n\n        if self.epoch_log_interval <= 0:\n            pass\n\n        elif (epoch %  self.epoch_log_interval == 0):\n            for data, metric in evals_log.items():\n                for metric_name, log in metric.items():\n                    score = log[-1][0] if isinstance(log[-1], tuple) else log[-1]\n                    logger.info(f\"XGBLogging epoch {epoch} dataset {data} {metric_name} {score}\")\n\n        return False\n\n# Making sklearn pipeline outputs as dataframe:-\nfrom sklearn import set_config\npd.set_option('display.max_columns', 1000)\npd.set_option('display.max_rows', 200)\nprint(f\"---> Imports- part 2 done\")\n","metadata":{"execution":{"iopub.execute_input":"2025-05-30T15:48:46.509897Z","iopub.status.busy":"2025-05-30T15:48:46.509520Z","iopub.status.idle":"2025-05-30T15:48:46.516693Z","shell.execute_reply":"2025-05-30T15:48:46.515784Z"},"papermill":{"duration":0.080842,"end_time":"2025-05-30T15:48:46.518257","exception":false,"start_time":"2025-05-30T15:48:46.437415","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a myimports.py\n\nprint(f\"---> Seeding everything\")\n\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\n\nseed_everything(2024)\nprint(f\"\\n---> Imports done\")","metadata":{"execution":{"iopub.execute_input":"2025-05-30T15:48:46.672206Z","iopub.status.busy":"2025-05-30T15:48:46.671853Z","iopub.status.idle":"2025-05-30T15:48:46.678001Z","shell.execute_reply":"2025-05-30T15:48:46.677035Z"},"papermill":{"duration":0.085069,"end_time":"2025-05-30T15:48:46.679640","exception":false,"start_time":"2025-05-30T15:48:46.594571","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **TRAINING ELEMENTS**","metadata":{"papermill":{"duration":0.072298,"end_time":"2025-05-30T15:48:46.829153","exception":false,"start_time":"2025-05-30T15:48:46.756855","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%writefile -a myutils.py\n\nclass ParticipantVisibleError(Exception):\n    \"\"\"Errors raised here will be shown directly to the competitor.\"\"\"\n    pass\n\nclass CompetitionMetric:\n    \"\"\"Hierarchical macro F1 for the CMI 2025 challenge.\"\"\"\n    def __init__(self):\n        self.target_gestures = [\n            'Above ear - pull hair',\n            'Cheek - pinch skin',\n            'Eyebrow - pull hair',\n            'Eyelash - pull hair',\n            'Forehead - pull hairline',\n            'Forehead - scratch',\n            'Neck - pinch skin',\n            'Neck - scratch',\n        ]\n        self.non_target_gestures = [\n            'Write name on leg',\n            'Wave hello',\n            'Glasses on/off',\n            'Text on phone',\n            'Write name in air',\n            'Feel around in tray and pull out an object',\n            'Scratch knee/leg skin',\n            'Pull air toward your face',\n            'Drink from bottle/cup',\n            'Pinch knee/leg skin'\n        ]\n        self.all_classes = self.target_gestures + self.non_target_gestures\n\n    def calculate_hierarchical_f1(\n        self,\n        sol: pd.DataFrame,\n        sub: pd.DataFrame\n    ) -> float:\n\n        # Validate gestures\n        invalid_types = {i for i in sub['gesture'].unique() if i not in self.all_classes}\n        if invalid_types:\n            raise ParticipantVisibleError(\n                f\"Invalid gesture values in submission: {invalid_types}\"\n            )\n\n        # Compute binary F1 (Target vs Non-Target)\n        y_true_bin = sol['gesture'].isin(self.target_gestures).values\n        y_pred_bin = sub['gesture'].isin(self.target_gestures).values\n        f1_binary = f1_score(\n            y_true_bin,\n            y_pred_bin,\n            pos_label=True,\n            zero_division=0,\n            average='binary'\n        )\n\n        # Build multi-class labels for gestures\n        y_true_mc = sol['gesture'].apply(lambda x: x if x in self.target_gestures else 'non_target')\n        y_pred_mc = sub['gesture'].apply(lambda x: x if x in self.target_gestures else 'non_target')\n\n        # Compute macro F1 over all gesture classes\n        f1_macro = f1_score(\n            y_true_mc,\n            y_pred_mc,\n            average='macro',\n            zero_division=0\n        )\n\n        return 0.5 * f1_binary + 0.5 * f1_macro\n\nclass Utils:\n    \"\"\"\n    This class creates and uses several utility methods to be used across the code\n    \"\"\"\n\n    def __init__(self):\n        pass\n\n    def ScoreMetric(\n        self, \n        solution, \n        submission, \n        row_id_column_name = \"sequence_id\"\n    )-> float:\n        \"\"\"\n        This method calculates the metric for the competition\n        Inputs- solution, submission:- input truth and predictions\n                row_id_column_name :- id - index column name\n        Output- float:- competition metric\n        \"\"\";\n\n        for col in (row_id_column_name, 'gesture'):\n            if col not in solution.columns:\n                raise ParticipantVisibleError(f\"Solution file missing required column: '{col}'\")\n            if col not in submission.columns:\n                raise ParticipantVisibleError(f\"Submission file missing required column: '{col}'\")\n\n        metric = CompetitionMetric()\n        return metric.calculate_hierarchical_f1(solution, submission)\n\n    def pp_preds(self, ypreds : np.ndarray)-> np.ndarray :\n        \"Post-processes the predictions using min-max values from the training data\"\n        return np.clip( ypreds , a_min = 0, a_max = 1 )\n\n    def CleanMemory(self):\n        \"This method cleans the memory off unused objects and displays the cleaned state RAM usage\"\n\n        collect();\n        libc.malloc_trim(0)\n        pid        = getpid()\n        py         = Process(pid)\n        memory_use = py.memory_info()[0] / 2. ** 30\n        return f\"\\nRAM usage = {memory_use :.4} GB\"\n\ndef reduce_mem_usage(\n    dataframe, dataset: str\n):  \n    \"\"\"\n    Reducing memory for the dataset based on datatypes\n    Source - https://www.kaggle.com/competitions/drw-crypto-market-prediction/discussion/580485\n\n    Inputs - \n    dataframe - pd.DataFrame\n    dataset   - str : dataset label\n\n    Returns - \n    dataframe - reduced memory dataset\n    \"\"\"\n    \n    print(f'---> Reducing memory usage for: {dataset}')\n    initial_mem_usage = dataframe.memory_usage().sum() / 1024**2\n\n    for col in tqdm( dataframe.columns ):\n        col_type = dataframe[col].dtype\n\n        c_min = dataframe[col].min()\n        c_max = dataframe[col].max()\n        if str(col_type)[:3] == 'int':\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                dataframe[col] = dataframe[col].astype(np.int8)\n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                dataframe[col] = dataframe[col].astype(np.int16)\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                dataframe[col] = dataframe[col].astype(np.int32)\n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                dataframe[col] = dataframe[col].astype(np.int64)\n        else:\n            try:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    dataframe[col] = dataframe[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    dataframe[col] = dataframe[col].astype(np.float32)\n                else:\n                    dataframe[col] = dataframe[col].astype(np.float64)\n            except:\n                pass\n\n    final_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    print('---> Memory usage before: {:.2f} MB'.format(initial_mem_usage))\n    print('---> Memory usage after: {:.2f} MB'.format(final_mem_usage))\n\n    dec = float(100 * (initial_mem_usage - final_mem_usage) / initial_mem_usage)\n    print(f'---> Decreased memory usage by {round(dec, 4)} percent \\n')\n    return dataframe\n\nutils = Utils()\ncollect()\nprint()","metadata":{"execution":{"iopub.execute_input":"2025-05-30T15:48:46.979576Z","iopub.status.busy":"2025-05-30T15:48:46.978859Z","iopub.status.idle":"2025-05-30T15:48:46.986172Z","shell.execute_reply":"2025-05-30T15:48:46.985252Z"},"papermill":{"duration":0.086665,"end_time":"2025-05-30T15:48:46.987566","exception":false,"start_time":"2025-05-30T15:48:46.900901","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a myutils.py\n\nclass AdversarialCVMaker:\n    \"\"\"\n    This class assists in adversarial CV between the train and test data with the below steps-\n\n    1. Consider any classifier as a base model, I prefer any boosted tree model as I don't have to focus too much on preprocessing\n    2. Load the train and test set features\n    3. Make a new target column with 1 for test set occurrances and 0 for train-set\n    4. Classify to predict the test set instances with the features and new target from the step above\n    \n    If the AUC score hovers around 50% (random model), then we can be sure that the train and test set have similar distributions \n    Else, if our model is able to differentiate between the train and test data, then our model is unlikely to generalize as-is.\n    In this case, further adjustments may be necesary \n    \"\"\"\n\n    def __init__(self, n_splits: int = 5) :\n        self.model = \\\n        LGBMC(\n            n_estimators     = 200,\n            learning_rate    = 0.02,\n            max_depth        = 3, \n            colsample_bytree = 0.50,\n            objective        = \"binary\",\n            metric           = \"auc\",\n            random_state     = 42,\n            device           = \"gpu\" if torch.cuda.is_available() else \"cpu\",\n        )\n\n        self.n_splits = n_splits\n\n    @staticmethod\n    def scorer(ytrue, ypreds):\n        return roc_auc_score( ytrue, ypreds )\n\n    def make_cv(\n        self, Xtrain, Xtest, **fit_params,\n        ):\n        \"Fits the model with the auxilary target and calculates the AUC score for the CV\"\n\n        df = \\\n        pd.concat(\n            [Xtrain.assign(**{\"target\" : 0}), \n             Xtest.assign(**{\"target\" : 1}),\n            ], \n            axis=0, ignore_index = True\n        )\n\n        cv     = StratifiedKFold(n_splits = self.n_splits, random_state = 42, shuffle = True)\n        scores = 0\n        \n        for train_idx, dev_idx in cv.split(df, df[\"target\"]) :\n            Xtr  = df.loc[train_idx].drop(\"target\", axis=1)\n            Xdev = df.loc[dev_idx].drop(\"target\", axis=1)\n            ytr  = df.loc[train_idx, \"target\"]\n            ydev = df.loc[dev_idx, \"target\"]\n\n            cat_cols = list( Xdev.select_dtypes(include = [\"string\", \"category\", \"object\"]).columns )\n\n            if len(cat_cols) > 0 :\n                Xtr[cat_cols]  = Xtr[cat_cols].astype(\"category\")\n                Xdev[cat_cols] = Xdev[cat_cols].astype(\"category\")\n            else:\n                pass\n                \n            model = clone(self.model)\n            model.fit(Xtr, ytr)\n            dev_preds = model.predict_proba(Xdev)[:,1]\n            score = self.scorer(ydev, dev_preds)\n            scores += score\n\n        score = scores / self.n_splits\n\n        PrintColor(\n            f\"\\n---> Overall adversarial CV score = {score :,.4f}\"\n        )\n\n        if score > 0.60 :\n            PrintColor(\n                f\"---> Check for test-train distribution shift\\n\", color = Fore.RED\n            )\n        else:\n            PrintColor(\n                f\"---> Train-test distributions are similar\\n\", color = Fore.GREEN\n            )\n","metadata":{"execution":{"iopub.execute_input":"2025-05-30T15:48:47.133049Z","iopub.status.busy":"2025-05-30T15:48:47.132738Z","iopub.status.idle":"2025-05-30T15:48:47.139237Z","shell.execute_reply":"2025-05-30T15:48:47.138307Z"},"papermill":{"duration":0.081315,"end_time":"2025-05-30T15:48:47.141530","exception":false,"start_time":"2025-05-30T15:48:47.060215","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a training.py\n\ndef MakePermImp(\n        method, mdl, X, y, ygrp,\n        myscorer, \n        n_repeats = 2,\n        state = 42,\n        ntop: int = 15,\n        **params,\n):\n    \"\"\"\n    This function makes the permutation importance for the provided model and returns the importance scores for all features\n    \n    Note-\n    myscorer - scikit-learn -> metrics -> make_scorer object with the corresponding eval metric and relevant details\n    \"\"\"\n\n    cv        = PredefinedSplit(ygrp)\n    n_splits  = ygrp.nunique()\n    drop_cols = [\"Source\", \"id\", \"Id\", \"Label\", \"fold_nb\"]\n\n    for fold_nb, (train_idx, dev_idx) in tqdm(enumerate(cv.split(X, y))):\n        Xtr  = X.iloc[train_idx].drop(drop_cols, axis=1, errors = \"ignore\")\n        Xdev = X.iloc[dev_idx].drop(drop_cols, axis=1, errors = \"ignore\")\n        ytr  = y.loc[Xtr.index]\n        ydev = y.loc[Xdev.index]\n\n        model = clone(mdl)\n        sel_cols = list(Xdev.columns)\n        model.fit(Xtr, ytr)\n\n        imp_ = permutation_importance(model,\n                                      Xdev, ydev,\n                                      scoring = myscorer,\n                                      n_repeats = n_repeats,\n                                      random_state = state,\n                                      )[\"importances_mean\"]\n        imp_ = pd.Series(index = sel_cols, data = imp_)\n\n        display(\n            imp_.\\\n            sort_values(ascending = False).\\\n            head(ntop).\\\n            to_frame().\\\n            transpose().\\\n            style.\\\n            format(formatter = '{:,.3f}').\\\n            background_gradient(\"icefire\", axis=1).\\\n            set_caption(f\"Top {ntop} features\")\n            )\n\n        return imp_","metadata":{"execution":{"iopub.execute_input":"2025-05-30T15:48:47.282717Z","iopub.status.busy":"2025-05-30T15:48:47.282411Z","iopub.status.idle":"2025-05-30T15:48:47.288830Z","shell.execute_reply":"2025-05-30T15:48:47.287754Z"},"papermill":{"duration":0.079155,"end_time":"2025-05-30T15:48:47.290360","exception":false,"start_time":"2025-05-30T15:48:47.211205","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a training.py\n\nclass ModelTrainer:\n    \"This class trains the provided model on the train-test data and returns the predictions and fitted models\"\n\n    def __init__(\n        self,\n        problem_type   : str   = \"multiclass\", \n        es             : int   = 100,\n        target         : str   = \"\",\n        metric_lbl     : str   = \"auc\",\n        orig_req       : bool  = False,\n        orig_all_folds : bool  = False,\n        drop_cols      : list  = [\"Source\", \"id\", \"Id\", \"Label\", \"fold_nb\"],\n        pp_preds       : bool  = False,\n    ):\n        \"\"\"\n        Key parameters-\n        es_iter  - early stopping rounds for boosted trees\n        pp_preds - do you want to post-process predictions (true/ false boolean)\n        \"\"\"\n\n        self.problem_type   = problem_type\n        self.es_iter        = es\n        self.target         = target\n        self.drop_cols      = drop_cols + [self.target]\n        self.metric_lbl     = metric_lbl\n        self.orig_req       = orig_req\n        self.orig_all_folds = orig_all_folds\n        self.pp_preds       = pp_preds\n\n        if self.metric_lbl == \"rmse\" :\n            self.ScoreMetric = lambda x,y : root_mean_squared_error( x, self.PostProcessPreds( y))\n        elif self.metric_lbl == \"mae\" :\n            self.ScoreMetric = lambda x,y : mean_absolute_error( x, self.PostProcessPreds( y))\n        elif self.metric_lbl == \"accuracy\" :\n            self.ScoreMetric = lambda x,y : accuracy_score( np.uint8( x ), np.uint8(self.PostProcessPreds( y ) ) )\n        elif self.metric_lbl == \"auc\" :\n            self.ScoreMetric = lambda x,y : roc_auc_score(  x , self.PostProcessPreds( y ))\n        elif self.metric_lbl == \"auc_multi\" :\n            self.ScoreMetric = lambda x,y : roc_auc_score( \n                x , self.PostProcessPreds( y ), average = \"macro\", multi_class = \"ovo\"\n            )\n        elif self.metric_lbl == \"kappa\" :\n            self.ScoreMetric = lambda x,y : cohen_kappa_score( np.uint8( x ), np.uint8(self.PostProcessPreds( y ) ), \n                                                              weights = \"quadratic\" \n                                                             )\n        elif self.metric_lbl == \"logloss\" :\n            self.ScoreMetric = lambda x,y : log_loss(  x , self.PostProcessPreds( y ) )\n        elif self.metric_lbl == \"rmsle\" :\n            self.ScoreMetric = lambda x,y : root_mean_squared_log_error( x, self.PostProcessPreds( y))\n        else:\n            self.ScoreMetric = utils.ScoreMetric\n\n    def PlotFtreImp(\n        self, \n        ftreimp: pd.Series, \n        method: str,\n        ntop: int = 50,\n        title_specs: dict = {'fontsize': 12,'fontweight' : 'bold','color': '#992600'},\n        **params,\n    ):\n        \"This function plots the feature importances for the model provided\"\n\n        print()\n        \n        with sns.axes_style(\"white\"):\n            fig, ax = plt.subplots(1, 1, figsize = (25, 7.5))\n    \n            ftreimp.sort_values(ascending = False).\\\n            head(ntop).\\\n            plot.bar(ax = ax, color = \"#1285c7\")\n            ax.set_title(\n                f\"Feature Importances - {method}\", \n                **title_specs\n            )\n    \n            plt.tight_layout()\n            plt.show()\n        print()\n\n    def PostProcessPreds(self, ypred):\n        \"This method post-processes predictions optionally\"\n        if self.pp_preds :\n            return np.clip(ypred, a_min = 0, a_max = 1)\n        else:\n            return ypred\n            \n    def LoadData(\n            self, X, y, Xtest,\n            train_idx : list = [],\n            dev_idx   : list = [],\n            ):\n        \"This method loads the train and test data for the model fold using/ not using the original data\"\n\n        try:\n            mysrc = X[\"Source\"]\n        except:\n            X[\"Source\"] , Xtest[\"Source\"] = (\"Competition\", \"Competition\")\n\n        if self.orig_req == False:\n            Xtr  = X.iloc[train_idx].query(\"Source == 'Competition'\").drop(self.drop_cols, axis=1, errors = \"ignore\")\n            ytr  = y.iloc[Xtr.index]\n            Xdev = X.iloc[dev_idx].query(\"Source == 'Competition'\").drop(self.drop_cols, axis=1, errors = \"ignore\")\n            ydev = y.iloc[Xdev.index]\n\n        elif self.orig_req == True and self.orig_all_folds == True:\n            Xtr  = X.iloc[train_idx].query(\"Source == 'Competition'\").drop(self.drop_cols, axis=1, errors = \"ignore\")\n            ytr  = y.iloc[Xtr.index]\n            Xdev = X.iloc[dev_idx].query(\"Source == 'Competition'\").drop(self.drop_cols, axis=1, errors = \"ignore\")\n            ydev = y.iloc[Xdev.index]\n\n            orig_x = X.query(\"Source == 'Original'\")[Xtr.columns]\n            orig_y = y.iloc[orig_x.index]\n\n            Xtr = pd.concat([Xtr, orig_x], axis = 0, ignore_index = True)\n            ytr = pd.concat([ytr, orig_y], axis = 0, ignore_index = True)\n\n        elif self.orig_req == True and self.orig_all_folds == False:\n            Xtr  = X.iloc[train_idx].drop(self.drop_cols, axis=1, errors = \"ignore\")\n            ytr  = y.iloc[Xtr.index]\n            Xdev = X.iloc[dev_idx].query(\"Source == 'Competition'\").drop(self.drop_cols, axis=1, errors = \"ignore\")\n            ydev = y.iloc[Xdev.index]\n\n        Xt = Xtest[Xdev.columns]\n\n        print(f\"\\n---> Shapes = {Xtr.shape} {ytr.shape} -- {Xdev.shape} {ydev.shape} -- {Xt.shape}\")\n        return (Xtr, ytr, Xdev, ydev, Xt)\n    \n    def MakePreds(self, X, fitted_model):\n        \"This method creates the model predictions based on the model provided, with optional post-processing\"\n\n        if self.problem_type == \"regression\":\n            if isinstance(fitted_model, (TNC, TNR)) == True:\n                return self.PostProcessPreds(fitted_model.predict(X.to_numpy()).flatten())\n            else:\n                return self.PostProcessPreds(fitted_model.predict(X))\n                \n        elif self.problem_type == \"binary\":\n            if isinstance(fitted_model, (TNC, TNR)) == True:\n                return self.PostProcessPreds(fitted_model.predict_proba(X.to_numpy()[:,1]).flatten())\n            else:\n                return self.PostProcessPreds(fitted_model.predict_proba(X)[:, 1])\n                \n        elif self.problem_type == \"multiclass\":\n            if isinstance(fitted_model, (TNC, TNR)) == True:\n                return self.PostProcessPreds(fitted_model.predict_proba(X.to_numpy()))\n            else:\n                return self.PostProcessPreds(fitted_model.predict_proba(X))\n\n    def MakeOfflineModel(\n        self, X, y, ygrp, Xtest, mdl, method,\n        test_preds_req   : bool = True,\n        ftreimp_plot_req : bool = True,\n        ntop             : int  = 50,\n        **params,\n    ):\n        \"\"\"\n        This method trains the provided model on the dataset and cross-validates appropriately\n\n        Inputs-\n        X, y, ygrp       - training data components (Xtrain, ytrain, fold_nb)\n        Xtest            - test data (optional)\n        model            - model object for training\n        method           - model method label\n        test_preds_req   - boolean flag to extract test set predictions\n        ftreimp_plot_req - boolean flag to plot tree feature importances\n        ntop             - top n features for feature importances plot\n\n        Returns-\n        oof_preds, test_preds - prediction arrays\n        fitted_models         - fitted model list for test set\n        ftreimp               - feature importances across selected features\n        mdl_best_iter         - model average best iteration across folds\n        \"\"\"\n\n        if self.problem_type != \"multiclass\" :\n            oof_preds     = np.zeros(len(X.loc[X.Source == \"Competition\"]))\n            orig_preds    = np.zeros(len(X.loc[X.Source == \"Original\"]))\n        else:\n            num_tgt_cls   = len( np.unique(y) )\n            oof_preds     = np.zeros((len(X.loc[X.Source == \"Competition\"]), num_tgt_cls) )\n            orig_preds    = np.zeros((len(X.loc[X.Source == \"Original\"]),  num_tgt_cls) )\n            \n        test_preds    = []\n        mdl_best_iter = []\n        ftreimp       = 0\n\n        scores, tr_scores, fitted_models = [], [], []\n\n        if self.orig_req == True:\n            cv = PredefinedSplit(ygrp)\n        elif self.orig_req == False:\n            X  = X.loc[X.Source == \"Competition\"]\n            y  = y.iloc[X.index]\n            cv = PredefinedSplit(ygrp.iloc[0 : len(X)])\n\n        n_splits = ygrp.nunique()\n\n        for fold_nb, (train_idx, dev_idx) in tqdm(enumerate(cv.split(X, y))):\n            Xtr, ytr, Xdev, ydev, Xt = \\\n            self.LoadData(X, y, Xtest, train_idx, dev_idx)\n\n            model = clone(mdl)\n\n            if \"CB\" in method and self.es_iter > 0:\n                model.fit(Xtr, ytr,\n                          eval_set = [(Xdev, ydev)],\n                          verbose = 0,\n                          early_stopping_rounds = self.es_iter,\n                          )\n                best_iter = model.get_best_iteration()\n\n            elif \"LGB\" in method and self.es_iter > 0:\n                model.fit(Xtr, ytr,\n                          eval_set = [(Xdev, ydev)],\n                          callbacks = [log_evaluation(0),\n                                       early_stopping(stopping_rounds = self.es_iter, verbose = False,),\n                                       ],\n                          )\n                best_iter = model.best_iteration_\n\n            elif \"XGB\" in method and self.es_iter > 0:\n                model.fit(Xtr, ytr,\n                          eval_set = [(Xdev, ydev)],\n                          verbose  = 0,\n                          )\n                best_iter = model.best_iteration\n\n            else:\n                model.fit(Xtr, ytr)\n                best_iter = -1\n\n            fitted_models.append(model)\n\n            try:\n                ftreimp += model.feature_importances_\n            except:\n                try:\n                    ftreimp += model.coef_.flatten()\n                except:\n                    pass\n\n            try:\n                ftreimp += model[\"M\"].feature_importances_\n            except:\n                try:\n                    ftreimp += model[\"M\"].coef_.flatten()\n                except:\n                    pass               \n            \n            dev_preds = self.MakePreds(Xdev, model)\n            oof_preds[Xdev.index] = dev_preds\n\n            train_preds  = self.MakePreds(Xtr, model)\n            tr_score     = self.ScoreMetric(ytr.values.flatten(), train_preds)\n            score        = self.ScoreMetric(ydev.values.flatten(), dev_preds)\n\n            scores.append(score)\n            tr_scores.append(tr_score)\n\n            nspace = 15 - len(method) - 2 if fold_nb <= 9 else 15 - len(method) - 1\n\n            if self.es_iter > 0 :\n                PrintColor(f\"{method} Fold{fold_nb} {' ' * nspace} OOF = {score:.6f} | Train = {tr_score:.6f} | Iter = {best_iter:,.0f} \")\n            else:\n                PrintColor(f\"{method} Fold{fold_nb} {' ' * nspace} OOF = {score:.6f} | Train = {tr_score:.6f} \")\n                \n            mdl_best_iter.append(best_iter)\n\n            if test_preds_req:\n                test_preds.append(self.MakePreds(Xt, model))\n            else:\n                pass\n\n        if self.problem_type != \"multiclass\" :\n            test_preds = np.mean(np.stack(test_preds, axis = 1), axis=1)\n        else:\n            df_list = []\n            for cnt in range(n_splits) :\n                df = pd.DataFrame(test_preds[cnt])\n                df_list.append(df)\n                del df\n            test_preds = pd.concat(df_list).groupby(level = 0).mean().to_numpy()\n            del df_list\n\n        try:\n            ftreimp = pd.Series(ftreimp, index = Xdev.columns)\n        except:\n            pass\n        mdl_best_iter = np.uint16(np.amax(mdl_best_iter))\n\n        if ftreimp_plot_req :\n            print()\n            try:\n                self.PlotFtreImp(ftreimp, method = method, ntop = ntop,)\n            except:\n                pass\n        else:\n            pass\n\n        PrintColor(f\"\\n---> {np.mean(scores):.6f} +- {np.std(scores):.6f} | OOF\", color = Fore.RED)\n        PrintColor(f\"---> {np.mean(tr_scores):.6f} +- {np.std(tr_scores):.6f} | Train\", color = Fore.RED)\n\n        if self.es_iter <= 0 :\n            pass\n        else:\n            PrintColor(\n                f\"---> Max best iteration = {mdl_best_iter :,.0f}\",\n                color = Fore.RED\n            )\n\n        return (fitted_models, oof_preds, test_preds, ftreimp, mdl_best_iter)\n\n    def MakeOnlineModel(\n        self, X, y, Xtest, model, method,\n        test_preds_req : bool = False,\n    ):\n        \"This method refits the model on the complete train data and returns the model fitted object and predictions\"\n\n        try:\n            model.early_stopping_rounds = None\n        except:\n            pass\n\n        if \"TN\" in method:\n            model.fit(\n                X.to_numpy(), y.to_numpy().reshape(-1,1),\n                max_epochs  = 100,\n                batch_size  = 128,\n                virtual_batch_size = 64,\n                )\n        else:\n            try:\n                model.fit(X, y, verbose = 0)\n            except:\n                model.fit(X, y,)\n\n        oof_preds  = self.MakePreds(X, model)\n        if test_preds_req:\n            test_preds = self.MakePreds(Xtest[X.columns], model)\n        else:\n            test_preds = 0\n            \n        return (model, oof_preds, test_preds)","metadata":{"execution":{"iopub.execute_input":"2025-05-30T15:48:47.432774Z","iopub.status.busy":"2025-05-30T15:48:47.432479Z","iopub.status.idle":"2025-05-30T15:48:47.444972Z","shell.execute_reply":"2025-05-30T15:48:47.444023Z"},"papermill":{"duration":0.085662,"end_time":"2025-05-30T15:48:47.446333","exception":false,"start_time":"2025-05-30T15:48:47.360671","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **LAMA IMPORTS**","metadata":{"papermill":{"duration":0.078271,"end_time":"2025-05-30T15:48:47.598061","exception":false,"start_time":"2025-05-30T15:48:47.519790","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%writefile -a myimports_lama.py\n\nfrom warnings import filterwarnings\nfilterwarnings('ignore')\n\nimport os, re, joblib, tempfile, ctypes\nfrom os import path, walk, getpid\nfrom psutil import Process\nfrom collections import Counter\nfrom itertools import product\nfrom gc import collect\n\nlibc = ctypes.CDLL(\"libc.so.6\")\n\nfrom IPython.display import display_html, clear_output\nfrom pprint import pprint\nfrom functools import partial\nfrom copy import deepcopy\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom colorama import Fore, Style, init\nfrom tqdm.notebook import tqdm\n\n# Essential DS libraries\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport polars.selectors as cs\nfrom sklearn.metrics import *\n\nfrom sklearn.model_selection import *\nfrom sklearn.preprocessing import *\nimport torch\nimport torch.nn as nn\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\n\n# LightAutoML presets, task and report generation\nfrom lightautoml.automl.presets.tabular_presets import TabularAutoML\nfrom lightautoml.tasks import Task\n\n# Color printing\ndef PrintColor(text: str, color = Fore.BLUE, style = Style.BRIGHT):\n    \"Prints color outputs using colorama using a text F-string\"\n    print(style + color + text + Style.RESET_ALL)\n\nprint(f\"---> CUDA available = {torch.cuda.is_available()}\\n\\n\")","metadata":{"execution":{"iopub.execute_input":"2025-05-30T15:48:47.746497Z","iopub.status.busy":"2025-05-30T15:48:47.746191Z","iopub.status.idle":"2025-05-30T15:48:47.752319Z","shell.execute_reply":"2025-05-30T15:48:47.751449Z"},"papermill":{"duration":0.081996,"end_time":"2025-05-30T15:48:47.753836","exception":false,"start_time":"2025-05-30T15:48:47.671840","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **AUTOGLUON IMPORTS**","metadata":{"papermill":{"duration":0.07174,"end_time":"2025-05-30T15:48:47.899006","exception":false,"start_time":"2025-05-30T15:48:47.827266","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%writefile -a myimports_ag.py\n\nimport numpy as np, pandas as pd\nimport polars as pl\nimport polars.selectors as cs\nimport re, os, joblib, logging\nfrom gc import collect\n\nfrom IPython.display import display_html, clear_output\nfrom pprint import pprint\nfrom tqdm.notebook import tqdm\nfrom colorama import Fore, Back, Style\nfrom os import path, walk, getpid\nfrom psutil import Process\nimport ctypes\nlibc = ctypes.CDLL(\"libc.so.6\")\n\nfrom warnings import filterwarnings\nfilterwarnings(\"ignore\")\n\nfrom sklearn.model_selection import *\nfrom sklearn.metrics import *\nfrom autogluon.tabular import TabularPredictor, TabularDataset\nfrom autogluon.core.metrics import make_scorer as ag_make_scorer\n\n# Color printing\ndef PrintColor(text: str, color = Fore.BLUE, style = Style.BRIGHT):\n    \"Prints color outputs using colorama using a text F-string\"\n    print(style + color + text + Style.RESET_ALL)","metadata":{"execution":{"iopub.execute_input":"2025-05-30T15:48:48.043550Z","iopub.status.busy":"2025-05-30T15:48:48.043266Z","iopub.status.idle":"2025-05-30T15:48:48.049152Z","shell.execute_reply":"2025-05-30T15:48:48.048358Z"},"papermill":{"duration":0.079856,"end_time":"2025-05-30T15:48:48.050403","exception":false,"start_time":"2025-05-30T15:48:47.970547","status":"completed"},"tags":[]},"outputs":[],"execution_count":null}]}