{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.14"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":9997028,"sourceType":"datasetVersion","datasetId":6106113}],"dockerImageVersionId":30761,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":3218.371246,"end_time":"2024-12-04T21:28:02.846701","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-12-04T20:34:24.475455","version":"2.6.0"},"colab":{"provenance":[{"file_id":"1dR7OQn9lN5beldkirnwLdkaYoG4LUNlx","timestamp":1732282140660}]}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"","metadata":{"papermill":{"duration":0.007392,"end_time":"2024-12-04T20:34:27.496132","exception":false,"start_time":"2024-12-04T20:34:27.488740","status":"completed"},"tags":[]},"attachments":{}},{"cell_type":"markdown","source":"# **FOREWORD**\n\nThis kernel blends 2-more component models from my private models in version 7. <br>\nI use the artefacts from the component models and fuse them here <br>\n\nI record these experiments in the **Ensemble** tab\n\n","metadata":{"papermill":{"duration":0.006002,"end_time":"2024-12-04T20:34:27.508710","exception":false,"start_time":"2024-12-04T20:34:27.502708","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# **COMMON SCRIPTS**","metadata":{"papermill":{"duration":0.006276,"end_time":"2024-12-04T20:34:27.521904","exception":false,"start_time":"2024-12-04T20:34:27.515628","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%time \n\nimport os, joblib, random, torch, warnings, optuna\nfrom tqdm import tqdm\nfrom pathlib import Path\nfrom concurrent.futures import ThreadPoolExecutor\nfrom typing import Dict\nfrom colorama import Fore, Style, Back\nfrom pprint import pprint\nfrom collections.abc import Callable\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nwarnings.filterwarnings('ignore')\n\nimport numpy as np\nimport pandas as pd\n\nfrom sklearn.metrics import (make_scorer, \ncohen_kappa_score, mean_squared_error, confusion_matrix, ConfusionMatrixDisplay)\n\nfrom sklearn.compose import ColumnTransformer, make_column_selector\nfrom sklearn.model_selection import StratifiedKFold, cross_val_score, cross_val_predict, PredefinedSplit as PDS\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.pipeline import Pipeline\nfrom scipy.optimize import minimize\nfrom sklearn.inspection import permutation_importance\nfrom sklearn.base import clone\n\nimport lightgbm as lgb\nfrom lightgbm import LGBMRegressor as LGBMR\nfrom catboost import CatBoostRegressor as CBR\nfrom xgboost import XGBRegressor as XGBR\n\n\nSEED  = 42\n\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\n\nseed_everything(42)\n\n# Color printing\ndef PrintColor(text: str, color = Fore.BLUE, style = Style.BRIGHT):\n    \"Prints color outputs using colorama using a text F-string\"\n    print(style + color + text + Style.RESET_ALL)\n    \nprint(f\"\\n---> Imports done\")\n\nKAPPA_SCORER = \\\nmake_scorer(\n    cohen_kappa_score,\n    greater_is_better=True,\n    weights ='quadratic',\n)\n\nmyscorer = \\\nmake_scorer(\n    mean_squared_error,\n    greater_is_better= False,\n    squared =False,\n)\n\ndef ScoreMetric(ytrue, ypreds):\n    \"Scoring metric for the competition\"\n    \n    score = \\\n    cohen_kappa_score(\n        np.uint8(np.round(ytrue)),\n        np.uint8(np.round(ypreds)),\n        weights = \"quadratic\"\n    )\n    return score\n\n\nclass MyLGBMR(lgb.LGBMRegressor):\n    '''\n    Custom LightGBM Regressor\n\n    It optimizes threshold values during fitting.\n    Main goal is preventing overfit on validation data.\n    '''\n\n    @staticmethod\n    def ScoreMetric(ytrue, ypreds)-> float:\n        \"Scoring metric for the competition\"\n\n        return cohen_kappa_score(\n            np.uint8(np.round(ytrue,0)),\n            np.uint8(np.round(ypreds,0)),\n            weights = \"quadratic\"\n        )\n\n    def _threshold_rounder(self, y_pred, thresholds):\n        output = np.zeros_like(y_pred, dtype=int)\n\n        for i in range(len(thresholds)):\n            output += (y_pred >= thresholds[i]).astype(int)\n        return output\n\n    def _eval_preds(self, thresholds, y_true, y_pred):\n        y_pred = self._threshold_rounder(y_pred, thresholds)\n        score  = self.ScoreMetric(y_true, y_pred)\n        return -score\n\n    def fit(\n        self, X, y, verbose = False, **kwargs\n    ):\n        super().fit(X, y, **kwargs)\n        y_pred           = super().predict(X, **kwargs)\n        self.base_score_ = self.ScoreMetric(y, y_pred)\n\n        self.n_classes_ = len(np.unique(y))\n        init_th = np.linspace(0, self.n_classes_ - 1, self.n_classes_) + 0.5\n        init_th = init_th[0: -1]\n\n        if verbose :\n            pprint(init_th)\n\n        self.optimizer = \\\n        minimize(\n            self._eval_preds,\n            x0     = init_th,\n            args   = (y, y_pred),\n            method = 'Nelder-Mead',\n        )\n\n        self.optimized_score_ = -self._eval_preds(self.optimizer.x, y, y_pred)\n        return self\n\n    def predict(\n        self,\n        X : np.ndarray | pd.DataFrame,\n        base_preds: bool = False,\n        **kwargs\n    ):\n        y_pred = super().predict(X, **kwargs)\n\n        if base_preds:\n            return y_pred\n        else:\n            return self._threshold_rounder(y_pred, self.optimizer.x)\n\n    def get_optimization_info(self)-> Dict:\n        \"Provides the scores before and after the optimization and thresholds\"\n\n        return {\n            'base_score'     : self.base_score_,\n            'optimized_score': self.optimized_score_,\n            'thresholds'     : self.optimizer.x\n        }\n\nclass MyCBR(CBR):\n    '''\n    Custom Catboost Regressor\n\n    It optimizes threshold values during fitting.\n    Main goal is preventing overfit on validation data.\n    '''\n\n    @staticmethod\n    def ScoreMetric(ytrue, ypreds)-> float:\n        \"Scoring metric for the competition\"\n\n        return cohen_kappa_score(\n            np.uint8(np.round(ytrue,0)),\n            np.uint8(np.round(ypreds,0)),\n            weights = \"quadratic\"\n        )\n\n    def _threshold_rounder(self, y_pred, thresholds):\n        output = np.zeros_like(y_pred, dtype=int)\n\n        for i in range(len(thresholds)):\n            output += (y_pred >= thresholds[i]).astype(int)\n        return output\n\n    def _eval_preds(self, thresholds, y_true, y_pred):\n        y_pred = self._threshold_rounder(y_pred, thresholds)\n        score  = self.ScoreMetric(y_true, y_pred)\n        return -score\n\n    def fit(\n        self, X, y, verbose = False, **kwargs\n    ):\n        super().fit(X, y, **kwargs)\n        y_pred           = super().predict(X, **kwargs)\n        self.base_score_ = self.ScoreMetric(y, y_pred)\n\n        self.n_classes_ = len(np.unique(y))\n        init_th = np.linspace(0, self.n_classes_ - 1, self.n_classes_) + 0.5\n        init_th = init_th[0: -1]\n\n        if verbose :\n            pprint(init_th)\n\n        self.optimizer = \\\n        minimize(\n            self._eval_preds,\n            x0     = init_th,\n            args   = (y, y_pred),\n            method = 'Nelder-Mead',\n        )\n\n        self.optimized_score_ = -self._eval_preds(self.optimizer.x, y, y_pred)\n        return self\n\n    def predict(\n        self,\n        X : np.ndarray | pd.DataFrame,\n        base_preds: bool = False,\n        **kwargs\n    ):\n        y_pred = super().predict(X, **kwargs)\n\n        if base_preds:\n            return y_pred\n        else:\n            return self._threshold_rounder(y_pred, self.optimizer.x)\n\n    def get_optimization_info(self)-> Dict:\n        \"Provides the scores before and after the optimization and thresholds\"\n\n        return {\n            'base_score'     : self.base_score_,\n            'optimized_score': self.optimized_score_,\n            'thresholds'     : self.optimizer.x\n        }\n\nclass MyXGBR(XGBR):\n    '''\n    Custom XGBRegressor\n\n    It optimizes threshold values during fitting.\n    Main goal is preventing overfit on validation data.\n    '''\n\n    @staticmethod\n    def ScoreMetric(ytrue, ypreds)-> float:\n        \"Scoring metric for the competition\"\n\n        return cohen_kappa_score(\n            np.uint8(np.round(ytrue,0)),\n            np.uint8(np.round(ypreds,0)),\n            weights = \"quadratic\"\n        )\n\n    def _threshold_rounder(self, y_pred, thresholds):\n        output = np.zeros_like(y_pred, dtype=int)\n\n        for i in range(len(thresholds)):\n            output += (y_pred >= thresholds[i]).astype(int)\n        return output\n\n    def _eval_preds(self, thresholds, y_true, y_pred):\n        y_pred = self._threshold_rounder(y_pred, thresholds)\n        score  = self.ScoreMetric(y_true, y_pred)\n        return -score\n\n    def fit(\n        self, X, y, verbose = False, **kwargs\n    ):\n        super().fit(X, y, **kwargs)\n        y_pred           = super().predict(X, **kwargs)\n        self.base_score_ = self.ScoreMetric(y, y_pred)\n\n        self.n_classes_ = len(np.unique(y))\n        init_th = np.linspace(0, self.n_classes_ - 1, self.n_classes_) + 0.5\n        init_th = init_th[0: -1]\n\n        if verbose :\n            pprint(init_th)\n\n        self.optimizer = \\\n        minimize(\n            self._eval_preds,\n            x0     = init_th,\n            args   = (y, y_pred),\n            method = 'Nelder-Mead',\n        )\n\n        self.optimized_score_ = -self._eval_preds(self.optimizer.x, y, y_pred)\n        return self\n\n    def predict(\n        self,\n        X : np.ndarray | pd.DataFrame,\n        base_preds: bool = False,\n        **kwargs\n    ):\n        y_pred = super().predict(X, **kwargs)\n\n        if base_preds:\n            return y_pred\n        else:\n            return self._threshold_rounder(y_pred, self.optimizer.x)\n\n    def get_optimization_info(self)-> Dict:\n        \"Provides the scores before and after the optimization and thresholds\"\n\n        return {\n            'base_score'     : self.base_score_,\n            'optimized_score': self.optimized_score_,\n            'thresholds'     : self.optimizer.x\n        }\n\nclass ModelTrainer:\n    \"This class trains the provided model on the train-test data and returns the predictions and fitted models\"\n\n    def __init__(\n        self,\n        drop_cols : list,\n        ntop      : int = 50,\n        verbose   : bool = True,\n        test_preds_req : bool = True,\n    ):\n        self.drop_cols = drop_cols\n        self.ntop      = ntop\n        self.verbose   = verbose\n        self.test_preds_req = test_preds_req\n\n    def MakeOfflineModel(\n        self,\n        X, y, ygrp, Xtest,\n        model            : Callable,\n        method           : str,\n        ftreimp_plot_req : bool = True,\n        **kwargs,\n    ):\n        \"\"\"\n        This function trains the provided model on the dataset and cross-validates appropriately\n\n        Inputs-\n        X, y, ygrp       - training data components (Xtrain, ytrain, fold_nb)\n        Xtest            - test data (optional)\n        model            - model object for training\n        method           - model method label\n        ftreimp_plot_req - boolean flag to plot tree feature importances\n\n        Returns-\n        base_oof, adj_oof, mdl_preds - prediction arrays\n        fitted_models                - fitted model list for test set\n        ftreimp                      - feature importances across selected features\n        thresholds                   - all fold level threshold arrays\n        \"\"\"\n\n        PrintColor(f\"\\n ===== {method.upper()} =====  \\n\")\n\n        mdl_preds     = np.zeros(len(Xtest))\n        base_oof      = np.zeros(len(X))\n        adj_oof       = np.zeros(len(X))\n        thresholds    = {}\n        fitted_models = []\n\n        scores, bscores, ftreimp = 0,0,0\n\n        cv = PDS(ygrp[0 : len(X)])\n        n_splits = ygrp.nunique()\n\n        for fold_nb, (train_idx, dev_idx) in enumerate(cv.split(X, y)):\n\n            PrintColor(f\"------- FOLD {fold_nb} ------- \", color = Fore.BLACK)\n            \n            Xtr  = X.iloc[train_idx]\n            ytr  = y.iloc[train_idx]\n            Xdev = X.iloc[dev_idx]\n            ydev = y.iloc[dev_idx]\n\n            model.fit(Xtr, ytr)\n            fitted_models.append(model)\n\n            thresholds[f\"Fold{fold_nb}\"] = model[\"M\"].optimizer.x\n\n            try:\n                ftreimp += model[\"M\"].feature_importances_\n            except:\n                pass\n\n            base_preds = model.predict(Xdev, **{\"base_preds\" : True})\n            opt_preds  = model.predict(Xdev, **{\"base_preds\" : False})\n            base_score = model[\"M\"].ScoreMetric(ydev, base_preds)\n            opt_score  = model[\"M\"].ScoreMetric(ydev, opt_preds)\n            scores    += opt_score / n_splits\n            bscores   += base_score/ n_splits\n\n            base_oof[dev_idx] = base_preds\n            adj_oof[dev_idx]  = opt_preds\n\n            if self.verbose:\n                PrintColor(\n                    f\"---> OOF | Base = {base_score :.6f} Optimized = {opt_score :.6f}\",\n                    color = Fore.RED\n                )\n                \n                PrintColor(\n                    f\"\\n---> Optimization information\", \n                    color = Fore.MAGENTA,\n                )\n                pprint(model[\"M\"].get_optimization_info())\n                print()\n\n            if self.test_preds_req :\n                try:\n                    mdl_preds = \\\n                    mdl_preds + \\\n                    model.predict(\n                        Xtest.drop(self.drop_cols, axis=1, errors = \"ignore\"),\n                        base_preds = False\n                    )\n\n                except:\n                    mdl_preds = mdl_preds + model.predict(Xtest, base_preds = False)\n            print()\n\n        PrintColor(f\"---> Overall scores - \", color = Fore.RED)\n        PrintColor(f\"---> Optimized OOF = {scores:.6f} | Base OOF = {bscores  :.6f}\")\n\n        mdl_preds = mdl_preds / n_splits\n        ftreimp   = pd.Series(ftreimp, index = Xdev.columns)\n\n        if ftreimp_plot_req :\n            self.PlotFtreImp(ftreimp, method,)\n        else:\n            pass\n\n        return (base_oof, adj_oof, mdl_preds, fitted_models, ftreimp, thresholds,)\n\n    def PlotFtreImp(\n            self,\n            ftreimp     : pd.Series,\n            method      : str,\n            title_specs : dict = {'fontsize': 12, 'fontweight' : 'bold','color': '#992600'},\n            **params,\n        ):\n            \"This method plots the feature importances for the model provided\"\n\n            print()\n\n            with sns.axes_style(\"white\"):\n                fig, ax = plt.subplots(1, 1, figsize = (25, 7.5))\n\n                ftreimp.sort_values(ascending = False).\\\n                head(self.ntop).\\\n                plot.bar(ax = ax, color = \"tab:blue\")\n                ax.set_title(f\"Feature Importances - {method}\", **title_specs)\n\n                plt.tight_layout()\n                plt.show()\n\n            print()\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"papermill":{"duration":8.96364,"end_time":"2024-12-04T20:34:36.492063","exception":false,"start_time":"2024-12-04T20:34:27.528423","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time \n\ntarget_col = \"sii\"\nmethod     = \"LGBM1R\"\n\nn_refits   = 100","metadata":{"papermill":{"duration":0.01721,"end_time":"2024-12-04T20:34:36.517424","exception":false,"start_time":"2024-12-04T20:34:36.500214","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **DATA LOADS**","metadata":{"papermill":{"duration":0.007335,"end_time":"2024-12-04T20:34:36.532534","exception":false,"start_time":"2024-12-04T20:34:36.525199","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%time \n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    \n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname):\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df\n\ndef make_sec_ftre(\n    X : pd.DataFrame, \n    label : str = \"Train\",\n)-> pd.DataFrame:\n    \"This function uses the csv customer/ participant file and extracts features\"\n\n    PrintColor(\n        f\"\\n{'-' * 10} {label.upper()} PREPROCESSING {'-' * 10}\\n\", \n        color = Fore.RED\n    )\n    \n    df = X.copy()\n    \n    PrintColor(f\"---> Physical section features = {df.shape}\", color = Fore.CYAN)\n    df.loc[df[\"Physical-Weight\"] < 10, \"Physical-Weight\"] = np.nan\n    df.loc[df[\"Physical-Height\"] < 5,  \"Physical-Height\"] = np.nan\n    \n    for col in ['Physical-BMI', 'BIA-BIA_BMI']:\n        df.loc[df[col] <= 5, col] = np.nan\n        \n    for col in [\"Physical-Systolic_BP\", \"Physical-Diastolic_BP\"]:\n        df.loc[df[col] <= 0, col] = np.nan\n        \n    df[\"Imp_BMI\"]    = df['BIA-BIA_BMI'].fillna(df['Physical-BMI'])\n    df[\"Change_BMI\"] = df['BIA-BIA_BMI'] - df['Physical-BMI']\n    \n    df[\"Grp_PBMI\"]   = \\\n    np.select(\n        [df['Physical-BMI'].isna() == True, \n         df['Physical-BMI'] < 18.5, \n         df['Physical-BMI'] < 25, \n         df['Physical-BMI'] < 30, \n         df['Physical-BMI'] < 35, \n         df['Physical-BMI'] < 40\n        ],\n        [\"NA\", \"Underweight\", \"Normal\", \"Overweight\", \"Obese1\", \"Obese2\"],\n        \"Obese3\" \n    )\n    df[\"Grp_PBMI\"] = df[\"Grp_PBMI\"]\n    \n    df[\"Grp_BIABMI\"]   = \\\n    np.select(\n        [df['BIA-BIA_BMI'].isna() == True, \n         df['BIA-BIA_BMI'] < 18.5, \n         df['BIA-BIA_BMI'] < 25, \n         df['BIA-BIA_BMI'] < 30, \n         df['BIA-BIA_BMI'] < 35, \n         df['BIA-BIA_BMI'] < 40\n        ],\n        [\"NA\", \"Underweight\", \"Normal\", \"Overweight\", \"Obese1\", \"Obese2\"],\n        \"Obese3\"  \n    )\n    df[\"Grp_BIABMI\"] = df[\"Grp_BIABMI\"]\n    \n    df[\"Grp_Imp_BMI\"]   = \\\n    np.select(\n        [df['Imp_BMI'].isna() == True,\n         df['Imp_BMI'] < 18.5, \n         df['Imp_BMI'] < 25, \n         df['Imp_BMI'] < 30, \n         df['Imp_BMI'] < 35, \n         df['Imp_BMI'] < 40\n        ],\n        [\"NA\", \"Underweight\", \"Normal\", \"Overweight\", \"Obese1\", \"Obese2\"],\n        \"Obese3\" \n    )\n    df[\"Grp_Imp_BMI\"] = df[\"Grp_Imp_BMI\"]\n\n    df[\"Ratio_WaistHeight\"] = (df[\"Physical-Waist_Circumference\"] / df[\"Physical-Height\"]).replace([np.inf, -1*np.inf], np.nan)\n    df[\"Diff_BP\"]           = df[\"Physical-Systolic_BP\"] - df[\"Physical-Diastolic_BP\"]\n    \n    df.loc[df[\"Diff_BP\"] < 0, [\"Physical-Systolic_BP\", \"Physical-Diastolic_BP\"]] = \\\n    df.loc[df[\"Diff_BP\"] < 0, [\"Physical-Diastolic_BP\", \"Physical-Systolic_BP\"]].values\n    \n    df[\"Diff_BP\"]  = df[\"Physical-Systolic_BP\"] - df[\"Physical-Diastolic_BP\"]\n    df[\"Grp_BP\"] = \\\n    np.select(\n        [(df[\"Physical-Systolic_BP\"].isna() == True) | (df[\"Physical-Diastolic_BP\"].isna() == True),\n         (df[\"Physical-Systolic_BP\"] < 90) | (df[\"Physical-Diastolic_BP\"] < 60) ,\n         (df[\"Physical-Systolic_BP\"].between(90, 119, inclusive = \"both\")) & (df[\"Physical-Diastolic_BP\"].between(60, 79, inclusive = \"both\")),\n         (df[\"Physical-Systolic_BP\"].between(120,129, inclusive = \"both\")) & (df[\"Physical-Diastolic_BP\"] < 80),\n         (df[\"Physical-Systolic_BP\"].between(130,139, inclusive = \"both\")) | (df[\"Physical-Diastolic_BP\"].between(80, 89, inclusive = \"both\")), \n         (df[\"Physical-Systolic_BP\"] >= 140) | (df[\"Physical-Diastolic_BP\"] >= 90),\n        ],\n        [\"NA\", \"Low\", \"Normal\", \"Elevated\", \"High1\", \"High2\"],\n        \"High3\",\n    )\n    df[\"Grp_BP\"] = df[\"Grp_BP\"].astype(\"category\")\n\n    df[\"Grp_HeartRate\"] = \\\n        np.select(\n            [df[\"Physical-HeartRate\"].isna() == True,\n             df[\"Physical-HeartRate\"] < 60,\n             df[\"Physical-HeartRate\"].between(60, 100, inclusive = \"left\"),\n            ], \n            [\"NA\", \"Bradycardia\", \"Normal\"], \"Tachycardia\"\n        )\n    df[\"Grp_HeartRate\"] = df[\"Grp_HeartRate\"].astype(\"category\")\n            \n    for col in df.filter(regex = \"Grp_\", axis = 1).columns:\n        if df[col].dtype == \"category\":\n            pass\n        else:\n            try:\n                df[col] = df[col].astype(\"category\")\n            except:\n                pass\n\n    df = df.replace([np.inf, -1*np.inf], np.nan)\n    PrintColor(f\"---> All features complete = {df.shape} -- {label}\\n\", color = Fore.RED)\n    return df\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"papermill":{"duration":0.032398,"end_time":"2024-12-04T20:34:36.572003","exception":false,"start_time":"2024-12-04T20:34:36.539605","status":"completed"},"tags":[],"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time \n\n# Loading the time series dataset\nroot     = Path('/kaggle/input/child-mind-institute-problematic-internet-use')\nts_train = load_time_series(root / \"series_train.parquet\")\nts_test  = load_time_series(root / \"series_test.parquet\")\n\nPrintColor(f\"\\n\\n---> Shape = {ts_train.shape} {ts_test.shape}\")","metadata":{"papermill":{"duration":99.81919,"end_time":"2024-12-04T20:36:16.398110","exception":false,"start_time":"2024-12-04T20:34:36.578920","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **MODEL-1  VERSION 7_10**\n\nThis is a proxy model - fitted on PCIAT-1-20 and then used for sii calculation","metadata":{"papermill":{"duration":0.025162,"end_time":"2024-12-04T20:36:16.449406","exception":false,"start_time":"2024-12-04T20:36:16.424244","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%time \n\n# Preparing the dataset for the inference\ndf_train = pd.read_csv(root / 'train.csv')\ndf_test  = pd.read_csv(root / 'test.csv')\ndf_subm  = pd.read_csv(root / 'sample_submission.csv', index_col='id')\n\n# Making new features:-\ndf_train = make_sec_ftre(df_train, \"Train\", )\nprint()\ndf_test  = make_sec_ftre(df_test,  \"Test\", )\n\ntime_series_cols = ts_train.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ndf_train = pd.merge(df_train, ts_train, how=\"left\", on='id')\ndf_test  = pd.merge(df_test, ts_test,   how=\"left\", on='id')\ndf_train = df_train.set_index('id')\ndf_test  = df_test.set_index('id')\ndf_train = df_train.dropna(subset= [\"sii\"])\n\nfeature_cols = list(df_test.drop(\"id\", axis = 1, errors = \"ignore\").columns)\ncat_cols     = [c for c in feature_cols if \"Season\" in c]\nnum_cols     = list(df_test.select_dtypes(np.number).columns)\nnum_cols     = sorted(list(set(num_cols).intersection(set(feature_cols))))\n\nPrintColor(f\"\\n\\n---> Shape = {df_train.shape} {df_test.shape}\")\n\n# Preparing the targets\ntargets_df = \\\n(df_train.\n filter(regex = r\"PCIAT|sii\", axis=1).\n iloc[:, 1:].\n dropna(subset = \"sii\", axis=0).\n fillna(0).\n astype(np.uint8)\n)\n\nPrintColor(f\"\\n---> Targets \\n\")\ntargets = np.array(targets_df.columns)[0: 20]\nwith np.printoptions(linewidth = 150):\n    print(np.array(targets))\n\nprint()","metadata":{"papermill":{"duration":0.236232,"end_time":"2024-12-04T20:36:16.711068","exception":false,"start_time":"2024-12-04T20:36:16.474836","status":"completed"},"tags":[],"trusted":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time \n\n# Feature Engineering\nX     = df_train[feature_cols]\ny     = df_train[target_col]\nXtest = df_test[feature_cols]\ncv    = StratifiedKFold(5, shuffle=True, random_state=SEED)\n\nygrp  = np.zeros(len(X))\nfor fold_nb, (_, dev_idx) in enumerate(cv.split(X, y)):\n    ygrp[dev_idx] = fold_nb\n\nygrp = pd.Series(ygrp, name = \"fold_nb\", dtype = np.uint8)\n\nxform = SimpleImputer(strategy = \"mean\")\nX[num_cols]     = xform.fit_transform(X[num_cols])\nXtest[num_cols] = xform.transform(Xtest[num_cols])\n\nprint(f\"---> Imputed numerics\")\n\nxform = \\\nOrdinalEncoder(dtype=np.int32,\n               handle_unknown='use_encoded_value',\n               unknown_value=-1,\n               encoded_missing_value=-2,\n              )\nX[cat_cols]     = xform.fit_transform(X[cat_cols])\nXtest[cat_cols] = xform.transform(Xtest[cat_cols])\nprint(f\"---> Encoded categories\")\n\nPrintColor(f\"\\n\\n---> Shape = {X.shape} {Xtest.shape}\")","metadata":{"papermill":{"duration":0.121585,"end_time":"2024-12-04T20:36:16.860404","exception":false,"start_time":"2024-12-04T20:36:16.738819","status":"completed"},"tags":[],"trusted":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time \n\n# Initializing storage elements\nOOF_Preds     = {}\nAdj_OOF_Preds = {}\nFittedModels  = {}\nFtreImp       = {}\n\nOnlineModels  = {}\nOnl_Preds     = {}\nOnl_Mdl_Preds = {}\n\nlgb_params = \\\n{\n'objective'       : 'l2',\n'verbosity'       : -1,\n'n_iter'          : 200,\n'lambda_l1'       : 0.005116829730239727,\n'lambda_l2'       : 0.0011520776712645852,\n'learning_rate'   : 0.02376367323636638,\n'max_depth'       : 5,\n'num_leaves'      : 207,\n'colsample_bytree': 0.7759862336963801,\n'colsample_bynode': 0.5110355095943208,\n'bagging_fraction': 0.5485770314992224,\n'bagging_freq'    : 7,\n'min_data_in_leaf': 78,\n}\n\ndrop_cols = [\"Source\", \"id\", \"Id\", \"Label\", target_col, \"fold_nb\"]\nmd        = ModelTrainer(drop_cols = drop_cols, ntop  = 50, verbose = True)\nX         = X.drop(drop_cols, axis=1, errors = \"ignore\")\nXtest     = Xtest.drop(drop_cols, axis=1, errors = \"ignore\")\ncat_mdl_cols = list(Xtest.select_dtypes(\"category\").columns)\n\nprint(f\"\\n{'=' * 20} {method.upper()} MODEL TRAINING {'=' * 20}\\n\")\n\nfor mytarget in tqdm(targets_df.columns[0: 20]):\n    PrintColor(\n        f\"\\n ======== CURRENT TARGET = {mytarget} ======== \\n\"\n    )\n\n    model = Pipeline(steps = [(\"M\", MyLGBMR(random_state = 42, **lgb_params))])\n\n    PrintColor(\n        f\"\\n ======== OFFLINE MODELS - CV ======== \\n\", color = Fore.CYAN\n    )\n    (base_oof, adj_oof, mdl_preds, fitted_models, ftreimp, thresholds) = \\\n    md.MakeOfflineModel(\n        X,\n        targets_df[mytarget],\n        ygrp,\n        Xtest,\n        model  = model,\n        method = method,\n        ftreimp_plot_req = False,\n    )\n\n    FittedModels[mytarget] = fitted_models\n    FtreImp[mytarget]      = pd.Series(ftreimp, index = X.columns)\n    OOF_Preds[mytarget]    = base_oof\n    Adj_OOF_Preds[mytarget]= adj_oof   \n\n    PrintColor(\n        f\"\\n ======== FULL REFIT MODELS ======== \\n\", color = Fore.CYAN\n    )\n\n    reg_ = \\\n    VotingRegressor(\n        [(f\"{method}_{mystate}\" , MyLGBMR(random_state = mystate, **lgb_params))\n         for mystate in tqdm(list(range(0, n_refits, 1))) \n        ]\n    )\n    \n    onl_model = Pipeline(steps = [(\"M\", reg_)])\n    onl_model.fit(\n        X.drop(drop_cols, axis=1, errors = \"ignore\"), \n        targets_df[mytarget]\n    )\n\n    OnlineModels[mytarget]   = onl_model\n    Onl_Preds[mytarget]      = onl_model.predict(X)\n    Onl_Mdl_Preds[mytarget]  = onl_model.predict(Xtest)\n    \n","metadata":{"_kg_hide-output":true,"papermill":{"duration":1506.465528,"end_time":"2024-12-04T21:01:23.351614","exception":false,"start_time":"2024-12-04T20:36:16.886086","status":"completed"},"tags":[],"trusted":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time \n\n# Data storage and final CV score\npciat_total = pd.DataFrame(Adj_OOF_Preds).sum(axis=1)\nsii_proxy = \\\nnp.select(\n    [pciat_total <= 30, \n     pciat_total  < 50, \n     pciat_total  < 80\n    ],\n    [0,1,2],\n    3\n)\nscore = ScoreMetric(targets_df[target_col], sii_proxy)\nPrintColor(f\"\\n---> Final OOF score = {score :.6f} \\n\")\n\nPrintColor(f\"---> Prediction counts\")\nprint(np.bincount(sii_proxy))\n\nprint()\nfig, ax = plt.subplots(1,1, figsize = (4,4))\nConfusionMatrixDisplay.from_predictions(\n    targets_df[\"sii\"], \n    sii_proxy, \n    cmap = 'Blues',\n    colorbar= False,\n    text_kw = {\"fontweight\" : \"bold\", \"fontsize\" : 14},\n    ax = ax\n)\nax.set_title(\n    f\"Confusion matrix - {target_col}\\n\", \n    **{\"color\"      : \"maroon\", \n       \"fontweight\" : \"bold\", \n       \"fontsize\"   : 10,\n      }\n)\nplt.tight_layout()\nplt.show()\n\nprint();\n\n# Storing the relevant files:-\n(pd.DataFrame(Adj_OOF_Preds).\n assign(sii = sii_proxy).\n to_csv(f\"Adj_OOF_Preds_LGBM1RMLV7_10.csv\")\n)\n\njoblib.dump(OnlineModels, \"OnlineModels_LGBM1RMLV7_10.joblib\")\njoblib.dump(fitted_models,\"FittedModels_LGBM1RMLV7_10.joblib\")\n\n!ls\n\n# Submission\npciat_total = pd.DataFrame(Onl_Mdl_Preds).sum(axis=1)\nsii_proxy = \\\nnp.select(\n    [pciat_total <= 30, pciat_total < 50, pciat_total < 80],\n    [0,1,2],\n    3\n)\n\ndf_subm[\"model1\"]  = sii_proxy","metadata":{"papermill":{"duration":15.258084,"end_time":"2024-12-04T21:01:38.650058","exception":false,"start_time":"2024-12-04T21:01:23.391974","status":"completed"},"tags":[],"trusted":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **MODEL-2 VERSION 7_2**\n\nThis is a direct model fitted on sii target","metadata":{"papermill":{"duration":0.044486,"end_time":"2024-12-04T21:01:38.739015","exception":false,"start_time":"2024-12-04T21:01:38.694529","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%time \n\nroot     = Path('/kaggle/input/child-mind-institute-problematic-internet-use')\ndf_train = pd.read_csv(root / 'train.csv')\ndf_test  = pd.read_csv(root / 'test.csv')\n\ntime_series_cols = ts_train.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ndf_train = pd.merge(df_train, ts_train, how=\"left\", on='id')\ndf_test  = pd.merge(df_test, ts_test, how=\"left\", on='id')\ndf_train = df_train.set_index('id')\ndf_test  = df_test.set_index('id')\n\ncat_cols     = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n                'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', 'PAQ_A-Season', \n                'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season'\n               ]\n\nnum_cols     = ['Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-CGAS_Score', 'Physical-BMI', \n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', \n                'Fitness_Endurance-Time_Sec', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', \n                'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', \n                'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', \n                'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', \n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', \n                'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', \n                'BIA-BIA_Fat', 'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', \n                'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total',\n                'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'PreInt_EduHx-computerinternet_hoursday'\n               ]\n\ntabular_cols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex', \n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference', \n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins',\n                'Fitness_Endurance-Time_Sec', 'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', \n                'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU', \n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone',\n                'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season', 'BIA-BIA_Activity_Level_num', \n                'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', \n                'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', \n                'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season', 'PAQ_C-PAQ_C_Total', \n                'SDS-Season', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'PreInt_EduHx-Season', \n                'PreInt_EduHx-computerinternet_hoursday'\n               ]\n\nfeature_cols = tabular_cols + time_series_cols\nnum_cols     = num_cols + time_series_cols\ndf_train     = df_train.dropna(subset= [\"sii\"])\n\nPrintColor(f\"\\n\\n---> Shape = {df_train.shape} {df_test.shape}\")","metadata":{"papermill":{"duration":0.15787,"end_time":"2024-12-04T21:01:38.939749","exception":false,"start_time":"2024-12-04T21:01:38.781879","status":"completed"},"tags":[],"trusted":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time \n\nX     = df_train[feature_cols]\ny     = df_train[target_col]\nXtest = df_test[feature_cols]\ncv    = StratifiedKFold(5, shuffle=True, random_state=SEED)\n\nygrp  = np.zeros(len(X))\nfor fold_nb, (_, dev_idx) in enumerate(cv.split(X, y)):\n    ygrp[dev_idx] = fold_nb\n\nygrp = pd.Series(ygrp, name = \"fold_nb\", dtype = np.uint8)\n\nxform = SimpleImputer(strategy = \"mean\")\nX[num_cols]     = xform.fit_transform(X[num_cols])\nXtest[num_cols] = xform.transform(Xtest[num_cols])\n\nxform = \\\nOrdinalEncoder(dtype=np.int32,\n               handle_unknown='use_encoded_value',\n               unknown_value=-1,\n               encoded_missing_value=-2,\n              )\nX[cat_cols]     = xform.fit_transform(X[cat_cols])\nXtest[cat_cols] = xform.transform(Xtest[cat_cols])\n\nPrintColor(f\"\\n\\n---> Shape = {X.shape} {Xtest.shape}\")","metadata":{"papermill":{"duration":0.165804,"end_time":"2024-12-04T21:01:39.148079","exception":false,"start_time":"2024-12-04T21:01:38.982275","status":"completed"},"tags":[],"trusted":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time \n\nparams = {\n    'objective'       : 'l2',\n    'verbosity'       : -1,\n    'n_iter'          : 200,\n    'lambda_l1'       : 0.005116829730239727,\n    'lambda_l2'       : 0.0011520776712645852,\n    'learning_rate'   : 0.02376367323636638,\n    'max_depth'       : 5,\n    'num_leaves'      : 207,\n    'colsample_bytree': 0.7759862336963801,\n    'colsample_bynode': 0.5110355095943208,\n    'bagging_fraction': 0.5485770314992224,\n    'bagging_freq'    : 7,\n    'min_data_in_leaf': 78,\n}\n\nprint(f\"\\n{'=' * 20} {method.upper()} MODEL TRAINING {'=' * 20}\\n\")\ndrop_cols = [\"Source\", \"id\", \"Id\", \"Label\", target_col, \"fold_nb\"]\n\nmd    = ModelTrainer(drop_cols = drop_cols, ntop  = 50, verbose = True)\nmodel = Pipeline(steps = [(\"M\", MyLGBMR(random_state = 42, **params))])\n\n(base_oof, adj_oof, mdl_preds, fitted_models, ftreimp, thresholds) = \\\nmd.MakeOfflineModel(\n    X,\n    y,\n    ygrp,\n    Xtest,\n    model  = model,\n    method = method,\n    ftreimp_plot_req = True,\n)\n\nreg_ = \\\nVotingRegressor([\n    ('lgb_0', MyLGBMR(**params, random_state=12)),\n    ('lgb_1', MyLGBMR(**params, random_state=22)),\n    ('lgb_2', MyLGBMR(**params, random_state=32)),\n    ('lgb_3', MyLGBMR(**params, random_state=42)),\n    ('lgb_4', MyLGBMR(**params, random_state=52)),\n    ('lgb_5', MyLGBMR(**params, random_state=62)),\n    ('lgb_6', MyLGBMR(**params, random_state=72)),\n    ('lgb_7', MyLGBMR(**params, random_state=82)),\n    ('lgb_8', MyLGBMR(**params, random_state=92)),\n    ('lgb_9', MyLGBMR(**params, random_state=102)),\n    ('lgb_10', MyLGBMR(**params, random_state=777)),\n    ('lgb_11', MyLGBMR(**params, random_state=500)),\n    ('lgb_12', MyLGBMR(**params, random_state=1000)),\n    ('lgb_13', MyLGBMR(**params, random_state=5000)),\n]\n)\n\nmodel = Pipeline(steps = [(\"M\", reg_)])\nmodel.fit(X, y)\n\ndf_subm[\"model2\"] = model.predict(Xtest)\ndf_subm[\"model2\"] = np.uint8(df_subm[\"model2\"].round())\n\n\nprint()\nfig, ax = plt.subplots(1,1, figsize = (4,4))\nConfusionMatrixDisplay.from_predictions(\n    targets_df[\"sii\"], \n    adj_oof, \n    cmap = 'Blues',\n    colorbar= False,\n    text_kw = {\"fontweight\" : \"bold\", \"fontsize\" : 14},\n    ax = ax\n)\nax.set_title(\n    f\"Confusion matrix - {target_col}\\n\", \n    **{\"color\"      : \"maroon\", \n       \"fontweight\" : \"bold\", \n       \"fontsize\"   : 10,\n      }\n)\nplt.tight_layout()\nplt.show()\n\nprint();\n\n# Storing the relevant datasets\n(pd.DataFrame(adj_oof, columns = [\"sii\"]).\n to_csv(f\"Adj_OOF_Preds_LGBM1RLGBMV7_2.csv\")\n)\n\njoblib.dump(OnlineModels, \"OnlineModels_LGBM1RLGBMV7_2.joblib\")\njoblib.dump(fitted_models,\"FittedModels_LGBM1RLGBMV7_2.joblib\")\n\nprint()\n!ls","metadata":{"papermill":{"duration":26.125869,"end_time":"2024-12-04T21:02:05.315021","exception":false,"start_time":"2024-12-04T21:01:39.189152","status":"completed"},"tags":[],"trusted":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **MODEL-3 VERSION 7_24**","metadata":{"papermill":{"duration":0.043727,"end_time":"2024-12-04T21:02:05.403095","exception":false,"start_time":"2024-12-04T21:02:05.359368","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%time \n\nlgb_params = \\\n{\n'objective'       : 'l2',\n'verbosity'       : -1,\n'n_iter'          : 200,\n'lambda_l1'       : 0.005116829730239727,\n'lambda_l2'       : 0.0011520776712645852,\n'learning_rate'   : 0.02376367323636638,\n'max_depth'       : 5,\n'num_leaves'      : 207,\n'colsample_bytree': 0.7759862336963801,\n'colsample_bynode': 0.5110355095943208,\n'bagging_fraction': 0.5485770314992224,\n'bagging_freq'    : 7,\n'min_data_in_leaf': 78,\n}\n\nversion_nb = \"MLV7_24\"","metadata":{"papermill":{"duration":0.058566,"end_time":"2024-12-04T21:02:05.505497","exception":false,"start_time":"2024-12-04T21:02:05.446931","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time \n\ndef extract_stats(data):\n    return [\n        data.mean(), \n        data.std(), \n        data.max(), \n        data.min(), \n        data.diff().mean(), \n        data.diff().std(),\n    ]\n\ndef make_stats(df):\n    \"\"\"\n    Source - https://www.kaggle.com/code/mehrankazeminia/3-cmi-end-to-end\n    \"\"\"\n    \n    df[\"hours\"]  = df[\"time_of_day\"] // (3_600 * 1_000_000_000)\n    df[\"l2_xyz\"] = df[\"X\"] ** 2 +  df[\"Y\"] ** 2 + df[\"Z\"] ** 2\n    df[\"l1_xyz\"] = df[\"X\"].abs()+  df[\"Y\"].abs()+ df[\"Z\"].abs() \n    \n    features = [\n        df[\"non-wear_flag\"].mean(),\n        df[\"enmo\"][df[\"enmo\"] >= 0.05].sum(),\n    ]\n\n    # Mask1 - by day/ night/ sunrise-sunset\n    night   = ((df[\"hours\"] >= 22)  | (df[\"hours\"] <= 6))\n    day     = ((df[\"hours\"] <= 20) & (df[\"hours\"] >= 7))\n    no_mask = np.ones(len(df), dtype=bool)\n    \n    keys  = [\"enmo\", \"anglez\", \"light\", \"battery_voltage\"]\n    masks = [no_mask, night, day]\n      \n    # Iterate over keys and masks to generate the statistics\n    for key in keys:\n        for mask in masks:\n            filtered_data = df.loc[mask, key]\n            features.extend(extract_stats(filtered_data))\n\n    return features\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return make_stats(df), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = \\\n        list(\n            tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), \n                 total=len(ids))\n        )\n    \n    stats, indexes = zip(*results)\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"papermill":{"duration":0.065637,"end_time":"2024-12-04T21:02:05.619149","exception":false,"start_time":"2024-12-04T21:02:05.553512","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time \n\nroot     = Path('/kaggle/input/child-mind-institute-problematic-internet-use')\ndf_train = pd.read_csv(root / 'train.csv')\ndf_test  = pd.read_csv(root / 'test.csv')\n\nts_train = load_time_series(root / \"series_train.parquet\")\nts_test  = load_time_series(root / \"series_test.parquet\")\n\n# Making new features:-\ndf_train = make_sec_ftre(df_train, \"Train\", )\nprint()\ndf_test  = make_sec_ftre(df_test,  \"Test\", )\n\ntime_series_cols = ts_train.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ndf_train = pd.merge(df_train, ts_train, how=\"left\", on='id')\ndf_test  = pd.merge(df_test, ts_test, how=\"left\", on='id')\ndf_train = df_train.set_index('id')\ndf_test  = df_test.set_index('id')\ndf_train = df_train.dropna(subset= [\"sii\"])\n\nfeature_cols = list(df_test.drop(\"id\", axis = 1, errors = \"ignore\").columns)\ncat_cols     = [c for c in feature_cols if \"Season\" in c]\nnum_cols     = list(df_test.select_dtypes(np.number).columns)\nnum_cols     = sorted(list(set(num_cols).intersection(set(feature_cols))))\n\nPrintColor(f\"\\n\\n---> Shape = {df_train.shape} {df_test.shape}\")\n\ntargets_df = \\\n(df_train.\n filter(regex = r\"PCIAT|sii\", axis=1).\n iloc[:, 1:].\n dropna(subset = \"sii\", axis=0).\n fillna(0).\n astype(np.uint8)\n)\n\nPrintColor(f\"\\n---> Targets \\n\")\nwith np.printoptions(linewidth = 150):\n    print(np.array(targets_df.columns))","metadata":{"papermill":{"duration":109.615985,"end_time":"2024-12-04T21:03:55.282080","exception":false,"start_time":"2024-12-04T21:02:05.666095","status":"completed"},"tags":[],"trusted":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time \n\nX     = df_train[feature_cols]\ny     = df_train[target_col]\nXtest = df_test[feature_cols]\ncv    = StratifiedKFold(5, shuffle=True, random_state=SEED)\n\nygrp  = np.zeros(len(X))\nfor fold_nb, (_, dev_idx) in enumerate(cv.split(X, y)):\n    ygrp[dev_idx] = fold_nb\n\nygrp = pd.Series(ygrp, name = \"fold_nb\", dtype = np.uint8)\n\nxform = SimpleImputer(strategy = \"mean\")\nX[num_cols]     = xform.fit_transform(X[num_cols])\nXtest[num_cols] = xform.transform(Xtest[num_cols])\n\n\nxform = \\\nOrdinalEncoder(dtype=np.int32,\n               handle_unknown='use_encoded_value',\n               unknown_value=-1,\n               encoded_missing_value=-2,\n              )\nX[cat_cols]     = xform.fit_transform(X[cat_cols])\nXtest[cat_cols] = xform.transform(Xtest[cat_cols])\n\nPrintColor(f\"\\n\\n---> Shape = {X.shape} {Xtest.shape}\")","metadata":{"papermill":{"duration":0.162104,"end_time":"2024-12-04T21:03:55.512572","exception":false,"start_time":"2024-12-04T21:03:55.350468","status":"completed"},"tags":[],"trusted":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time \n\n# Initializing storage elements\nOOF_Preds     = {}\nAdj_OOF_Preds = {}\nFittedModels  = {}\nFtreImp       = {}\n\nOnlineModels  = {}\nOnl_Preds     = {}\nOnl_Mdl_Preds = {}\n\ndrop_cols = [\"Source\", \"id\", \"Id\", \"Label\", target_col, \"fold_nb\"]\nmd        = ModelTrainer(drop_cols = drop_cols, ntop  = 50, verbose = True)\nX         = X.drop(drop_cols, axis=1, errors = \"ignore\")\nXtest     = Xtest.drop(drop_cols, axis=1, errors = \"ignore\")\ncat_mdl_cols = list(Xtest.select_dtypes(\"category\").columns)\n\n \nprint(f\"\\n{'=' * 20} {method.upper()} MODEL TRAINING {'=' * 20}\\n\")\n\nfor mytarget in tqdm(targets_df.columns[0: 20]):\n    PrintColor(\n        f\"\\n ======== CURRENT TARGET = {mytarget} ======== \\n\"\n    )\n\n    model = Pipeline(steps = [(\"M\", MyLGBMR(random_state = 42, **lgb_params))])\n\n    PrintColor(\n        f\"\\n ======== OFFLINE MODELS - CV ======== \\n\", color = Fore.CYAN\n    )\n    (base_oof, adj_oof, mdl_preds, fitted_models, ftreimp, thresholds) = \\\n    md.MakeOfflineModel(\n        X,\n        targets_df[mytarget],\n        ygrp,\n        Xtest,\n        model  = model,\n        method = method,\n        ftreimp_plot_req = False,\n    )\n\n    FittedModels[mytarget] = fitted_models\n    FtreImp[mytarget]      = pd.Series(ftreimp, index = X.columns)\n    OOF_Preds[mytarget]    = base_oof\n    Adj_OOF_Preds[mytarget]= adj_oof   \n\n    PrintColor(\n        f\"\\n ======== FULL REFIT MODELS ======== \\n\", color = Fore.CYAN\n    )\n\n    reg_ = \\\n    VotingRegressor(\n        [(f\"{method}_{mystate}\" , MyLGBMR(random_state = mystate, **lgb_params))\n         for mystate in tqdm(list(range(0, n_refits, 1))) \n        ]\n    )\n    \n    onl_model = Pipeline(steps = [(\"M\", reg_)])\n    onl_model.fit(\n        X.drop(drop_cols, axis=1, errors = \"ignore\"), \n        targets_df[mytarget]\n    )\n\n    OnlineModels[mytarget]   = onl_model\n    Onl_Preds[mytarget]      = onl_model.predict(X)\n    Onl_Mdl_Preds[mytarget]  = onl_model.predict(Xtest)\n","metadata":{"papermill":{"duration":1423.958642,"end_time":"2024-12-04T21:27:39.534486","exception":false,"start_time":"2024-12-04T21:03:55.575844","status":"completed"},"tags":[],"trusted":true,"_kg_hide-output":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time \n\n# Data storage and final CV score\npciat_total = pd.DataFrame(Adj_OOF_Preds).sum(axis=1)\nsii_proxy = \\\nnp.select(\n    [pciat_total <= 30, \n     pciat_total  < 50, \n     pciat_total  < 80\n    ],\n    [0,1,2],\n    3\n)\nscore = ScoreMetric(targets_df[target_col], sii_proxy)\nPrintColor(f\"\\n---> Final OOF score = {score :.6f} \\n\")\n\npd.DataFrame(FtreImp).to_csv(f\"FtreImp_LGBM1RMLV7_24.csv\")\n(pd.DataFrame(Adj_OOF_Preds).\n assign(sii = sii_proxy).\n to_csv(f\"Adj_OOF_Preds_LGBM1RMLV7_24.csv\")\n)\npd.DataFrame(Onl_Preds).to_csv(f\"Onl_Preds_LGBM1RMLV7_24.csv\")\n\njoblib.dump(FittedModels, f\"FittedModels_LGBM1RMLV7_24.joblib\")\njoblib.dump(OnlineModels, f\"OnlineModels_LGBM1RMLV7_24.joblib\")\n\nPrintColor(f\"---> Prediction counts\")\nprint(np.bincount(sii_proxy))\n\nprint()\nfig, ax = plt.subplots(1,1, figsize = (4,4))\nConfusionMatrixDisplay.from_predictions(\n    targets_df[\"sii\"], \n    sii_proxy, \n    cmap = 'Blues',\n    colorbar= False,\n    text_kw = {\"fontweight\" : \"bold\", \"fontsize\" : 14},\n    ax = ax\n)\nax.set_title(\n    f\"Confusion matrix - {target_col}\\n\", \n    **{\"color\"      : \"maroon\", \n       \"fontweight\" : \"bold\", \n       \"fontsize\"   : 10,\n      }\n)\nplt.tight_layout()\nplt.show()\n\n\npciat_total = pd.DataFrame(Onl_Mdl_Preds).sum(axis=1)\nsii_proxy = \\\nnp.select(\n    [pciat_total <= 30, pciat_total < 50, pciat_total < 80],\n    [0,1,2],\n    3\n)\n\ndf_subm[\"model3\"]  = sii_proxy\n\n!ls\nprint()","metadata":{"papermill":{"duration":15.684552,"end_time":"2024-12-04T21:27:55.296713","exception":false,"start_time":"2024-12-04T21:27:39.612161","status":"completed"},"tags":[],"trusted":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **BLEND CV**","metadata":{"papermill":{"duration":0.079504,"end_time":"2024-12-04T21:27:55.458594","exception":false,"start_time":"2024-12-04T21:27:55.379090","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%time \n\noof_preds = \\\nnp.average(\n    np.c_[(\n        pd.read_csv(f\"/kaggle/working/Adj_OOF_Preds_LGBM1RMLV7_10.csv\")[\"sii\"].values,\n        pd.read_csv(f\"/kaggle/working/Adj_OOF_Preds_LGBM1RLGBMV7_2.csv\")[\"sii\"].values,\n        pd.read_csv(f\"/kaggle/working/Adj_OOF_Preds_LGBM1RMLV7_24.csv\")[\"sii\"].values\n    )\n    ],\n    axis=1,\n    weights = [0.3, 0.5, 0.2]\n)\n\nPrintColor(f\"---> Fused CV score = {ScoreMetric(y, oof_preds) :.6f}\")\nprint(\"\\n\\n\\n\")\ndisplay(df_subm.head(10))","metadata":{"papermill":{"duration":0.156793,"end_time":"2024-12-04T21:27:55.701741","exception":false,"start_time":"2024-12-04T21:27:55.544948","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **SUBMISSION**","metadata":{"papermill":{"duration":0.084585,"end_time":"2024-12-04T21:27:55.875436","exception":false,"start_time":"2024-12-04T21:27:55.790851","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%time \n\ndf_subm[\"sii\"] = \\\nnp.uint8(\n    np.average(\n        df_subm[[\"model1\", \"model2\", \"model3\"]].values,\n        axis=1,\n        weights = [0.3, 0.5, 0.2]\n    ).round()\n)\ndf_subm[[\"sii\"]].to_csv(\"submission.csv\")\n\n!ls\nprint()\n!head submission.csv","metadata":{"papermill":{"duration":2.758671,"end_time":"2024-12-04T21:27:58.722907","exception":false,"start_time":"2024-12-04T21:27:55.964236","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null}]}