{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:white; font-size:150%; text-align:left;padding:3.0px; background: #607BB0; border-bottom: 8px solid #53599A; border-radius: 5px;\">Description<br><div>\n\n**First and foremost** this notebook is inspired/based on blog post by Max Halford. Please refer to his original blog post here for more details https://maxhalford.github.io/blog/lightgbm-focal-loss/ . I have simply used his idea and applied it to competitions data. Others who use LGBM are welcome to try his focal loss custom function for training model. Public **LB** score for model with `binary` objective (see notebook  version 3) is very similiar to **LB** score by model using `FocalLoss` objective (see notebook  version 4). Notebook version 5 is intented to have results from models trained using different objectives, hence long run time.\n    \n## <div style= \"font-family: Cambria; font-weight:bold; letter-spacing: 0px; color:white; font-size:150%; text-align:left;padding:3.0px; background: #607BB0; border-bottom: 8px solid #53599A; border-radius: 5px;\"> Table of contents<br><div>\n    \n<a id=\"toc\"></a>\n- [1 Prepare project](#1)\n- [2 Helper functions](#2)\n- [3 Custom loss function](#3)\n- [4 Load train data](#4)\n- [5 Train model using `binary` objective](#5)\n- [6 Train model using `FocalLoss` objective](#6)\n- [7 Validate models](#7)\n    - [First model (`binary` objective)](#7.1)\n    - [Second model (`FocalLoss` objective)](#7.2)\n- [8 Submit predictions](#8)","metadata":{}},{"cell_type":"markdown","source":"<a id=\"1\"></a>\n# <b>1 <span style='color:#53599A'>Prepare project</span></b>\n\nLoad required python packages.\n\n<div>\n<br>\n<a href=\"#toc\" style=\"background-color: #607BB0; color: #ffffff; padding: 7px 10px; text-decoration: none; border-radius: 50px;\">Back to top</a><a id=\"toc\"></a>\n</div>","metadata":{}},{"cell_type":"code","source":"# suppress enjoying warnings from seaborn\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\n# for fiding file names\nfrom pathlib import Path\nfrom glob import glob\nimport gc\nimport re\n\n# data processing libraries\nimport polars as pl\nimport numpy as np\nimport pandas as pd\n\n# for custom objective\nfrom scipy import optimize\nfrom scipy import special\n\n# for visualizing data\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# sklearn functions\nfrom sklearn.model_selection import train_test_split, StratifiedGroupKFold\nfrom sklearn.preprocessing import LabelEncoder\n# for validating models\nfrom sklearn.metrics import roc_auc_score, roc_curve\n\n# LightGBM modeling\nimport lightgbm as lgb\n\n# define default colors for plots in notebook\nfrom matplotlib import cycler\nfrom matplotlib.colors import LinearSegmentedColormap\nCOLORS = [\"#068D9D\", \"#53599A\", \"#607BB0\", \"#6D9DC5\", \"#77BECF\", \"#80DED9\", \"#AEECEF\"]\nplt.rc('axes', facecolor='#E6E6E6', edgecolor='none', axisbelow=True, grid=True, prop_cycle=cycler('color', COLORS))\n\n# project CONSTANTS\nROOT = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR = ROOT / \"parquet_files\" / \"test\"\nSEED = 123456789","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-09T16:53:37.781764Z","iopub.execute_input":"2024-05-09T16:53:37.782591Z","iopub.status.idle":"2024-05-09T16:53:40.379998Z","shell.execute_reply.started":"2024-05-09T16:53:37.782550Z","shell.execute_reply":"2024-05-09T16:53:40.378938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    root_dir = Path(\"/kaggle/input/home-credit-credit-risk-model-stability/\")\n    train_dir = Path(\"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train\")\n    test_dir = Path(\"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/test\")","metadata":{"execution":{"iopub.status.busy":"2024-05-09T16:53:40.382056Z","iopub.execute_input":"2024-05-09T16:53:40.382608Z","iopub.status.idle":"2024-05-09T16:53:40.388277Z","shell.execute_reply.started":"2024-05-09T16:53:40.382568Z","shell.execute_reply":"2024-05-09T16:53:40.387243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"2\"></a>\n# <b>2 <span style='color:#53599A'>Helper functions</span></b>\n\n<div>\n<br>\n<a href=\"#toc\" style=\"background-color: #607BB0; color: #ffffff; padding: 7px 10px; text-decoration: none; border-radius: 50px;\">Back to top</a><a id=\"toc\"></a>\n</div>","metadata":{}},{"cell_type":"code","source":"def reduce_mem_usage(df, float16_as32=True):\n    \"\"\" \n    iterate through all the columns of a dataframe and modify the data type\n    to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    if float16_as32:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        df[col] = df[col].astype(np.float16)                    \n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-09T16:53:40.389800Z","iopub.execute_input":"2024-05-09T16:53:40.390207Z","iopub.status.idle":"2024-05-09T16:53:40.404714Z","shell.execute_reply.started":"2024-05-09T16:53:40.390169Z","shell.execute_reply":"2024-05-09T16:53:40.403525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Pipeline:\n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.95:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n\n        return df","metadata":{"execution":{"iopub.status.busy":"2024-05-09T16:53:40.407496Z","iopub.execute_input":"2024-05-09T16:53:40.407853Z","iopub.status.idle":"2024-05-09T16:53:40.422422Z","shell.execute_reply.started":"2024-05-09T16:53:40.407824Z","shell.execute_reply":"2024-05-09T16:53:40.421501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Aggregator:\n    def __init__(\n        self, \n        num_aggregators=[pl.max, pl.min, pl.first, pl.last, pl.mean], \n        str_aggregators=[pl.max, pl.min, pl.first, pl.last],  # n_unique\n        group_aggregators=[pl.max, pl.min, pl.first, pl.last],\n        str_mode=True\n    ):\n        self.num_aggregators = num_aggregators\n        self.str_aggregators = str_aggregators\n        self.group_aggregators = group_aggregators\n        self.str_mode=str_mode\n    \n    def num_expr(self, df_cols):\n        cols = [col for col in df_cols if col[-1] in (\"P\", \"A\")]\n        expr_all = []\n        for method in self.num_aggregators:\n            expr = [method(col).alias(f\"{method.__name__}_{col}\") for col in cols]\n            expr_all += expr\n\n        return expr_all\n\n    def date_expr(self, df_cols):\n        cols = [col for col in df_cols if col[-1] in (\"D\",)]\n        expr_all = []\n        for method in self.num_aggregators:\n            expr = [method(col).alias(f\"{method.__name__}_{col}\") for col in cols]  \n            expr_all += expr\n\n        return expr_all\n\n    def str_expr(self, df_cols):\n        cols = [col for col in df_cols if col[-1] in (\"M\",)]\n        \n        expr_all = []\n        for method in self.str_aggregators:\n            expr = [method(col).alias(f\"{method.__name__}_{col}\") for col in cols]  \n            expr_all += expr\n            \n        if self.str_mode:\n            expr_mode = [\n                pl.col(col)\n                .drop_nulls()\n                .mode()\n                .first()\n                .alias(f\"mode_{col}\")\n                for col in cols\n            ]\n        else:\n            expr_mode = []\n\n        return expr_all + expr_mode\n\n    def other_expr(self, df_cols):\n        cols = [col for col in df_cols if col[-1] in (\"T\", \"L\")]\n        \n        expr_all = []\n        for method in self.str_aggregators:\n            expr = [method(col).alias(f\"{method.__name__}_{col}\") for col in cols]  \n            expr_all += expr\n\n        return expr_all\n    \n    def count_expr(self, df_cols):\n        cols = [col for col in df_cols if \"num_group\" in col]\n\n        expr_all = []\n        for method in self.group_aggregators:\n            expr = [method(col).alias(f\"{method.__name__}_{col}\") for col in cols]  \n            expr_all += expr\n\n        return expr_all\n\n    def get_exprs(self, df_cols):\n        exprs = (\n            self.num_expr(df_cols) + \n            self.date_expr(df_cols) + \n            self.str_expr(df_cols) + \n            self.other_expr(df_cols) + \n            self.count_expr(df_cols)\n        )\n\n        return exprs","metadata":{"execution":{"iopub.status.busy":"2024-05-09T16:53:40.424272Z","iopub.execute_input":"2024-05-09T16:53:40.424705Z","iopub.status.idle":"2024-05-09T16:53:40.441877Z","shell.execute_reply.started":"2024-05-09T16:53:40.424649Z","shell.execute_reply":"2024-05-09T16:53:40.440741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_files_by_path(pattern_path, aggregator, depth=None, agg_chunks=False, num_group1_filter=None):\n    chunks = []\n    for i, path in enumerate(glob(str(pattern_path))):\n        print(\"  chunk: \", i)\n        chunk = pl.read_parquet(path).pipe(Pipeline.set_table_dtypes)\n        if agg_chunks:\n            if num_group1_filter != None:\n                chunk = chunk.filter((pl.col(\"num_group1\") == num_group1_filter)).drop(columns=[\"num_group1\"])\n            chunk = chunk.group_by(\"case_id\").agg(aggregator.get_exprs(chunk.columns))\n        chunks.append(chunk)\n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    \n    if depth in [1, 2]:\n        print(f\"  agg, depth {depth}\")\n        if num_group1_filter != None:\n            df = df.filter((pl.col(\"num_group1\") == num_group1_filter)).drop(columns=[\"num_group1\"])\n        df = df.group_by(\"case_id\").agg(aggregator.get_exprs(df.columns))\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-09T16:53:40.443517Z","iopub.execute_input":"2024-05-09T16:53:40.443926Z","iopub.status.idle":"2024-05-09T16:53:40.458537Z","shell.execute_reply.started":"2024-05-09T16:53:40.443889Z","shell.execute_reply":"2024-05-09T16:53:40.457521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_files(files_arr, data_dir, aggregator, mode=\"train\", agg_chunks=False, num_group1_filter=None):\n    base_file_name = f\"{mode}_base.parquet\"\n    print(\"  files: \", base_file_name)\n    feats_df = read_files_by_path(\n        data_dir / base_file_name, data_dir, mode\n    )\n    \n    for i, file_name in enumerate(files_arr):\n        depth = re.findall(\"\\w_(\\d)\", file_name)\n        if len(depth) > 0:\n            depth = depth[0]\n        else:\n            continue\n        print(\"  files: \", file_name, f\"(depth: {depth})\")\n        files_df = read_files_by_path(\n            data_dir / f\"{mode}{file_name}\", \n            aggregator, int(depth), agg_chunks, \n            num_group1_filter=num_group1_filter\n        )\n        feats_df = feats_df.join(\n            files_df, \n            how=\"left\", on=\"case_id\", suffix=f\"_{depth}_{i}\"\n        )\n        del files_df\n        gc.collect()\n\n    return feats_df","metadata":{"execution":{"iopub.status.busy":"2024-05-09T16:53:40.460227Z","iopub.execute_input":"2024-05-09T16:53:40.460578Z","iopub.status.idle":"2024-05-09T16:53:40.470254Z","shell.execute_reply.started":"2024-05-09T16:53:40.460546Z","shell.execute_reply":"2024-05-09T16:53:40.469113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n    return df_data","metadata":{"execution":{"iopub.status.busy":"2024-05-09T16:53:40.471803Z","iopub.execute_input":"2024-05-09T16:53:40.472148Z","iopub.status.idle":"2024-05-09T16:53:40.487090Z","shell.execute_reply.started":"2024-05-09T16:53:40.472120Z","shell.execute_reply":"2024-05-09T16:53:40.485659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_df(\n    files_arr, data_dir, aggregator, \n    mode=\"train\", cat_cols=None, train_cols=[], \n    agg_chunks=False, feat_eng=True, num_group1_filter=None\n):\n    print()\n    print(\"Collecting data...\")\n    feats_df = read_files(files_arr, data_dir, aggregator, mode=mode, agg_chunks=agg_chunks)\n    print(\"  feats_df shape:\\t\", feats_df.shape)\n    \n    if feat_eng:\n        print(\"Feature Engineering...\")\n        feats_df = feats_df.with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n        neworder = feats_df.columns\n        neworder.remove(\"month_decision\")\n        neworder.remove(\"weekday_decision\")\n        neworder.insert(5, \"month_decision\")\n        neworder.insert(6, \"weekday_decision\")\n        feats_df = feats_df.select(neworder)\n    \n    feats_df = feats_df.pipe(Pipeline.handle_dates)\n    \n    print(\"Filter cols...\")\n    if mode == \"train\":\n        feats_df = feats_df.pipe(Pipeline.filter_cols)\n    else:\n        train_cols = feats_df.columns if len(train_cols) == 0 else train_cols\n        feats_df = feats_df.select([col for col in train_cols if col != \"target\"])\n    print(\"  feats_df shape:\\t\", feats_df.shape)\n    \n    print(\"Convert to pandas...\")\n    feats_df = to_pandas(feats_df, cat_cols)\n    return feats_df","metadata":{"execution":{"iopub.status.busy":"2024-05-09T16:53:40.488325Z","iopub.execute_input":"2024-05-09T16:53:40.488642Z","iopub.status.idle":"2024-05-09T16:53:40.500126Z","shell.execute_reply.started":"2024-05-09T16:53:40.488612Z","shell.execute_reply":"2024-05-09T16:53:40.498914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"3\"></a>\n# <b>3 <span style='color:#53599A'>Custom loss function</span></b>\n\n<div>\n<br>\n<a href=\"#toc\" style=\"background-color: #607BB0; color: #ffffff; padding: 7px 10px; text-decoration: none; border-radius: 50px;\">Back to top</a><a id=\"toc\"></a>\n</div>","metadata":{}},{"cell_type":"code","source":"class FocalLoss:\n    \n    \"\"\"\n    Function taken from: https://maxhalford.github.io/blog/lightgbm-focal-loss/\n    \"\"\"\n\n    def __init__(self, gamma, alpha=None):\n        self.alpha = alpha\n        self.gamma = gamma\n\n    def at(self, y):\n        if self.alpha is None:\n            return np.ones_like(y)\n        return np.where(y, self.alpha, 1 - self.alpha)\n\n    def pt(self, y, p):\n        p = np.clip(p, 1e-15, 1 - 1e-15)\n        return np.where(y, p, 1 - p)\n\n    def __call__(self, y_true, y_pred):\n        at = self.at(y_true)\n        pt = self.pt(y_true, y_pred)\n        return -at * (1 - pt) ** self.gamma * np.log(pt)\n\n    def grad(self, y_true, y_pred):\n        y = 2 * y_true - 1  # {0, 1} -> {-1, 1}\n        at = self.at(y_true)\n        pt = self.pt(y_true, y_pred)\n        g = self.gamma\n        return at * y * (1 - pt) ** g * (g * pt * np.log(pt) + pt - 1)\n\n    def hess(self, y_true, y_pred):\n        y = 2 * y_true - 1  # {0, 1} -> {-1, 1}\n        at = self.at(y_true)\n        pt = self.pt(y_true, y_pred)\n        g = self.gamma\n\n        u = at * y * (1 - pt) ** g\n        du = -at * y * g * (1 - pt) ** (g - 1)\n        v = g * pt * np.log(pt) + pt - 1\n        dv = g * np.log(pt) + g + 1\n\n        return (du * v + u * dv) * y * (pt * (1 - pt))\n\n    def init_score(self, y_true):\n        res = optimize.minimize_scalar(\n            lambda p: self(y_true, p).sum(),\n            bounds=(0, 1),\n            method='bounded'\n        )\n        p = res.x\n        log_odds = np.log(p / (1 - p))\n        return log_odds\n\n    def lgb_obj(self, preds, train_data):\n        y = train_data.get_label()\n        p = special.expit(preds)\n        return self.grad(y, p), self.hess(y, p)\n\n    def lgb_eval(self, preds, train_data):\n        y = train_data.get_label()\n        p = special.expit(preds)\n        is_higher_better = False\n        return 'focal_loss', self(y, p).mean(), is_higher_better","metadata":{"execution":{"iopub.status.busy":"2024-05-09T16:53:40.504635Z","iopub.execute_input":"2024-05-09T16:53:40.505171Z","iopub.status.idle":"2024-05-09T16:53:40.522904Z","shell.execute_reply.started":"2024-05-09T16:53:40.505136Z","shell.execute_reply":"2024-05-09T16:53:40.521163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"4\"></a>\n# <b>4 <span style='color:#53599A'>Load train data</span></b>\n\n<div>\n<br>\n<a href=\"#toc\" style=\"background-color: #607BB0; color: #ffffff; padding: 7px 10px; text-decoration: none; border-radius: 50px;\">Back to top</a><a id=\"toc\"></a>\n</div>","metadata":{}},{"cell_type":"code","source":"base_files = [\n    \"_static_cb_0.parquet\",\n    \"_static_0_*.parquet\",\n    \"_applprev_1_*.parquet\",\n    \"_tax_registry_a_1.parquet\",\n    \"_tax_registry_b_1.parquet\",\n    \"_tax_registry_c_1.parquet\",\n    \"_other_1.parquet\",\n    \"_person_1.parquet\",\n    \"_deposit_1.parquet\",\n    \"_debitcard_1.parquet\",\n    \"_credit_bureau_b_1.parquet\",\n    \"_credit_bureau_b_2.parquet\",\n]\nbase_agg = Aggregator(\n    num_aggregators = [pl.max, pl.min, pl.first, pl.last, pl.mean],\n    str_aggregators = [pl.max, pl.min, pl.first, pl.last],\n    group_aggregators = [pl.max, pl.min, pl.first, pl.last],\n    str_mode = True\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-09T16:53:40.524252Z","iopub.execute_input":"2024-05-09T16:53:40.524564Z","iopub.status.idle":"2024-05-09T16:53:40.539422Z","shell.execute_reply.started":"2024-05-09T16:53:40.524540Z","shell.execute_reply":"2024-05-09T16:53:40.538160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nif __name__ == '__main__':\n    train_base_df = prepare_df(base_files, CFG.train_dir, base_agg)\n    cat_cols_base = list(train_base_df.select_dtypes(\"category\").columns)\n    gc.collect()\n    display(train_base_df)","metadata":{"execution":{"iopub.status.busy":"2024-05-09T16:53:40.540830Z","iopub.execute_input":"2024-05-09T16:53:40.541216Z","iopub.status.idle":"2024-05-09T16:55:53.148174Z","shell.execute_reply.started":"2024-05-09T16:53:40.541180Z","shell.execute_reply":"2024-05-09T16:55:53.147081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# all columns for training\nTRAIN_COLS = train_base_df.columns[3:].values.tolist()","metadata":{"execution":{"iopub.status.busy":"2024-05-09T16:55:53.149257Z","iopub.execute_input":"2024-05-09T16:55:53.149546Z","iopub.status.idle":"2024-05-09T16:55:53.154570Z","shell.execute_reply.started":"2024-05-09T16:55:53.149522Z","shell.execute_reply":"2024-05-09T16:55:53.153525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ncase_ids_train, case_ids_test = train_test_split(train_base_df['case_id'], train_size=0.5, random_state=SEED)\n\nX_train = train_base_df.loc[train_base_df['case_id'].isin(case_ids_train), TRAIN_COLS + [\"WEEK_NUM\"]]\nX_test = train_base_df.loc[train_base_df['case_id'].isin(case_ids_test), TRAIN_COLS + [\"WEEK_NUM\"] + ['target']]\n\ny_train = train_base_df.loc[train_base_df['case_id'].isin(case_ids_train), 'target']\ny_test = train_base_df.loc[train_base_df['case_id'].isin(case_ids_test), 'target']","metadata":{"execution":{"iopub.status.busy":"2024-05-09T16:55:53.156046Z","iopub.execute_input":"2024-05-09T16:55:53.156384Z","iopub.status.idle":"2024-05-09T16:56:00.694234Z","shell.execute_reply.started":"2024-05-09T16:55:53.156357Z","shell.execute_reply":"2024-05-09T16:56:00.692980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"5\"></a>\n# <b>5 <span style='color:#53599A'>Train model using `binary` objective</span></b>\n\n<div>\n<br>\n<a href=\"#toc\" style=\"background-color: #607BB0; color: #ffffff; padding: 7px 10px; text-decoration: none; border-radius: 50px;\">Back to top</a><a id=\"toc\"></a>\n</div>","metadata":{}},{"cell_type":"code","source":"%%time\nNO_MODELS = 5\n\n# use week numbers for group spliting\nweeks = X_train['WEEK_NUM']\n\ncv = StratifiedGroupKFold(n_splits = NO_MODELS, shuffle=False)\n\nensemble_models_0 = {}\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(cv.split(X_train, y_train, groups=weeks)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    # parameters were taken from this notebook:\n    # https://www.kaggle.com/code/greysky/home-credit-baseline\n    lgb_params = {\n        \"boosting_type\": \"gbdt\",\n        'objective' : 'binary',\n        'metric' : 'auc',\n        \"learning_rate\": 0.05,\n        \"n_estimators\": 1500,\n        \"colsample_bytree\": 0.8, \n        \"colsample_bynode\": 0.8,\n        \"verbose\": -1,\n        \"reg_alpha\": 0.1,\n        \"reg_lambda\": 10,\n        \"extra_trees\": True,\n        'num_leaves': 64,\n        \"random_state\": SEED + i\n    }\n            \n    # TRAIN DATA\n    train_x = X_train.iloc[train_index][TRAIN_COLS]\n    train_y = y_train.iloc[train_index]\n\n    # VALID DATA\n    valid_x = X_train.iloc[test_index][TRAIN_COLS]\n    valid_y = y_train.iloc[test_index]\n    \n    # CONVERT TO DataSet\n    train_dataset = lgb.Dataset(train_x, label=train_y)\n    valid_dataset = lgb.Dataset(valid_x, label=valid_y, reference=train_dataset)\n    \n    # TRAIN MODEL\n    model = lgb.train(\n        lgb_params,\n        train_dataset,\n        valid_sets = valid_dataset,\n        callbacks = [lgb.log_evaluation(100), lgb.early_stopping(50)]\n    )\n    \n    # VALIDATE MODELS\n    preds = model.predict(valid_x, num_iteration=model.best_iteration)\n    \n    # SCORE ON VALID SAMPLE\n    auc = roc_auc_score(valid_y, preds)\n    gini = 2 * auc - 1\n    print(f\"\\nValidation AUC: {auc:.3f}\")\n    print(f\"Validation GINI: {gini:.3f}\")\n\n    # SAVE MODEL\n    ensemble_models_0[f'model_{i}'] = model\n        \n    print()","metadata":{"execution":{"iopub.status.busy":"2024-05-09T16:56:00.695839Z","iopub.execute_input":"2024-05-09T16:56:00.696257Z","iopub.status.idle":"2024-05-09T17:31:24.415571Z","shell.execute_reply.started":"2024-05-09T16:56:00.696218Z","shell.execute_reply":"2024-05-09T17:31:24.414226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"6\"></a>\n# <b>6 <span style='color:#53599A'>Train model using `FocalLoss` objective</span></b>\n\n<div>\n<br>\n<a href=\"#toc\" style=\"background-color: #607BB0; color: #ffffff; padding: 7px 10px; text-decoration: none; border-radius: 50px;\">Back to top</a><a id=\"toc\"></a>\n</div>","metadata":{}},{"cell_type":"code","source":"%%time\nNO_MODELS = 5\n\nfl = FocalLoss(alpha=None, gamma=0)\n\n# use week numbers for group spliting\nweeks = X_train['WEEK_NUM']\n\ncv = StratifiedGroupKFold(n_splits = NO_MODELS, shuffle=False)\n\nensemble_models_1 = {}\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(cv.split(X_train, y_train, groups=weeks)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    # parameters were taken from this notebook:\n    # https://www.kaggle.com/code/greysky/home-credit-baseline\n    lgb_params = {\n        \"boosting_type\": \"gbdt\",\n        'objective' : fl.lgb_obj,\n        'metric' : 'auc',\n        \"learning_rate\": 0.05,\n        \"n_estimators\": 1500,\n        \"colsample_bytree\": 0.8, \n        \"colsample_bynode\": 0.8,\n        \"verbose\": -1,\n        \"reg_alpha\": 0.1,\n        \"reg_lambda\": 10,\n        \"extra_trees\": True,\n        'num_leaves': 64,\n        \"random_state\": SEED + i\n    }\n            \n    # TRAIN DATA\n    train_x = X_train.iloc[train_index][TRAIN_COLS]\n    train_y = y_train.iloc[train_index]\n\n    # VALID DATA\n    valid_x = X_train.iloc[test_index][TRAIN_COLS]\n    valid_y = y_train.iloc[test_index]\n    \n    # CONVERT TO DataSet\n    train_dataset = lgb.Dataset(train_x, label=train_y)\n    valid_dataset = lgb.Dataset(valid_x, label=valid_y, reference=train_dataset)\n    \n    # TRAIN MODEL\n    model = lgb.train(\n        lgb_params,\n        train_dataset,\n        valid_sets = valid_dataset,\n        callbacks = [lgb.log_evaluation(100), lgb.early_stopping(50)],\n        feval = fl.lgb_eval\n    )\n    \n    # VALIDATE MODELS\n    preds = model.predict(valid_x, num_iteration=model.best_iteration)\n    \n    # SCORE ON VALID SAMPLE\n    auc = roc_auc_score(valid_y, preds)\n    gini = 2 * auc - 1\n    print(f\"\\nValidation AUC: {auc:.3f}\")\n    print(f\"Validation GINI: {gini:.3f}\")\n\n    # SAVE MODEL\n    ensemble_models_1[f'model_{i}'] = model\n        \n    print()","metadata":{"execution":{"iopub.status.busy":"2024-05-09T17:31:24.417165Z","iopub.execute_input":"2024-05-09T17:31:24.417523Z","iopub.status.idle":"2024-05-09T18:08:59.748216Z","shell.execute_reply.started":"2024-05-09T17:31:24.417493Z","shell.execute_reply":"2024-05-09T18:08:59.746795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"7\"></a>\n# <b>7 <span style='color:#53599A'>Validate models</span></b>\n\n<div>\n<br>\n<a href=\"#toc\" style=\"background-color: #607BB0; color: #ffffff; padding: 7px 10px; text-decoration: none; border-radius: 50px;\">Back to top</a><a id=\"toc\"></a>\n</div>","metadata":{}},{"cell_type":"code","source":"def validate_model(data, output = \"Chart\"):\n    \"\"\"\n    Input:\n        preds, array like object\n        target, array like object\n        chart, boolean, if true return matplolib else table of summary statistics\n    Output:\n        pandas DataFrame or matplolib object\n        \n    Needs equal length prediction and taget column vectors\n    \"\"\"\n    # create DataFrame for plotting and summary stats\n    _df = data.copy()\n    _roc = data.loc[:, [\"WEEK_NUM\", \"target\", \"pred\"]]\\\n                   .sort_values(\"WEEK_NUM\").groupby(\"WEEK_NUM\")[[\"target\", \"pred\"]]\\\n                   .apply(lambda x: roc_auc_score(x[\"target\"], x[\"pred\"]))\n    _df['ROC_AUC'] = _df['WEEK_NUM'].map(_roc)\n    _df['GINI'] = 2 * _df['ROC_AUC'] - 1\n    \n    # stability score\n    x = _df[['WEEK_NUM', 'GINI']].drop_duplicates()['WEEK_NUM']\n    y = _df[['WEEK_NUM', 'GINI']].drop_duplicates()['GINI']\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = _df[['WEEK_NUM', 'GINI']].drop_duplicates()['GINI'].mean()\n    w_fallingrate = 88.0\n    w_resstd = -0.5 \n    # calculate competition score\n    score = avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std\n    \n    if output == \"Chart\":\n        fig, ax = plt.subplots(2, 2, figsize=(16, 10))\n        \n        # temporal changes of gini scores\n        _df_plot = _df.groupby('WEEK_NUM')['GINI'].mean().reset_index()\n        ax[0, 0].plot(_df_plot['WEEK_NUM'], _df_plot['GINI'], \"o-\", label = 'Weekly GINI scores')\n        # add linear fit\n        ax[0, 0].plot(_df_plot['WEEK_NUM'], _df_plot['WEEK_NUM'] * a + b, \"r--\", label = f'linear fit, k={a:.3f}')\n        ax[0, 0].set_ylabel(\"Gini score\")\n        ax[0, 0].set_xlabel(\"Week number\")\n        \n        # GINI score boxplot\n        sns.boxplot(data=_df_plot, x=\"GINI\", ax = ax[0, 1], color = COLORS[2])\n        ax[0, 1].set_xlabel(\"Gini score\")\n        ax[0, 1].set_title(f\"Q1: {np.quantile(_df_plot['GINI'].values, 0.25):.3f}, Q3: {np.quantile(_df_plot['GINI'].values, 0.75):.3f}\")\n        \n        # GINI score histogram\n        ax[1, 1].hist(_df_plot['GINI'], bins=20, label=\"Weekly GINI scores\", color = COLORS[2])\n        ax[1, 1].set_ylabel(\"Count\")\n        ax[1, 1].set_xlabel(\"Gini score\")\n        ax[1, 1].set_title(f\"Mean: {_df_plot['GINI'].mean():.3f}, STD: {_df_plot['GINI'].std():.3f}\")\n        \n        # GINI vs number of observations\n        _df_plot = _df.groupby('WEEK_NUM').agg({\"GINI\": ['mean', \"count\"]})\n        sns.regplot(x=_df_plot.iloc[:, 1], y=_df_plot.iloc[:, 0], ax=ax[1, 0], label=f'Corr. {_df_plot.corr().values[0, 1]:.3f}', color = COLORS[1])\n        ax[1, 0].set_ylabel(\"Weekly Gini score\")\n        ax[1, 0].set_xlabel(\"Number of applications\")\n        \n        # add white backgrounds to legends\n        for i in range(2):\n            for ii in range(2):\n                legend = ax[i, ii].legend(frameon=1)\n                frame = legend.get_frame()\n                frame.set_facecolor('w')\n        \n    # return DataFrame with summary stats\n    elif output == \"Stats\":\n        _df_stats = pd.DataFrame({\"dataset\": [\"test\"]})\n        _df_stats['Avg. weekly ROC AUC'] = _df[['WEEK_NUM', 'ROC_AUC']].drop_duplicates()['ROC_AUC'].mean().round(3)\n        _df_stats['Std. of weekly ROC AUC'] = _df[['WEEK_NUM', 'ROC_AUC']].drop_duplicates()['ROC_AUC'].std().round(3)\n        _df_stats['Avg. weekly GINI'] = _df[['WEEK_NUM', 'GINI']].drop_duplicates()['GINI'].mean().round(3)\n        _df_stats['Std. of weekly GINI'] = _df[['WEEK_NUM', 'GINI']].drop_duplicates()['GINI'].std().round(3)\n        _df_stats['Stability score'] = round(score, 3)\n        \n        return _df_stats\n    else:\n        return _df","metadata":{"execution":{"iopub.status.busy":"2024-05-09T18:08:59.749727Z","iopub.execute_input":"2024-05-09T18:08:59.750112Z","iopub.status.idle":"2024-05-09T18:08:59.772377Z","shell.execute_reply.started":"2024-05-09T18:08:59.750038Z","shell.execute_reply":"2024-05-09T18:08:59.771302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"7.1\"></a>\n## <b>7.1 <span style='color:#53599A'>First model (`binary` objective)</span></b>\n\n<div>\n<br>\n<a href=\"#toc\" style=\"background-color: #607BB0; color: #ffffff; padding: 7px 10px; text-decoration: none; border-radius: 50px;\">Back to top</a><a id=\"toc\"></a>\n</div>","metadata":{}},{"cell_type":"code","source":"%%time\n# first model with binary objective\ndf_preds = pd.DataFrame()\nfor i in range(NO_MODELS):\n    model = ensemble_models_0[f'model_{i}'] \n    df_preds[f'preds_{i}'] = model.predict(X_test[TRAIN_COLS], num_iteration=model.best_iteration)\n    \n# add average predictions\nX_test['pred'] = df_preds.mean(axis = 1).values","metadata":{"execution":{"iopub.status.busy":"2024-05-09T18:08:59.773741Z","iopub.execute_input":"2024-05-09T18:08:59.774061Z","iopub.status.idle":"2024-05-09T18:12:21.998979Z","shell.execute_reply.started":"2024-05-09T18:08:59.774034Z","shell.execute_reply":"2024-05-09T18:12:21.997773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validate_model(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-09T18:12:22.000567Z","iopub.execute_input":"2024-05-09T18:12:22.001184Z","iopub.status.idle":"2024-05-09T18:12:26.455601Z","shell.execute_reply.started":"2024-05-09T18:12:22.001128Z","shell.execute_reply":"2024-05-09T18:12:26.454177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validate_model(X_test, \"Stats\")","metadata":{"execution":{"iopub.status.busy":"2024-05-09T18:12:26.457042Z","iopub.execute_input":"2024-05-09T18:12:26.457464Z","iopub.status.idle":"2024-05-09T18:12:29.579893Z","shell.execute_reply.started":"2024-05-09T18:12:26.457425Z","shell.execute_reply":"2024-05-09T18:12:29.578906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"7.2\"></a>\n## <b>7.2 <span style='color:#53599A'>Second model (`FocalLoss` objective)</span></b>\n\n<div>\n<br>\n<a href=\"#toc\" style=\"background-color: #607BB0; color: #ffffff; padding: 7px 10px; text-decoration: none; border-radius: 50px;\">Back to top</a><a id=\"toc\"></a>\n</div>","metadata":{}},{"cell_type":"code","source":"%%time\n# second model with FocalLoss objective\ndf_preds = pd.DataFrame()\nfor i in range(NO_MODELS):\n    model = ensemble_models_1[f'model_{i}'] \n    df_preds[f'preds_{i}'] = model.predict(X_test[TRAIN_COLS], num_iteration=model.best_iteration)\n    \n# add average predictions\nX_test['pred'] = df_preds.mean(axis = 1).values","metadata":{"execution":{"iopub.status.busy":"2024-05-09T18:12:29.581215Z","iopub.execute_input":"2024-05-09T18:12:29.581595Z","iopub.status.idle":"2024-05-09T18:15:39.163407Z","shell.execute_reply.started":"2024-05-09T18:12:29.581562Z","shell.execute_reply":"2024-05-09T18:15:39.161983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validate_model(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-09T18:15:39.165132Z","iopub.execute_input":"2024-05-09T18:15:39.165601Z","iopub.status.idle":"2024-05-09T18:15:43.484210Z","shell.execute_reply.started":"2024-05-09T18:15:39.165553Z","shell.execute_reply":"2024-05-09T18:15:43.482837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validate_model(X_test, \"Stats\")","metadata":{"execution":{"iopub.status.busy":"2024-05-09T18:15:43.485842Z","iopub.execute_input":"2024-05-09T18:15:43.486235Z","iopub.status.idle":"2024-05-09T18:15:46.619257Z","shell.execute_reply.started":"2024-05-09T18:15:43.486202Z","shell.execute_reply":"2024-05-09T18:15:46.618063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# clean memory\ndel X_train, X_test, y_train, y_test, case_ids_train, case_ids_test, df_preds\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-09T18:15:46.622868Z","iopub.execute_input":"2024-05-09T18:15:46.623211Z","iopub.status.idle":"2024-05-09T18:15:46.833908Z","shell.execute_reply.started":"2024-05-09T18:15:46.623185Z","shell.execute_reply":"2024-05-09T18:15:46.832504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"8\"></a>\n# <b>8 <span style='color:#53599A'>Submit predictions</span></b>\n\n<div>\n<br>\n<a href=\"#toc\" style=\"background-color: #607BB0; color: #ffffff; padding: 7px 10px; text-decoration: none; border-radius: 50px;\">Back to top</a><a id=\"toc\"></a>\n</div>","metadata":{}},{"cell_type":"code","source":"if __name__ == '__main__':\n    test_base_df = prepare_df(\n        base_files, CFG.test_dir, base_agg, mode=\"test\", cat_cols=cat_cols_base, train_cols=train_base_df.columns\n    )\n    display(test_base_df)","metadata":{"execution":{"iopub.status.busy":"2024-05-09T18:15:46.835536Z","iopub.execute_input":"2024-05-09T18:15:46.836010Z","iopub.status.idle":"2024-05-09T18:15:48.515210Z","shell.execute_reply.started":"2024-05-09T18:15:46.835970Z","shell.execute_reply":"2024-05-09T18:15:48.514244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# which model to use to make submission predictions\nmodel_number = 1\nassert model_number in [0, 1]","metadata":{"execution":{"iopub.status.busy":"2024-05-09T18:15:48.516618Z","iopub.execute_input":"2024-05-09T18:15:48.516945Z","iopub.status.idle":"2024-05-09T18:15:48.523967Z","shell.execute_reply.started":"2024-05-09T18:15:48.516918Z","shell.execute_reply":"2024-05-09T18:15:48.522942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf_preds = pd.DataFrame()\nfor i in range(NO_MODELS):\n    if model_number:\n        model = ensemble_models_1[f'model_{i}']\n    else:\n        model = ensemble_models_0[f'model_{i}']\n    df_preds[f'preds_{i}'] = model.predict(test_base_df[TRAIN_COLS], num_iteration=model.best_iteration)","metadata":{"execution":{"iopub.status.busy":"2024-05-09T18:15:48.525387Z","iopub.execute_input":"2024-05-09T18:15:48.525799Z","iopub.status.idle":"2024-05-09T18:15:49.057198Z","shell.execute_reply.started":"2024-05-09T18:15:48.525764Z","shell.execute_reply":"2024-05-09T18:15:49.055763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_sub = df_sub.set_index(\"case_id\")\ndf_sub[\"score\"] = df_preds.mean(axis = 1).values","metadata":{"execution":{"iopub.status.busy":"2024-05-09T18:15:49.062797Z","iopub.execute_input":"2024-05-09T18:15:49.063155Z","iopub.status.idle":"2024-05-09T18:15:49.082149Z","shell.execute_reply.started":"2024-05-09T18:15:49.063125Z","shell.execute_reply":"2024-05-09T18:15:49.080974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-05-09T18:15:49.083377Z","iopub.execute_input":"2024-05-09T18:15:49.083699Z","iopub.status.idle":"2024-05-09T18:15:49.094939Z","shell.execute_reply.started":"2024-05-09T18:15:49.083653Z","shell.execute_reply":"2024-05-09T18:15:49.093881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}