{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":50160,"databundleVersionId":7921029}],"dockerImageVersionId":31329,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\nimport os\nimport gc\n\nimport re\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport joblib\n\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import TimeSeriesSplit, GroupKFold, StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.metrics import roc_auc_score\n\nimport lightgbm as lgb","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:15.720737Z","iopub.execute_input":"2026-05-13T09:16:15.721032Z","iopub.status.idle":"2026-05-13T09:16:23.651127Z","shell.execute_reply.started":"2026-05-13T09:16:15.721008Z","shell.execute_reply":"2026-05-13T09:16:23.650458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:23.652540Z","iopub.execute_input":"2026-05-13T09:16:23.653164Z","iopub.status.idle":"2026-05-13T09:16:23.661618Z","shell.execute_reply.started":"2026-05-13T09:16:23.653137Z","shell.execute_reply":"2026-05-13T09:16:23.660939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Prepare Functions for Data Preprocessing and Table Aggregation","metadata":{}},{"cell_type":"code","source":"class DataPrep:\n\n    # Special columns:\n    # case_id - This is the unique identifier for each credit case. You'll need this ID to join relevant tables to the base table.\n    # date_decision - This refers to the date when a decision was made regarding the approval of the loan.\n    # WEEK_NUM - This is the week number used for aggregation. In the test sample, WEEK_NUM continues sequentially from the last training value of WEEK_NUM.\n    # MONTH - This column represents the month and is intended for aggregation purposes.\n    # target - This is the target value, determined after a certain period based on whether or not the client defaulted on the specific credit case (loan).\n    # num_group1 - This is an indexing column used for the historical records of case_id in both depth=1 and depth=2 tables.\n    # num_group2 - This is the second indexing column for depth=2 tables' historical records of case_id. The order of num_group1 and num_group2 is important and will be clarified in feature definitions.\n    # All other raw columns in the tables serve as predictors. Their definitions can be found in the file feature_definitions.csv. For depth=0 tables, predictors can be directly used as features. However, for tables with depth>0, you may need to employ aggregation functions that will condense the historical records associated with each case_id into a single feature. In case num_group1 or num_group2 stands for person index (this is clear with predictor definitions) the zero index has special meaning. When num_groupN=0 it is the applicant (the person who applied for a loan).\n    \n    # Various predictors were transformed, therefore we have the following notation for similar groups of transformations\n    # P - Transform DPD (Days past due)\n    # M - Masking categories\n    # A - Transform amount\n    # D - Transform date\n    # T - Unspecified Transform\n    # L - Unspecified Transform\n    \n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n        return df\n\n    # Handle dates\n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n\n    # Filter columns\n    # If the column name is not in the reserved list and the null value ratio of the column is greater than 0.95, then delete the column.\n    # If the column name is not in the reserved list, the column data type is String, and the number of unique values is 1 or greater than 200, then delete the column.\n    @staticmethod\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.98:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 50):\n                    df = df.drop(col)\n\n        return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:23.664851Z","iopub.execute_input":"2026-05-13T09:16:23.665816Z","iopub.status.idle":"2026-05-13T09:16:23.801818Z","shell.execute_reply.started":"2026-05-13T09:16:23.665783Z","shell.execute_reply":"2026-05-13T09:16:23.801010Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Aggregator:\n    # Generate maximum aggregate expression for numeric columns\n    @staticmethod\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]+ \\\n        [pl.mean(col).alias(f\"mean_{col}\") for col in cols]+ \\\n        [pl.var(col).alias(f\"mean_{col}\") for col in cols]\n        return expr_max\n\n    # Generate maximum aggregate expression for date type columns\n    @staticmethod \n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\",)]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_last = [pl.last(col).alias(f\"min_{col}\") for col in cols]\n        expr_first = [pl.first(col).alias(f\"min_{col}\") for col in cols]\n        expr_mean = [pl.mean(col).alias(f\"min_{col}\") for col in cols]\n        return expr_max+expr_min+expr_first+expr_mean\n\n    # Generate a maximum aggregate expression for a column of type string\n    @staticmethod\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        expr_max = [pl.max(col).alias(f\"first_{col}\") for col in cols]\n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        return expr_max+expr_last\n\n\n    # Generate maximum aggregate expressions for columns of other types\n    @staticmethod\n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_mean = [pl.mean(col).alias(f\"mean_{col}\") for col in cols]\n        return expr_max+expr_mean\n\n\n    # Generate the maximum aggregate expression for a specific column \"num_group\"\n    @staticmethod\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        return expr_max\n\n\n    # Get all types of aggregate expressions\n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n        return exprs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:23.803700Z","iopub.execute_input":"2026-05-13T09:16:23.804052Z","iopub.status.idle":"2026-05-13T09:16:23.818444Z","shell.execute_reply.started":"2026-05-13T09:16:23.804009Z","shell.execute_reply":"2026-05-13T09:16:23.817819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Read a single file and preprocess it\ndef read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(DataPrep.set_table_dtypes)\n    # If the depth parameter is 1 or 2, the data is aggregated by \"case_id\"\n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:23.819370Z","iopub.execute_input":"2026-05-13T09:16:23.819624Z","iopub.status.idle":"2026-05-13T09:16:23.832068Z","shell.execute_reply.started":"2026-05-13T09:16:23.819604Z","shell.execute_reply":"2026-05-13T09:16:23.831327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Read multiple files and preprocess them\ndef read_files(regex_path, depth=None):\n    chunks = []\n    \n    for path in glob(str(regex_path)):\n        df = pl.read_parquet(path)\n        df = df.pipe(DataPrep.set_table_dtypes)\n        if depth in [1, 2]:\n            df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        chunks.append(df)\n    \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:23.832942Z","iopub.execute_input":"2026-05-13T09:16:23.833212Z","iopub.status.idle":"2026-05-13T09:16:23.845952Z","shell.execute_reply.started":"2026-05-13T09:16:23.833179Z","shell.execute_reply":"2026-05-13T09:16:23.845291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature engineering function for adding new features and merging dataframes\ndef data_preprocessing(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n    df_base = df_base.pipe(DataPrep.handle_dates)\n    return df_base","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:23.846721Z","iopub.execute_input":"2026-05-13T09:16:23.846992Z","iopub.status.idle":"2026-05-13T09:16:23.857054Z","shell.execute_reply.started":"2026-05-13T09:16:23.846972Z","shell.execute_reply":"2026-05-13T09:16:23.856276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    return df_data, cat_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:23.857916Z","iopub.execute_input":"2026-05-13T09:16:23.858183Z","iopub.status.idle":"2026-05-13T09:16:23.868393Z","shell.execute_reply.started":"2026-05-13T09:16:23.858153Z","shell.execute_reply":"2026-05-13T09:16:23.867594Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load Data Sets","metadata":{}},{"cell_type":"code","source":"ROOT = Path(\"/kaggle/input/competitions/home-credit-credit-risk-model-stability\")\n\nTRAIN_DIR = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR = ROOT / \"parquet_files\" / \"test\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:23.869352Z","iopub.execute_input":"2026-05-13T09:16:23.869537Z","iopub.status.idle":"2026-05-13T09:16:23.881555Z","shell.execute_reply.started":"2026-05-13T09:16:23.869518Z","shell.execute_reply":"2026-05-13T09:16:23.880829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pl.read_parquet(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\")\ndf.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:23.882299Z","iopub.execute_input":"2026-05-13T09:16:23.882565Z","iopub.status.idle":"2026-05-13T09:16:24.102291Z","shell.execute_reply.started":"2026-05-13T09:16:23.882545Z","shell.execute_reply":"2026-05-13T09:16:24.101513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Read all training data sets into a variable\ntraining_data_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_1_*.parquet\",1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR / \"train_applprev_2.parquet\", 2),\n        read_file(TRAIN_DIR / \"train_person_2.parquet\", 2),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n    ]\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:24.103401Z","iopub.execute_input":"2026-05-13T09:16:24.103900Z","iopub.status.idle":"2026-05-13T09:16:31.246949Z","shell.execute_reply.started":"2026-05-13T09:16:24.103876Z","shell.execute_reply":"2026-05-13T09:16:31.242265Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load and Pre-process all training datasets\ndf_train = data_preprocessing(**training_data_store)\nprint(\"train data shape:\\t\", df_train.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.247616Z","iopub.status.idle":"2026-05-13T09:16:31.247919Z","shell.execute_reply.started":"2026-05-13T09:16:31.247748Z","shell.execute_reply":"2026-05-13T09:16:31.247808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Read all test data sets into a variable\ntest_data_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n        read_files(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\",1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_applprev_2.parquet\", 2),\n        read_file(TEST_DIR / \"test_person_2.parquet\", 2),\n        read_files(TEST_DIR / \"test_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n    ]\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.249247Z","iopub.status.idle":"2026-05-13T09:16:31.249647Z","shell.execute_reply.started":"2026-05-13T09:16:31.249447Z","shell.execute_reply":"2026-05-13T09:16:31.249471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load and Pre-process all test datasets\ndf_test = data_preprocessing(**test_data_store)\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.251273Z","iopub.status.idle":"2026-05-13T09:16:31.251932Z","shell.execute_reply.started":"2026-05-13T09:16:31.251719Z","shell.execute_reply":"2026-05-13T09:16:31.251746Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# Remove useless columns\ndf_train = df_train.pipe(DataPrep.filter_cols)\ndf_test = df_test.select([col for col in df_train.columns if col != \"target\"])\n\nprint(\"train data shape:\\t\", df_train.shape)\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.252992Z","iopub.status.idle":"2026-05-13T09:16:31.253444Z","shell.execute_reply.started":"2026-05-13T09:16:31.253269Z","shell.execute_reply":"2026-05-13T09:16:31.253292Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert to Pandas\ndf_train, cat_cols = to_pandas(df_train)\ndf_test, cat_cols = to_pandas(df_test, cat_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.254580Z","iopub.status.idle":"2026-05-13T09:16:31.254954Z","shell.execute_reply.started":"2026-05-13T09:16:31.254732Z","shell.execute_reply":"2026-05-13T09:16:31.254747Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.256225Z","iopub.status.idle":"2026-05-13T09:16:31.256622Z","shell.execute_reply.started":"2026-05-13T09:16:31.256415Z","shell.execute_reply":"2026-05-13T09:16:31.256439Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA and Feature Engineering","metadata":{}},{"cell_type":"code","source":"df_feature_definition = pd.read_csv(ROOT / \"feature_definitions.csv\")\ndf_feature_definition.head(20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.257503Z","iopub.status.idle":"2026-05-13T09:16:31.257826Z","shell.execute_reply.started":"2026-05-13T09:16:31.257661Z","shell.execute_reply":"2026-05-13T09:16:31.257676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Training Data Shape & Column Suffix Counts\n\ndef cols_by_suffix(df, sfx):\n    return [c for c in df.columns if c.endswith(sfx)]\n    \nprint(\"Training Data Shape (rows, cols):\", df_train.shape)\nfor s in [\"P\",\"A\",\"M\",\"D\",\"T\",\"L\"]:\n    print(f\"#{s}-suffix columns:\", len(cols_by_suffix(df_train, s)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.259281Z","iopub.status.idle":"2026-05-13T09:16:31.259558Z","shell.execute_reply.started":"2026-05-13T09:16:31.259445Z","shell.execute_reply":"2026-05-13T09:16:31.259459Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Target Distribution\nif \"target\" in df_train.columns:\n    n = len(df_train)\n    n_pos = df_train[\"target\"].sum()\n    pos_rate = 100.0 * df_train[\"target\"].mean()\n    print(f\"\\nThere are {n:,} total number of samples.\\nThere are {int(n_pos):,} positive number of samples.\\nThe positive percentage is {pos_rate:.2f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.260482Z","iopub.status.idle":"2026-05-13T09:16:31.260834Z","shell.execute_reply.started":"2026-05-13T09:16:31.260663Z","shell.execute_reply":"2026-05-13T09:16:31.260682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Top Features with Missing Values\nmissing = (df_train.isna().mean().sort_values(ascending=False) * 100).round(2)\nprint(\"\\nTop-15 missingness (%):\")\ndisplay(missing.head(15).to_frame(\"missing_%\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.262773Z","iopub.status.idle":"2026-05-13T09:16:31.263069Z","shell.execute_reply.started":"2026-05-13T09:16:31.262944Z","shell.execute_reply":"2026-05-13T09:16:31.262966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Quick numeric summaries on 3 least-missing numeric feature columns\nexclude_cols = [\"case_id\", \"target\", \"WEEK_NUM\"]\n\nnum_cols = [\n    col for col in df_train.select_dtypes(include=[np.number]).columns\n    if col not in exclude_cols\n]\n\nif num_cols:\n    miss_num = df_train[num_cols].isna().mean()\n    chosen = miss_num.sort_values().head(3).index.tolist()\n\n    desc = pd.concat({\n        c: pd.Series({\n            \"mean\": df_train[c].mean(),\n            \"std\": df_train[c].std(),\n            \"p01\": df_train[c].quantile(0.01),\n            \"p50\": df_train[c].quantile(0.50),\n            \"p99\": df_train[c].quantile(0.99),\n            \"missing_%\": df_train[c].isna().mean() * 100,\n        }) for c in chosen\n    }, axis=1).T\n\n    print(\"\\nNumeric summaries (sampled feature columns):\")\n    display(desc)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.265358Z","iopub.status.idle":"2026-05-13T09:16:31.265720Z","shell.execute_reply.started":"2026-05-13T09:16:31.265586Z","shell.execute_reply":"2026-05-13T09:16:31.265616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" # Draw Histograms For Each Chosen Feature\nplot_sample = df_train[chosen].sample(n=min(len(df_train), 200_000), random_state=42)\nfor c in chosen:\n    plt.figure()\n    plot_sample[c].dropna().hist(bins=50)\n    plt.title(f\"Histogram: {c}\")\n    plt.xlabel(c); plt.ylabel(\"count\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.267021Z","iopub.status.idle":"2026-05-13T09:16:31.267453Z","shell.execute_reply.started":"2026-05-13T09:16:31.267259Z","shell.execute_reply":"2026-05-13T09:16:31.267316Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train Models","metadata":{}},{"cell_type":"code","source":"# # ================================================================\n# # MODELLING: LR, RF, LightGBM with robust preprocessing\n# # Metrics: AUC, Gini, Gini Stability (OOF); Pick best and submit\n# # ================================================================\n# import os, gc, numpy as np, pandas as pd\n# from pandas.api.types import is_numeric_dtype, is_bool_dtype, is_categorical_dtype\n\n# from sklearn import set_config\n# from sklearn.model_selection import StratifiedKFold, StratifiedShuffleSplit, train_test_split\n# from sklearn.metrics import roc_auc_score\n# from sklearn.compose import ColumnTransformer\n# from sklearn.preprocessing import OneHotEncoder, OrdinalEncoder, MaxAbsScaler, FunctionTransformer\n# from sklearn.impute import SimpleImputer\n# from sklearn.pipeline import Pipeline, make_pipeline\n# from sklearn.linear_model import LogisticRegression\n# from sklearn.ensemble import RandomForestClassifier\n# from sklearn.base import clone\n# from scipy import sparse\n# from sklearn.preprocessing import StandardScaler\n\n# from lightgbm import LGBMClassifier\n# import lightgbm as lgb\n\n# # ---------------- thread caps to reduce RAM spikes ----------------\n# os.environ[\"OMP_NUM_THREADS\"] = \"2\"\n# os.environ[\"OPENBLAS_NUM_THREADS\"] = \"2\"\n# os.environ[\"MKL_NUM_THREADS\"] = \"2\"\n# os.environ[\"NUMEXPR_NUM_THREADS\"] = \"2\"\n# os.environ[\"VECLIB_MAXIMUM_THREADS\"] = \"2\"\n\n# # Ensure sklearn transformers return NumPy arrays by default (not pandas)\n# set_config(transform_output=\"default\")\n\n# # --------------------- competition stability metric ---------------------\n# def gini_stability(base, score_col=\"score\", w_fallingrate=88.0, w_resstd=-0.5):\n#     gini_in_time = (\n#         base.loc[:, [\"WEEK_NUM\", \"target\", score_col]]\n#             .sort_values(\"WEEK_NUM\")\n#             .groupby(\"WEEK_NUM\")[[\"target\", score_col]]\n#             .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[score_col]) - 1)\n#             .tolist()\n#     )\n#     x = np.arange(len(gini_in_time), dtype=float)\n#     y = np.array(gini_in_time, dtype=float)\n#     a, b = np.polyfit(x, y, 1)                    # trend\n#     res_std = np.std(y - (a*x + b))               # volatility\n#     return y.mean() + w_fallingrate*min(0.0, a) + w_resstd*res_std\n\n# # --------------------- Optional quick subsample for speed ---------------------\n# FAST_SAMPLE = True\n# SAMPLE_ROWS = 450_000\n\n# df = df_train.replace([np.inf, -np.inf], np.nan).copy()\n# if FAST_SAMPLE and len(df) > SAMPLE_ROWS:\n#     idx = np.arange(len(df))\n#     _, idx_small = train_test_split(idx, train_size=SAMPLE_ROWS,\n#                                     stratify=df[\"target\"], random_state=42)\n#     df = df.iloc[idx_small].copy()\n\n# TARGET_COL = \"target\"\n# WEEK_COL   = \"WEEK_NUM\"\n# drop_cols  = {TARGET_COL, WEEK_COL, \"case_id\"}\n\n# # --------------------- Robust feature split & dtype hygiene ---------------------\n# X = df.drop(columns=list(drop_cols)).copy()\n# y = df[TARGET_COL].astype(int)\n# week_num = df[WEEK_COL].values\n\n# # initial numeric guess by dtype (exclude bools)\n# num_cols = [c for c in X.columns if is_numeric_dtype(X[c]) and not is_bool_dtype(X[c])]\n# cat_cols_all = [c for c in X.columns if c not in num_cols]\n\n# # demote any \"numeric\" columns that actually contain non-numeric tokens\n# bad_num = []\n# for c in list(num_cols):\n#     coerced = pd.to_numeric(X[c], errors=\"coerce\")\n#     frac_bad = ((~X[c].isna()) & (coerced.isna())).mean()\n#     if frac_bad > 0.0:\n#         bad_num.append(c)\n# for c in bad_num:\n#     num_cols.remove(c)\n#     if c not in cat_cols_all:\n#         cat_cols_all.append(c)\n\n# # set categoricals to category dtype; downcast numerics for memory\n# for c in cat_cols_all:\n#     if not is_categorical_dtype(X[c]):\n#         X[c] = X[c].astype(\"category\")\n# for c in num_cols:\n#     if X[c].dtype == \"float64\":\n#         X[c] = X[c].astype(\"float32\")\n#     elif X[c].dtype == \"int64\":\n#         X[c] = X[c].astype(\"int32\")\n\n# # low-card subset for LR one-hot (keeps LR sparse matrix small)\n# low_card_cats = [c for c in cat_cols_all if X[c].nunique(dropna=True) <= 30]\n\n# # helpers: keep matrices memory-friendly\n# to_csr32 = FunctionTransformer(\n#     lambda A: sparse.csr_matrix(A, dtype=np.float32) if not sparse.issparse(A) else A.astype(np.float32),\n#     accept_sparse=True\n# )\n# to_numpy32 = FunctionTransformer(\n#     lambda A: (A.to_numpy(dtype=np.float32) if isinstance(A, pd.DataFrame) else A.astype(np.float32, copy=False)),\n#     accept_sparse=False\n# )\n\n# # helper: convert any DataFrame/ndarray to a numpy object array of strings\n# cat_to_object = FunctionTransformer(\n#     lambda A: (\n#         A.astype(str).to_numpy(object) if isinstance(A, pd.DataFrame)\n#         else np.asarray(A).astype(str)\n#     ),\n#     accept_sparse=False, feature_names_out=\"one-to-one\"\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.268277Z","iopub.status.idle":"2026-05-13T09:16:31.268630Z","shell.execute_reply.started":"2026-05-13T09:16:31.268428Z","shell.execute_reply":"2026-05-13T09:16:31.268443Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # =============================== 1) LOGISTIC REGRESSION ===============================\n# to_csr_strict = FunctionTransformer(\n#     lambda A: (A.tocsr().astype(np.float32) if sparse.issparse(A)\n#                else sparse.csr_matrix(A, dtype=np.float32)),\n#     accept_sparse=True\n# )\n\n# pre_lr = ColumnTransformer(\n#     transformers=[\n#         # Numeric: impute -> force CSR -> Standardize (sparse-safe with_mean=False)\n#         (\"num\", make_pipeline(SimpleImputer(strategy=\"median\"),\n#                               to_csr_strict,\n#                               StandardScaler(with_mean=False)),\n#          num_cols),\n\n#         # Categorical (low-card only for LR): OHE sparse\n#         (\"cat_low\", OneHotEncoder(handle_unknown=\"ignore\", sparse=True), low_card_cats),\n#     ],\n#     remainder=\"drop\",\n#     sparse_threshold=1.0   # keep the combined matrix sparse\n# )\n\n# lr_pipe = Pipeline([\n#     (\"prep\", pre_lr),\n#     (\"clf\", LogisticRegression(\n#         solver=\"saga\", C=1.0, tol=1e-3, max_iter=150,\n#         n_jobs=-1, random_state=42\n#     )),\n# ])\n\n# # =============================== 2) RANDOM FOREST ===============================\n# pre_tree = ColumnTransformer(\n#     transformers=[\n#         (\"num\", SimpleImputer(strategy=\"median\"), num_cols),\n#         (\"cat\", make_pipeline(\n#             cat_to_object,  # <--- NEW: force string/object dtype so imputer won't try to float-cast\n#             SimpleImputer(strategy=\"most_frequent\"),\n#             OrdinalEncoder(handle_unknown=\"use_encoded_value\", unknown_value=-1)\n#         ), cat_cols_all),\n#     ],\n#     remainder=\"drop\",\n#     n_jobs=1\n# )\n\n# rf_pipe = Pipeline([\n#     (\"prep\", pre_tree),\n#     (\"to32\", to_numpy32),\n#     (\"clf\", RandomForestClassifier(\n#         n_estimators=160, max_depth=12, max_features=\"sqrt\",\n#         max_samples=0.6, bootstrap=True,\n#         n_jobs=2, random_state=42, class_weight=\"balanced_subsample\"\n#     )),\n# ])\n\n# # =============================== 3) LIGHTGBM ===============================\n# try:\n#     lgb_est = LGBMClassifier(\n#         n_estimators=600, learning_rate=0.06, num_leaves=64,\n#         max_depth=-1, subsample=0.8, colsample_bytree=0.8,\n#         reg_lambda=1.0, n_jobs=2, objective=\"binary\", metric=\"auc\",\n#         device_type=\"gpu\"   # falls back to CPU if no GPU\n#     )\n# except TypeError:\n#     lgb_est = LGBMClassifier(\n#         n_estimators=600, learning_rate=0.06, num_leaves=64,\n#         max_depth=-1, subsample=0.8, colsample_bytree=0.8,\n#         reg_lambda=1.0, n_jobs=2, objective=\"binary\", metric=\"auc\",\n#         device=\"gpu\"\n#     )\n\n# lgb_pipe = Pipeline([\n#     (\"prep\", pre_tree),\n#     (\"to32\", to_numpy32),\n#     (\"clf\", lgb_est),\n# ])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.270170Z","iopub.status.idle":"2026-05-13T09:16:31.270519Z","shell.execute_reply.started":"2026-05-13T09:16:31.270395Z","shell.execute_reply":"2026-05-13T09:16:31.270419Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # =============================== CV runner (OOF + GS) ===============================\n# def cv_oof_report(name, pipeline, X, y, week_num, n_splits=3, random_state=42):\n#     skf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=random_state)\n#     oof = np.zeros(len(y), dtype=np.float32)\n\n#     for fold, (tr_idx, va_idx) in enumerate(skf.split(X, y), 1):\n#         X_tr, X_va = X.iloc[tr_idx], X.iloc[va_idx]\n#         y_tr, y_va = y.iloc[tr_idx], y.iloc[va_idx]\n\n#         model = clone(pipeline)\n#         fit_params = {}\n#         if name.lower().startswith(\"lgb\"):\n#             fit_params = {\n#                 \"clf__eval_set\": [(X_va, y_va)],\n#                 # Use callbacks for early stopping & silence logs\n#                 \"clf__callbacks\": [lgb.early_stopping(50, verbose=False),\n#                                    lgb.log_evaluation(period=-1)]\n#             }\n\n#         model.fit(X_tr, y_tr, **fit_params)\n#         oof[va_idx] = model.predict_proba(X_va)[:, 1].astype(np.float32)\n\n#         print(f\"[{name}] Fold {fold}/{n_splits} AUC={roc_auc_score(y_va, oof[va_idx]):.5f}\")\n\n#         del X_tr, X_va, y_tr, y_va, model\n#         gc.collect()\n\n#     auc  = roc_auc_score(y, oof)\n#     gini = 2*auc - 1\n#     base = pd.DataFrame({WEEK_COL: week_num, \"target\": y.values, \"score\": oof})\n#     gs   = gini_stability(base)\n\n#     print(f\"\\n[{name}] OOF AUC={auc:.5f} | Gini={gini:.5f} | GiniStability={gs:.5f}\\n\")\n#     return {\"model\": name, \"oof_auc\": auc, \"oof_gini\": gini, \"gini_stability\": gs, \"oof\": oof, \"pipe\": pipeline}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.271618Z","iopub.status.idle":"2026-05-13T09:16:31.272063Z","shell.execute_reply.started":"2026-05-13T09:16:31.271843Z","shell.execute_reply":"2026-05-13T09:16:31.271875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # =============================== Run CV for all models ===============================\n# out_lr  = cv_oof_report(\"LogReg_saga\",  lr_pipe, X, y, week_num)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.273459Z","iopub.status.idle":"2026-05-13T09:16:31.273845Z","shell.execute_reply.started":"2026-05-13T09:16:31.273659Z","shell.execute_reply":"2026-05-13T09:16:31.273681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# out_rf  = cv_oof_report(\"RandomForest\", rf_pipe, X, y, week_num)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.274481Z","iopub.status.idle":"2026-05-13T09:16:31.274873Z","shell.execute_reply.started":"2026-05-13T09:16:31.274672Z","shell.execute_reply":"2026-05-13T09:16:31.274696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# out_lgb = cv_oof_report(\"LGB_fast\",     lgb_pipe, X, y, week_num)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.276335Z","iopub.status.idle":"2026-05-13T09:16:31.276790Z","shell.execute_reply.started":"2026-05-13T09:16:31.276552Z","shell.execute_reply":"2026-05-13T09:16:31.276575Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# summary = pd.DataFrame([out_lr, out_rf, out_lgb]).drop(columns=[\"oof\", \"pipe\"]).sort_values(\"oof_auc\", ascending=False)\n# print(\"=== CV Summary ===\")\n# print(summary.to_string(index=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.277856Z","iopub.status.idle":"2026-05-13T09:16:31.278203Z","shell.execute_reply.started":"2026-05-13T09:16:31.278075Z","shell.execute_reply":"2026-05-13T09:16:31.278098Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # =============================== Pick best, refit, predict test, submit ===============================\n# SELECT_BY = \"gini_stability\"   # or \"oof_auc\"\n# best = max([out_lr, out_rf, out_lgb], key=lambda d: d[SELECT_BY])\n# print(f\"[BEST MODEL] {best['model']} by {SELECT_BY}: {best[SELECT_BY]:.5f}\")\n\n# best_pipe = clone(best[\"pipe\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.278728Z","iopub.status.idle":"2026-05-13T09:16:31.279126Z","shell.execute_reply.started":"2026-05-13T09:16:31.278934Z","shell.execute_reply":"2026-05-13T09:16:31.278949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Refit: LGB uses a tiny validation split for early stopping; others fit on full data\n# if best[\"model\"].lower().startswith(\"lgb\"):\n#     sss = StratifiedShuffleSplit(n_splits=1, test_size=0.10, random_state=42)\n#     tr_idx, va_idx = next(sss.split(X, y))\n#     X_tr, y_tr = X.iloc[tr_idx], y.iloc[tr_idx]\n#     X_va, y_va = X.iloc[va_idx], y.iloc[va_idx]\n\n#     best_pipe.fit(\n#         X_tr, y_tr,\n#         clf__eval_set=[(X_va, y_va)],\n#         clf__callbacks=[lgb.early_stopping(100, verbose=False),\n#                         lgb.log_evaluation(period=-1)]\n#     )\n\n#     del X_tr, X_va, y_tr, y_va\n#     gc.collect()\n# else:\n#     best_pipe.fit(X, y)\n\n\n# # --------------------- Build X_test aligned to training columns ---------------------\n# TEST_DROP = [\"case_id\", \"WEEK_NUM\"]\n# case_ids = df_test[\"case_id\"].values\n# X_test = df_test.drop(columns=[c for c in TEST_DROP if c in df_test.columns]).copy()\n\n# # add any train-only columns missing in test as NaN\n# missing_in_test = [c for c in X.columns if c not in X_test.columns]\n# for c in missing_in_test:\n#     X_test[c] = np.nan\n\n# # drop extras and match column order exactly\n# X_test = X_test[X.columns]\n\n# # enforce dtypes consistent with training split\n# for c in X_test.columns:\n#     if c in cat_cols_all:\n#         if not is_categorical_dtype(X_test[c]):\n#             X_test[c] = X_test[c].astype(\"category\")\n#     else:\n#         X_test[c] = pd.to_numeric(X_test[c], errors=\"coerce\").astype(\"float32\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.280537Z","iopub.status.idle":"2026-05-13T09:16:31.280947Z","shell.execute_reply.started":"2026-05-13T09:16:31.280706Z","shell.execute_reply":"2026-05-13T09:16:31.280730Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# LGB For Submission","metadata":{}},{"cell_type":"code","source":"# ---------------- Custom Voting Model ----------------\n# It averages predictions from multiple trained estimators (e.g., LGBM models).\nclass VotingModel(BaseEstimator, RegressorMixin):\n    def __init__(self, estimators):\n        super().__init__()\n        self.estimators = estimators\n\n    def fit(self, X, y=None):\n        # No fitting needed here; models are already trained.\n        return self\n\n    def predict(self, X):\n        # Average the predicted values from all models.\n        y_preds = [estimator.predict(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)\n\n    def predict_proba(self, X):\n        # Average the predicted probabilities from all models.\n        y_preds = [estimator.predict_proba(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.281832Z","iopub.status.idle":"2026-05-13T09:16:31.282192Z","shell.execute_reply.started":"2026-05-13T09:16:31.282004Z","shell.execute_reply":"2026-05-13T09:16:31.282027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ---------------- Data Preparation ----------------\n# Define training features (X), target labels (y), and group info (weeks)\ndf_train=reduce_mem_usage(df_train)\nX = df_train.drop(columns=[\"target\", \"case_id\", \"WEEK_NUM\"])\ny = df_train[\"target\"]\nweeks = df_train[\"WEEK_NUM\"]\n\n# ---------------- Cross-validation setup ----------------\n# StratifiedGroupKFold ensures balanced label distribution and that each week stays within a single fold (to prevent leakage)\ncv = StratifiedGroupKFold(n_splits=5, shuffle=False)\n\n# ---------------- LightGBM model parameters ----------------\n# LGBM model parameters after the cross-validation process\nparams = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"max_depth\": 10,\n    \"learning_rate\": 0.03,\n    \"max_bin\": 255,\n    \"n_estimators\": 3000,\n    \"colsample_bytree\": 0.8,\n    \"colsample_bynode\": 0.8,\n    \"verbose\": -1,\n    \"random_state\": 42,\n    \"reg_alpha\": 0.2,\n    \"reg_lambda\": 10,\n    \"extra_trees\": True,\n    \"num_leaves\": 64,\n    \"device\": \"gpu\", \n}\n\n# ---------------- Model training loop ----------------\nfitted_models = []\ncv_scores = []\nevals_results = []\n\n# Loop through each fold for training and validation\nfor idx_train, idx_valid in cv.split(X, y, groups=weeks):\n    X_train, y_train = X.iloc[idx_train], y.iloc[idx_train]\n    X_valid, y_valid = X.iloc[idx_valid], y.iloc[idx_valid]\n\n    print(\"Validation week range: \", (weeks.iloc[idx_valid].min(), weeks.iloc[idx_valid].max()))\n\n    # Initialize and train the LightGBM model\n    model = lgb.LGBMClassifier(**params)\n    evals_result = {}\n\n    model.fit(\n        X_train, y_train,\n        eval_set=[(X_train, y_train), (X_valid, y_valid)],\n        eval_names=[\"train\", \"valid\"],\n        callbacks=[\n            lgb.record_evaluation(evals_result),\n            lgb.log_evaluation(50),\n            lgb.early_stopping(50),\n        ]\n)\n\n    evals_results.append(evals_result)\n\n    # Save the trained model\n    fitted_models.append(model)\n\n    # Evaluate AUC on the validation fold\n    y_pred_valid = model.predict_proba(X_valid)[:, 1]\n    auc_score = roc_auc_score(y_valid, y_pred_valid)\n    cv_scores.append(auc_score)\n\n# ---------------- Ensemble & Results ----------------\n# Combine all fold models into a simple average voting model\nmodel = VotingModel(fitted_models)\n\n# Print all CV fold scores and their mean\nprint(\"CV AUC scores: \", cv_scores)\nprint(\"Average CV AUC score: \", sum(cv_scores) / len(cv_scores))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.283361Z","iopub.status.idle":"2026-05-13T09:16:31.283733Z","shell.execute_reply.started":"2026-05-13T09:16:31.283519Z","shell.execute_reply":"2026-05-13T09:16:31.283544Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_idx = np.argmax(cv_scores)\nlgb.plot_importance(fitted_models[best_idx], importance_type=\"split\", figsize=(10, 100))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.284554Z","iopub.status.idle":"2026-05-13T09:16:31.284855Z","shell.execute_reply.started":"2026-05-13T09:16:31.284672Z","shell.execute_reply":"2026-05-13T09:16:31.284685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport pandas as pd\nimport numpy as np\n\nlgb_auc_df = pd.DataFrame({\n    \"fold\": np.arange(1, len(cv_scores) + 1),\n    \"auc\": cv_scores,\n})\n\nmean_auc = lgb_auc_df[\"auc\"].mean()\n\nplt.figure(figsize=(8, 4))\nplt.bar(lgb_auc_df[\"fold\"], lgb_auc_df[\"auc\"], color=\"#4C78A8\")\nplt.axhline(mean_auc, color=\"red\", linestyle=\"--\", linewidth=1.5, label=f\"Mean AUC = {mean_auc:.5f}\")\n\nfor _, row in lgb_auc_df.iterrows():\n    plt.text(\n        row[\"fold\"],\n        row[\"auc\"] + 0.0003,\n        f\"{row['auc']:.5f}\",\n        ha=\"center\",\n        va=\"bottom\",\n        fontsize=9,\n    )\n\nplt.xlabel(\"Fold\")\nplt.ylabel(\"AUC\")\nplt.title(\"LightGBM CV AUC by Fold\")\nplt.ylim(lgb_auc_df[\"auc\"].min() - 0.003, lgb_auc_df[\"auc\"].max() + 0.003)\nplt.xticks(lgb_auc_df[\"fold\"])\nplt.grid(axis=\"y\", alpha=0.3)\nplt.legend()\nplt.tight_layout()\nplt.show()\n\ndisplay(lgb_auc_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.285980Z","iopub.status.idle":"2026-05-13T09:16:31.286295Z","shell.execute_reply.started":"2026-05-13T09:16:31.286129Z","shell.execute_reply":"2026-05-13T09:16:31.286161Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Save Models For Inference\n","metadata":{}},{"cell_type":"code","source":"# Save trained models and metadata for a separate inference-only notebook.\n# After this notebook finishes, save a Kaggle version and add its Output as input to the inference notebook.\nMODEL_OUTPUT_DIR = Path(\"/kaggle/working/models\")\nMODEL_OUTPUT_DIR.mkdir(parents=True, exist_ok=True)\n\nfeature_names = list(X.columns)\nmodel_metadata = {\n    \"feature_names\": feature_names,\n    \"cat_cols\": list(cat_cols),\n    \"cv_scores\": cv_scores,\n    \"params\": params,\n}\n\njoblib.dump(fitted_models, MODEL_OUTPUT_DIR / \"lgb_models.pkl\")\njoblib.dump(model_metadata, MODEL_OUTPUT_DIR / \"model_metadata.pkl\")\n\nprint(f\"Saved {len(fitted_models)} LightGBM models to {MODEL_OUTPUT_DIR}\")\nprint(f\"Saved {len(feature_names)} feature names and {len(cat_cols)} categorical columns\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.287293Z","iopub.status.idle":"2026-05-13T09:16:31.287604Z","shell.execute_reply.started":"2026-05-13T09:16:31.287471Z","shell.execute_reply":"2026-05-13T09:16:31.287488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_names = X.columns\n\nlgb_importance_df = pd.DataFrame({\n    \"feature\": feature_names,\n    \"importance\": np.mean(\n        [m.booster_.feature_importance(importance_type=\"gain\") for m in fitted_models],\n        axis=0\n    )\n}).sort_values(\"importance\", ascending=False)\n\ndisplay(lgb_importance_df.head(50))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.288439Z","iopub.status.idle":"2026-05-13T09:16:31.288762Z","shell.execute_reply.started":"2026-05-13T09:16:31.288588Z","shell.execute_reply":"2026-05-13T09:16:31.288603Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.colors as mcolors\nimport pandas as pd\nimport numpy as np\n\ntop_n = 20\n\ntable_df = lgb_importance_df.head(top_n).copy()\ntable_df = table_df.reset_index(drop=True)\ntable_df.insert(0, \"rank\", np.arange(1, len(table_df) + 1))\ntable_df[\"importance\"] = table_df[\"importance\"].round(4)\n\ntable_df[\"feature\"] = table_df[\"feature\"].astype(str).str.slice(0, 45)\n\nfig, ax = plt.subplots(figsize=(13, 11))\nax.axis(\"off\")\n\ntitle = f\"Top {top_n} LightGBM Feature Importance\"\nax.set_title(title, fontsize=20, fontweight=\"bold\", pad=22)\n\ncell_text = table_df[[\"rank\", \"feature\", \"importance\"]].values\ncol_labels = [\"Rank\", \"Feature\", \"Importance\"]\n\ntable = ax.table(\n    cellText=cell_text,\n    colLabels=col_labels,\n    cellLoc=\"left\",\n    colLoc=\"left\",\n    loc=\"center\",\n    colWidths=[0.10, 0.68, 0.22],\n)\n\ntable.auto_set_font_size(False)\ntable.set_fontsize(10)\ntable.scale(1, 1.45)\n\nheader_color = \"#1f2937\"\nheader_text = \"white\"\nstripe_color = \"#f3f4f6\"\nwhite = \"#ffffff\"\nedge_color = \"#d1d5db\"\n\nimportance_values = table_df[\"importance\"].astype(float).values\nnorm = mcolors.Normalize(vmin=importance_values.min(), vmax=importance_values.max())\ncmap = plt.cm.Blues\n\nfor (row, col), cell in table.get_celld().items():\n    cell.set_edgecolor(edge_color)\n    cell.set_linewidth(0.6)\n\n    if row == 0:\n        cell.set_facecolor(header_color)\n        cell.set_text_props(color=header_text, weight=\"bold\", fontsize=11)\n        cell.set_height(0.05)\n    else:\n        if col == 2:\n            value = importance_values[row - 1]\n            cell.set_facecolor(cmap(0.25 + 0.55 * norm(value)))\n            cell.set_text_props(color=\"#111827\", weight=\"bold\")\n        else:\n            cell.set_facecolor(stripe_color if row % 2 == 0 else white)\n            cell.set_text_props(color=\"#111827\")\n\n        if col == 0:\n            cell.set_text_props(weight=\"bold\", ha=\"center\")\n        if col == 2:\n            cell.set_text_props(ha=\"right\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.290557Z","iopub.status.idle":"2026-05-13T09:16:31.290936Z","shell.execute_reply.started":"2026-05-13T09:16:31.290723Z","shell.execute_reply":"2026-05-13T09:16:31.290745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.colors as mcolors\nimport pandas as pd\nimport numpy as np\n\n# Build LightGBM feature importance from fold models\nfeature_names = X.columns\n\nlgb_importance_df = pd.DataFrame({\n    \"feature\": feature_names,\n    \"importance\": np.mean(\n        [m.booster_.feature_importance(importance_type=\"gain\") for m in fitted_models],\n        axis=0\n    )\n}).sort_values(\"importance\", ascending=False)\n\n# Normalize to percentage scale\nlgb_importance_df[\"importance_pct\"] = (\n    lgb_importance_df[\"importance\"] / lgb_importance_df[\"importance\"].sum() * 100\n)\n\ndisplay(lgb_importance_df.head(50))\n\n# Pretty table image\ntop_n = 20\n\ntable_df = lgb_importance_df.head(top_n).copy()\ntable_df = table_df.reset_index(drop=True)\ntable_df.insert(0, \"rank\", np.arange(1, len(table_df) + 1))\ntable_df[\"importance_pct\"] = table_df[\"importance_pct\"].round(3)\ntable_df[\"feature\"] = table_df[\"feature\"].astype(str).str.slice(0, 45)\n\nfig, ax = plt.subplots(figsize=(13, 11))\nax.axis(\"off\")\n\ntitle = f\"Top {top_n} LightGBM Feature Importance\"\nax.set_title(title, fontsize=20, fontweight=\"bold\", pad=22)\n\ncell_text = table_df[[\"rank\", \"feature\", \"importance_pct\"]].values\ncol_labels = [\"Rank\", \"Feature\", \"Importance (%)\"]\n\ntable = ax.table(\n    cellText=cell_text,\n    colLabels=col_labels,\n    cellLoc=\"left\",\n    colLoc=\"left\",\n    loc=\"center\",\n    colWidths=[0.10, 0.68, 0.22],\n)\n\ntable.auto_set_font_size(False)\ntable.set_fontsize(10)\ntable.scale(1, 1.45)\n\nheader_color = \"#1f2937\"\nheader_text = \"white\"\nstripe_color = \"#f3f4f6\"\nwhite = \"#ffffff\"\nedge_color = \"#d1d5db\"\n\nimportance_values = table_df[\"importance_pct\"].astype(float).values\nnorm = mcolors.Normalize(vmin=importance_values.min(), vmax=importance_values.max())\ncmap = plt.cm.Blues\n\nfor (row, col), cell in table.get_celld().items():\n    cell.set_edgecolor(edge_color)\n    cell.set_linewidth(0.6)\n\n    if row == 0:\n        cell.set_facecolor(header_color)\n        cell.set_text_props(color=header_text, weight=\"bold\", fontsize=11)\n        cell.set_height(0.05)\n    else:\n        if col == 2:\n            value = importance_values[row - 1]\n            cell.set_facecolor(cmap(0.25 + 0.55 * norm(value)))\n            cell.set_text_props(color=\"#111827\", weight=\"bold\", ha=\"right\")\n        else:\n            cell.set_facecolor(stripe_color if row % 2 == 0 else white)\n            cell.set_text_props(color=\"#111827\")\n\n        if col == 0:\n            cell.set_text_props(weight=\"bold\", ha=\"center\")\n\nplt.tight_layout()\nplt.savefig(\"/kaggle/working/lgb_feature_importance_table_pretty.png\", dpi=250, bbox_inches=\"tight\")\nplt.show()\n\n# Save raw table too\nlgb_importance_df.to_csv(\"/kaggle/working/lgb_feature_importance.csv\", index=False)\nprint(\"Saved:\")\nprint(\"/kaggle/working/lgb_feature_importance_table_pretty.png\")\nprint(\"/kaggle/working/lgb_feature_importance.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.293011Z","iopub.status.idle":"2026-05-13T09:16:31.293361Z","shell.execute_reply.started":"2026-05-13T09:16:31.293176Z","shell.execute_reply":"2026-05-13T09:16:31.293198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nfold_id = 1\nresult = evals_results[fold_id - 1]\n\nplt.figure(figsize=(10, 5))\nplt.plot(result[\"train\"][\"auc\"], label=\"Train AUC\", linewidth=2)\nplt.plot(result[\"valid\"][\"auc\"], label=\"Valid AUC\", linewidth=2)\nplt.xlabel(\"Iteration\")\nplt.ylabel(\"AUC\")\nplt.title(f\"LightGBM Train vs Valid AUC - Fold {fold_id}\")\nplt.grid(alpha=0.3)\nplt.legend()\nplt.tight_layout()\nplt.savefig(f\"/kaggle/working/lgb_train_valid_auc_fold{fold_id}.png\", dpi=200, bbox_inches=\"tight\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.294048Z","iopub.status.idle":"2026-05-13T09:16:31.294405Z","shell.execute_reply.started":"2026-05-13T09:16:31.294207Z","shell.execute_reply":"2026-05-13T09:16:31.294229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\n\nfor i, result in enumerate(evals_results, 1):\n    plt.plot(result[\"valid\"][\"auc\"], label=f\"Fold {i}\", alpha=0.85)\n\nplt.xlabel(\"Iteration\")\nplt.ylabel(\"Valid AUC\")\nplt.title(\"LightGBM Validation AUC Curve by Fold\")\nplt.grid(alpha=0.3)\nplt.legend()\nplt.tight_layout()\nplt.savefig(\"/kaggle/working/lgb_valid_auc_curve_by_fold.png\", dpi=200, bbox_inches=\"tight\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T09:16:31.296032Z","iopub.status.idle":"2026-05-13T09:16:31.296375Z","shell.execute_reply.started":"2026-05-13T09:16:31.296199Z","shell.execute_reply":"2026-05-13T09:16:31.296221Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Training notebook output\nThis notebook intentionally stops after saving trained model artifacts under `/kaggle/working/models`. Use the separate inference notebook to generate `submission.csv`.\n","metadata":{}}]}