{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":50160,"databundleVersionId":7921029}],"dockerImageVersionId":31329,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Home Credit CatBoost Only\n","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.simplefilter(action=\"ignore\", category=FutureWarning)\n\nimport os\nimport gc\nimport re\nimport joblib\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.metrics import roc_auc_score\nfrom catboost import CatBoostClassifier, Pool\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T18:27:58.454308Z","iopub.execute_input":"2026-05-11T18:27:58.454904Z","iopub.status.idle":"2026-05-11T18:28:01.805420Z","shell.execute_reply.started":"2026-05-11T18:27:58.454871Z","shell.execute_reply":"2026-05-11T18:28:01.804431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T18:28:01.806486Z","iopub.execute_input":"2026-05-11T18:28:01.807032Z","iopub.status.idle":"2026-05-11T18:28:01.818688Z","shell.execute_reply.started":"2026-05-11T18:28:01.807002Z","shell.execute_reply":"2026-05-11T18:28:01.817717Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Pipeline\n","metadata":{}},{"cell_type":"code","source":"class DataPrep:\n\n    # Special columns:\n    # case_id - This is the unique identifier for each credit case. You'll need this ID to join relevant tables to the base table.\n    # date_decision - This refers to the date when a decision was made regarding the approval of the loan.\n    # WEEK_NUM - This is the week number used for aggregation. In the test sample, WEEK_NUM continues sequentially from the last training value of WEEK_NUM.\n    # MONTH - This column represents the month and is intended for aggregation purposes.\n    # target - This is the target value, determined after a certain period based on whether or not the client defaulted on the specific credit case (loan).\n    # num_group1 - This is an indexing column used for the historical records of case_id in both depth=1 and depth=2 tables.\n    # num_group2 - This is the second indexing column for depth=2 tables' historical records of case_id. The order of num_group1 and num_group2 is important and will be clarified in feature definitions.\n    # All other raw columns in the tables serve as predictors. Their definitions can be found in the file feature_definitions.csv. For depth=0 tables, predictors can be directly used as features. However, for tables with depth>0, you may need to employ aggregation functions that will condense the historical records associated with each case_id into a single feature. In case num_group1 or num_group2 stands for person index (this is clear with predictor definitions) the zero index has special meaning. When num_groupN=0 it is the applicant (the person who applied for a loan).\n    \n    # Various predictors were transformed, therefore we have the following notation for similar groups of transformations\n    # P - Transform DPD (Days past due)\n    # M - Masking categories\n    # A - Transform amount\n    # D - Transform date\n    # T - Unspecified Transform\n    # L - Unspecified Transform\n    \n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n        return df\n\n    # Handle dates\n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n\n    # Filter columns\n    # If the column name is not in the reserved list and the null value ratio of the column is greater than 0.95, then delete the column.\n    # If the column name is not in the reserved list, the column data type is String, and the number of unique values is 1 or greater than 200, then delete the column.\n    @staticmethod\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.95:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 50):\n                    df = df.drop(col)\n\n        return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T18:28:01.821097Z","iopub.execute_input":"2026-05-11T18:28:01.821566Z","iopub.status.idle":"2026-05-11T18:28:01.845638Z","shell.execute_reply.started":"2026-05-11T18:28:01.821526Z","shell.execute_reply":"2026-05-11T18:28:01.844705Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Aggregator:\n    # Generate maximum aggregate expression for numeric columns\n    @staticmethod\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]+ \\\n        [pl.min(col).alias(f\"min_{col}\") for col in cols]+ \\\n        [pl.mean(col).alias(f\"mean_{col}\") for col in cols]\n        return expr_max\n\n    # Generate maximum aggregate expression for date type columns\n    @staticmethod \n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\",)]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        return expr_max+expr_min\n\n    # Generate a maximum aggregate expression for a column of type string\n    @staticmethod\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        return expr_first+expr_last\n\n\n    # Generate maximum aggregate expressions for columns of other types\n    @staticmethod\n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_mean = [pl.mean(col).alias(f\"mean_{col}\") for col in cols]\n        return expr_max+expr_min+expr_mean\n\n\n    # Generate the maximum aggregate expression for a specific column \"num_group\"\n    @staticmethod\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        return expr_max\n\n\n    # Get all types of aggregate expressions\n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n        return exprs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T18:28:01.846622Z","iopub.execute_input":"2026-05-11T18:28:01.846892Z","iopub.status.idle":"2026-05-11T18:28:01.865922Z","shell.execute_reply.started":"2026-05-11T18:28:01.846869Z","shell.execute_reply":"2026-05-11T18:28:01.865163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Read a single file and preprocess it\ndef read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(DataPrep.set_table_dtypes)\n    # If the depth parameter is 1 or 2, the data is aggregated by \"case_id\"\n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T18:28:01.867003Z","iopub.execute_input":"2026-05-11T18:28:01.867436Z","iopub.status.idle":"2026-05-11T18:28:01.885111Z","shell.execute_reply.started":"2026-05-11T18:28:01.867399Z","shell.execute_reply":"2026-05-11T18:28:01.884458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Read multiple files and preprocess them\ndef read_files(regex_path, depth=None):\n    chunks = []\n    \n    for path in glob(str(regex_path)):\n        df = pl.read_parquet(path)\n        df = df.pipe(DataPrep.set_table_dtypes)\n        if depth in [1, 2]:\n            df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        chunks.append(df)\n    \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T18:28:01.886104Z","iopub.execute_input":"2026-05-11T18:28:01.886442Z","iopub.status.idle":"2026-05-11T18:28:01.902007Z","shell.execute_reply.started":"2026-05-11T18:28:01.886409Z","shell.execute_reply":"2026-05-11T18:28:01.901255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature engineering function for adding new features and merging dataframes\ndef data_preprocessing(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n    df_base = df_base.pipe(DataPrep.handle_dates)\n    return df_base","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T18:28:01.903077Z","iopub.execute_input":"2026-05-11T18:28:01.903505Z","iopub.status.idle":"2026-05-11T18:28:01.917880Z","shell.execute_reply.started":"2026-05-11T18:28:01.903469Z","shell.execute_reply":"2026-05-11T18:28:01.917019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    return df_data, cat_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T18:28:01.918887Z","iopub.execute_input":"2026-05-11T18:28:01.919099Z","iopub.status.idle":"2026-05-11T18:28:01.926157Z","shell.execute_reply.started":"2026-05-11T18:28:01.919079Z","shell.execute_reply":"2026-05-11T18:28:01.925397Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load Data\n","metadata":{}},{"cell_type":"code","source":"# Robust Kaggle input discovery. Imported notebooks may mount competition data under /kaggle/input/competitions.\ndef find_competition_root():\n    candidates = [\n        Path(\"/kaggle/input/home-credit-credit-risk-model-stability\"),\n        Path(\"/kaggle/input/competitions/home-credit-credit-risk-model-stability\"),\n    ]\n    for root in candidates:\n        if (root / \"parquet_files\" / \"train\").exists() and (root / \"parquet_files\" / \"test\").exists():\n            return root\n\n    for parquet_dir in Path(\"/kaggle/input\").rglob(\"parquet_files\"):\n        root = parquet_dir.parent\n        if (parquet_dir / \"train\").exists() and (parquet_dir / \"test\").exists():\n            return root\n\n    raise FileNotFoundError(\"Could not find Home Credit competition parquet_files under /kaggle/input\")\n\nROOT = find_competition_root()\nTRAIN_DIR = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR = ROOT / \"parquet_files\" / \"test\"\n\nprint(\"ROOT:\", ROOT)\nprint(\"TRAIN_DIR exists:\", TRAIN_DIR.exists())\nprint(\"TEST_DIR exists:\", TEST_DIR.exists())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T18:28:01.928623Z","iopub.execute_input":"2026-05-11T18:28:01.928872Z","iopub.status.idle":"2026-05-11T18:28:01.946535Z","shell.execute_reply.started":"2026-05-11T18:28:01.928850Z","shell.execute_reply":"2026-05-11T18:28:01.945690Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Read all training data sets into a variable\ntraining_data_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_1_*.parquet\",1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR / \"train_applprev_2.parquet\", 2),\n        read_file(TRAIN_DIR / \"train_person_2.parquet\", 2),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n    ]\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T18:28:01.947525Z","iopub.execute_input":"2026-05-11T18:28:01.947862Z","execution_failed":"2026-05-11T18:28:10.497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load and Pre-process all training datasets\ndf_train = data_preprocessing(**training_data_store)\nprint(\"train data shape:\\t\", df_train.shape)","metadata":{"trusted":true,"execution":{"execution_failed":"2026-05-11T18:28:10.498Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Read all test data sets into a variable\ntest_data_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n        read_files(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\",1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_applprev_2.parquet\", 2),\n        read_file(TEST_DIR / \"test_person_2.parquet\", 2),\n        read_files(TEST_DIR / \"test_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n    ]\n}","metadata":{"trusted":true,"execution":{"execution_failed":"2026-05-11T18:28:10.498Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load and Pre-process all test datasets\ndf_test = data_preprocessing(**test_data_store)\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"trusted":true,"execution":{"execution_failed":"2026-05-11T18:28:10.498Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Remove useless columns\ndf_train = df_train.pipe(DataPrep.filter_cols)\ndf_test = df_test.select([col for col in df_train.columns if col != \"target\"])\n\nprint(\"train data shape:\\t\", df_train.shape)\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"trusted":true,"execution":{"execution_failed":"2026-05-11T18:28:10.498Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert to Pandas\ndf_train, cat_cols = to_pandas(df_train)\ndf_test, cat_cols = to_pandas(df_test, cat_cols)","metadata":{"trusted":true,"execution":{"execution_failed":"2026-05-11T18:28:10.498Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# CatBoost Training\n","metadata":{}},{"cell_type":"code","source":"# ---------------- Data Preparation ----------------\ndf_train = reduce_mem_usage(df_train)\nX = df_train.drop(columns=[\"target\", \"case_id\", \"WEEK_NUM\"])\ny = df_train[\"target\"]\nweeks = df_train[\"WEEK_NUM\"]\n\n# ---------------- Fast CatBoost settings ----------------\n# Run full 5-fold validation for CatBoost.\nCAT_MAX_FOLDS = 5\n\n# ---------------- CatBoost frame preparation ----------------\ncat_features = [\n    col for col in X.columns\n    if str(X[col].dtype) in (\"category\", \"object\", \"string\")\n]\n\nprint(f\"CatBoost categorical features: {len(cat_features)}\")\n\n\ndef prepare_catboost_frame(df, cat_features):\n    # Avoid expensive astype(\"string\").astype(str) on million-row columns.\n    # Keep pandas category dtype where possible and only add/fill a missing category.\n    df_cb = df.copy(deep=False)\n    for col in cat_features:\n        if str(df_cb[col].dtype) == \"category\":\n            if \"__MISSING__\" not in df_cb[col].cat.categories:\n                df_cb[col] = df_cb[col].cat.add_categories([\"__MISSING__\"])\n            df_cb[col] = df_cb[col].fillna(\"__MISSING__\")\n        else:\n            df_cb[col] = df_cb[col].fillna(\"__MISSING__\")\n    return df_cb\n\n\nX_cb = prepare_catboost_frame(X, cat_features)\n\ncv = StratifiedGroupKFold(n_splits=5, shuffle=False)\n\ncat_params = {\n    \"eval_metric\": \"AUC\",\n    \"iterations\": 1500,\n    \"learning_rate\": 0.05,\n    #\"depth\": 6,\n    #\"l2_leaf_reg\": 10,\n    #\"random_strength\": 0.5,\n    #\"border_count\": 64,\n    \"od_type\": \"Iter\",\n    \"od_wait\": 50,\n    #\"random_seed\": 42,\n    \"task_type\": \"GPU\",\n    \"devices\": \"0\",\n    \"allow_writing_files\": False,\n    \"verbose\": 100,\n}\n\ncat_fitted_models = []\ncat_cv_scores = []\n\nfor fold, (idx_train, idx_valid) in enumerate(cv.split(X_cb, y, groups=weeks), 1):\n    if fold > CAT_MAX_FOLDS:\n        break\n\n    X_train, y_train = X_cb.iloc[idx_train], y.iloc[idx_train]\n    X_valid, y_valid = X_cb.iloc[idx_valid], y.iloc[idx_valid]\n\n    print(f\"CatBoost fold {fold} validation week range: \", (weeks.iloc[idx_valid].min(), weeks.iloc[idx_valid].max()))\n\n    train_pool = Pool(X_train, y_train, cat_features=cat_features)\n    valid_pool = Pool(X_valid, y_valid, cat_features=cat_features)\n\n    cat_model = CatBoostClassifier(**cat_params)\n    cat_model.fit(train_pool, eval_set=valid_pool, use_best_model=True)\n\n    cat_fitted_models.append(cat_model)\n\n    y_pred_valid = cat_model.predict_proba(valid_pool)[:, 1]\n    auc_score = roc_auc_score(y_valid, y_pred_valid)\n    cat_cv_scores.append(auc_score)\n    print(f\"CatBoost fold {fold} AUC: {auc_score:.6f}\")\n\nprint(\"CatBoost CV AUC scores: \", cat_cv_scores)\nprint(\"CatBoost Average CV AUC score: \", sum(cat_cv_scores) / len(cat_cv_scores))","metadata":{"trusted":true,"execution":{"execution_failed":"2026-05-11T18:28:10.498Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# CatBoost Prediction\n","metadata":{}},{"cell_type":"code","source":"# ---------------- Predict test with CatBoost ensemble ----------------\nX_test = df_test.drop(columns=[\"WEEK_NUM\"])\nX_test = X_test.set_index(\"case_id\")\nX_test_cb = prepare_catboost_frame(X_test, cat_features)\n\ntest_pool = Pool(X_test_cb, cat_features=cat_features)\ncat_pred = pd.Series(\n    np.mean([model.predict_proba(test_pool)[:, 1] for model in cat_fitted_models], axis=0),\n    index=X_test.index,\n    name=\"score\",\n)\n\nprint(\"Check CatBoost prediction null:\", cat_pred.isnull().any())\nprint(cat_pred.head())\n","metadata":{"trusted":true,"execution":{"execution_failed":"2026-05-11T18:28:10.498Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission\n","metadata":{}},{"cell_type":"code","source":"# Prepare CatBoost-only submission\n# This creates a sample-test submission in normal notebook runs.\n# For code competition hidden tests, use an inference-only notebook if training is too slow.\ndf_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\ndf_subm[\"score\"] = cat_pred.reindex(df_subm.index)\n\nprint(\"Check null:\", df_subm[\"score\"].isnull().any())\ndisplay(df_subm.head())\n\ndf_subm.to_csv(\"submission.csv\")\nprint(\"Saved submission.csv\")\n","metadata":{"trusted":true,"execution":{"execution_failed":"2026-05-11T18:28:10.498Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Save Models\n","metadata":{}},{"cell_type":"code","source":"# Save CatBoost models and metadata\nMODEL_OUTPUT_DIR = Path(\"/kaggle/working/models\")\nMODEL_OUTPUT_DIR.mkdir(parents=True, exist_ok=True)\n\nfor i, model in enumerate(cat_fitted_models):\n    model.save_model(MODEL_OUTPUT_DIR / f\"catboost_model_fold_{i}.cbm\")\n\ncat_metadata = {\n    \"feature_names\": list(X.columns),\n    \"cat_features\": cat_features,\n    \"cat_cv_scores\": cat_cv_scores,\n    \"cat_params\": cat_params,\n    \"cat_max_folds\": CAT_MAX_FOLDS,\n}\njoblib.dump(cat_metadata, MODEL_OUTPUT_DIR / \"cat_model_metadata.pkl\")\n\nprint(f\"Saved {len(cat_fitted_models)} CatBoost models to {MODEL_OUTPUT_DIR}\")\nprint(sorted(p.name for p in MODEL_OUTPUT_DIR.iterdir()))\n","metadata":{"trusted":true,"execution":{"execution_failed":"2026-05-11T18:28:10.498Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}