{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":50160,"databundleVersionId":7921029}],"dockerImageVersionId":31090,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\nimport os\nimport gc\n\nimport re\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import TimeSeriesSplit, GroupKFold, StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.metrics import roc_auc_score\n\nimport lightgbm as lgb","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:36:46.332974Z","iopub.execute_input":"2025-10-20T13:36:46.333423Z","iopub.status.idle":"2025-10-20T13:36:46.338374Z","shell.execute_reply.started":"2025-10-20T13:36:46.3334Z","shell.execute_reply":"2025-10-20T13:36:46.337449Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Prepare Functions for Data Preprocessing and Table Aggregation","metadata":{}},{"cell_type":"code","source":"class DataPrep:\n\n    # Special columns:\n    # case_id - This is the unique identifier for each credit case. You'll need this ID to join relevant tables to the base table.\n    # date_decision - This refers to the date when a decision was made regarding the approval of the loan.\n    # WEEK_NUM - This is the week number used for aggregation. In the test sample, WEEK_NUM continues sequentially from the last training value of WEEK_NUM.\n    # MONTH - This column represents the month and is intended for aggregation purposes.\n    # target - This is the target value, determined after a certain period based on whether or not the client defaulted on the specific credit case (loan).\n    # num_group1 - This is an indexing column used for the historical records of case_id in both depth=1 and depth=2 tables.\n    # num_group2 - This is the second indexing column for depth=2 tables' historical records of case_id. The order of num_group1 and num_group2 is important and will be clarified in feature definitions.\n    # All other raw columns in the tables serve as predictors. Their definitions can be found in the file feature_definitions.csv. For depth=0 tables, predictors can be directly used as features. However, for tables with depth>0, you may need to employ aggregation functions that will condense the historical records associated with each case_id into a single feature. In case num_group1 or num_group2 stands for person index (this is clear with predictor definitions) the zero index has special meaning. When num_groupN=0 it is the applicant (the person who applied for a loan).\n    \n    # Various predictors were transformed, therefore we have the following notation for similar groups of transformations\n    # P - Transform DPD (Days past due)\n    # M - Masking categories\n    # A - Transform amount\n    # D - Transform date\n    # T - Unspecified Transform\n    # L - Unspecified Transform\n    \n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n        return df\n\n    # Handle dates\n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n\n    # Filter columns\n    # If the column name is not in the reserved list and the null value ratio of the column is greater than 0.95, then delete the column.\n    # If the column name is not in the reserved list, the column data type is String, and the number of unique values is 1 or greater than 200, then delete the column.\n    @staticmethod\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.95:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n\n        return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:36:51.20628Z","iopub.execute_input":"2025-10-20T13:36:51.206542Z","iopub.status.idle":"2025-10-20T13:36:51.215631Z","shell.execute_reply.started":"2025-10-20T13:36:51.206525Z","shell.execute_reply":"2025-10-20T13:36:51.214888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Aggregator:\n    # Generate maximum aggregate expression for numeric columns\n    @staticmethod\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        return expr_max\n\n    # Generate maximum aggregate expression for date type columns\n    @staticmethod\n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\",)]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        return expr_max\n\n    # Generate a maximum aggregate expression for a column of type string\n    @staticmethod\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        return expr_max\n\n    # Generate maximum aggregate expressions for columns of other types\n    @staticmethod\n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        return expr_max\n\n    # Generate the maximum aggregate expression for a specific column \"num_group\"\n    @staticmethod\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        return expr_max\n\n    # Get all types of aggregate expressions\n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n        return exprs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:36:56.182328Z","iopub.execute_input":"2025-10-20T13:36:56.18288Z","iopub.status.idle":"2025-10-20T13:36:56.190291Z","shell.execute_reply.started":"2025-10-20T13:36:56.182832Z","shell.execute_reply":"2025-10-20T13:36:56.189382Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Read a single file and preprocess it\ndef read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(DataPrep.set_table_dtypes)\n    # If the depth parameter is 1 or 2, the data is aggregated by \"case_id\"\n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:36:59.313459Z","iopub.execute_input":"2025-10-20T13:36:59.313718Z","iopub.status.idle":"2025-10-20T13:36:59.317971Z","shell.execute_reply.started":"2025-10-20T13:36:59.313697Z","shell.execute_reply":"2025-10-20T13:36:59.317251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Read multiple files and preprocess them\ndef read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob(str(regex_path)):\n        chunks.append(pl.read_parquet(path).pipe(DataPrep.set_table_dtypes))\n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    # If the depth parameter is 1 or 2, the data is aggregated by \"case_id\"\n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:37:00.622296Z","iopub.execute_input":"2025-10-20T13:37:00.622578Z","iopub.status.idle":"2025-10-20T13:37:00.627189Z","shell.execute_reply.started":"2025-10-20T13:37:00.622557Z","shell.execute_reply":"2025-10-20T13:37:00.626413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature engineering function for adding new features and merging dataframes\ndef data_preprocessing(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n    df_base = df_base.pipe(DataPrep.handle_dates)\n    return df_base","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:37:01.846131Z","iopub.execute_input":"2025-10-20T13:37:01.846589Z","iopub.status.idle":"2025-10-20T13:37:01.851189Z","shell.execute_reply.started":"2025-10-20T13:37:01.846566Z","shell.execute_reply":"2025-10-20T13:37:01.850345Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    return df_data, cat_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:37:03.679738Z","iopub.execute_input":"2025-10-20T13:37:03.680027Z","iopub.status.idle":"2025-10-20T13:37:03.684178Z","shell.execute_reply.started":"2025-10-20T13:37:03.680009Z","shell.execute_reply":"2025-10-20T13:37:03.683363Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load Data Sets","metadata":{}},{"cell_type":"code","source":"ROOT = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\n\nTRAIN_DIR = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR = ROOT / \"parquet_files\" / \"test\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:37:05.815151Z","iopub.execute_input":"2025-10-20T13:37:05.815388Z","iopub.status.idle":"2025-10-20T13:37:05.819076Z","shell.execute_reply.started":"2025-10-20T13:37:05.815372Z","shell.execute_reply":"2025-10-20T13:37:05.818337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df = pl.read_parquet(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\")\n# df.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:37:06.935078Z","iopub.execute_input":"2025-10-20T13:37:06.935622Z","iopub.status.idle":"2025-10-20T13:37:06.938744Z","shell.execute_reply.started":"2025-10-20T13:37:06.935601Z","shell.execute_reply":"2025-10-20T13:37:06.9379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Read all training data sets into a variable\ntraining_data_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n        # read_file(TRAIN_DIR / \"train_credit_bureau_a_1_*.parquet\",1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR / \"train_applprev_2.parquet\", 2),\n        read_file(TRAIN_DIR / \"train_person_2.parquet\", 2),\n        # read_file(TRAIN_DIR / \"train_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n    ]\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:37:07.915108Z","iopub.execute_input":"2025-10-20T13:37:07.915375Z","iopub.status.idle":"2025-10-20T13:37:28.006123Z","shell.execute_reply.started":"2025-10-20T13:37:07.915356Z","shell.execute_reply":"2025-10-20T13:37:28.005529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load and Pre-process all training datasets\ndf_train = data_preprocessing(**training_data_store)\nprint(\"train data shape:\\t\", df_train.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:37:28.007281Z","iopub.execute_input":"2025-10-20T13:37:28.007527Z","iopub.status.idle":"2025-10-20T13:37:35.373156Z","shell.execute_reply.started":"2025-10-20T13:37:28.007502Z","shell.execute_reply":"2025-10-20T13:37:35.372319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Read all test data sets into a variable\ntest_data_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n        # read_file(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\",1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_applprev_2.parquet\", 2),\n        read_file(TEST_DIR / \"test_person_2.parquet\", 2),\n        # read_file(TEST_DIR / \"test_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n    ]\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:37:35.374077Z","iopub.execute_input":"2025-10-20T13:37:35.374281Z","iopub.status.idle":"2025-10-20T13:37:35.481964Z","shell.execute_reply.started":"2025-10-20T13:37:35.374265Z","shell.execute_reply":"2025-10-20T13:37:35.48142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load and Pre-process all test datasets\ndf_test = data_preprocessing(**test_data_store)\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:37:35.483257Z","iopub.execute_input":"2025-10-20T13:37:35.483454Z","iopub.status.idle":"2025-10-20T13:37:35.512599Z","shell.execute_reply.started":"2025-10-20T13:37:35.483438Z","shell.execute_reply":"2025-10-20T13:37:35.512003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Remove useless columns\ndf_train = df_train.pipe(DataPrep.filter_cols)\ndf_test = df_test.select([col for col in df_train.columns if col != \"target\"])\n\nprint(\"train data shape:\\t\", df_train.shape)\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:37:35.513258Z","iopub.execute_input":"2025-10-20T13:37:35.513491Z","iopub.status.idle":"2025-10-20T13:37:37.930367Z","shell.execute_reply.started":"2025-10-20T13:37:35.513468Z","shell.execute_reply":"2025-10-20T13:37:37.929652Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert to Pandas\ndf_train, cat_cols = to_pandas(df_train)\ndf_test, cat_cols = to_pandas(df_test, cat_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:37:37.931149Z","iopub.execute_input":"2025-10-20T13:37:37.931724Z","iopub.status.idle":"2025-10-20T13:37:48.926438Z","shell.execute_reply.started":"2025-10-20T13:37:37.931697Z","shell.execute_reply":"2025-10-20T13:37:48.925882Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.head(20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:37:48.927132Z","iopub.execute_input":"2025-10-20T13:37:48.927318Z","iopub.status.idle":"2025-10-20T13:37:48.956079Z","shell.execute_reply.started":"2025-10-20T13:37:48.927303Z","shell.execute_reply":"2025-10-20T13:37:48.955165Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA and Feature Engineering","metadata":{}},{"cell_type":"code","source":"df_feature_definition = pd.read_csv(ROOT / \"feature_definitions.csv\")\ndf_feature_definition.head(20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T14:11:47.387602Z","iopub.execute_input":"2025-10-07T14:11:47.387814Z","iopub.status.idle":"2025-10-07T14:11:47.407373Z","shell.execute_reply.started":"2025-10-07T14:11:47.387798Z","shell.execute_reply":"2025-10-07T14:11:47.406894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Training Data Shape & Column Suffix Counts\n\ndef cols_by_suffix(df, sfx):\n    return [c for c in df.columns if c.endswith(sfx)]\n    \nprint(\"Training Data Shape (rows, cols):\", df_train.shape)\nfor s in [\"P\",\"A\",\"M\",\"D\",\"T\",\"L\"]:\n    print(f\"#{s}-suffix columns:\", len(cols_by_suffix(df_train, s)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T14:11:47.407956Z","iopub.execute_input":"2025-10-07T14:11:47.408389Z","iopub.status.idle":"2025-10-07T14:11:47.413368Z","shell.execute_reply.started":"2025-10-07T14:11:47.408362Z","shell.execute_reply":"2025-10-07T14:11:47.412773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Target Distribution\nif \"target\" in df_train.columns:\n    n = len(df_train)\n    n_pos = df_train[\"target\"].sum()\n    pos_rate = 100.0 * df_train[\"target\"].mean()\n    print(f\"\\nThere are {n:,} total number of samples.\\nThere are {int(n_pos):,} positive number of samples.\\nThe positive percentage is {pos_rate:.2f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T14:11:47.414279Z","iopub.execute_input":"2025-10-07T14:11:47.414805Z","iopub.status.idle":"2025-10-07T14:11:47.449935Z","shell.execute_reply.started":"2025-10-07T14:11:47.414781Z","shell.execute_reply":"2025-10-07T14:11:47.449264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Top Features with Missing Values\nmissing = (df_train.isna().mean().sort_values(ascending=False) * 100).round(2)\nprint(\"\\nTop-15 missingness (%):\")\ndisplay(missing.head(15).to_frame(\"missing_%\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T14:11:47.450794Z","iopub.execute_input":"2025-10-07T14:11:47.451093Z","iopub.status.idle":"2025-10-07T14:11:48.741382Z","shell.execute_reply.started":"2025-10-07T14:11:47.45107Z","shell.execute_reply":"2025-10-07T14:11:48.740795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Quick numeric summaries on 3 least-missing numeric columns\nnum_cols = df_train.select_dtypes(include=[np.number]).columns.tolist()\nif num_cols:\n    miss_num = df_train[num_cols].isna().mean()\n    chosen = miss_num.sort_values().head(3).index.tolist()\n    desc = pd.concat({\n        c: pd.Series({\n            \"mean\": df_train[c].mean(),\n            \"std\": df_train[c].std(),\n            \"p01\": df_train[c].quantile(0.01),\n            \"p50\": df_train[c].quantile(0.50),\n            \"p99\": df_train[c].quantile(0.99),\n            \"missing_%\": df_train[c].isna().mean() * 100,\n        }) for c in chosen\n    }, axis=1).T\n    print(\"\\nNumeric summaries (sampled columns):\")\n    display(desc)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T14:11:48.742181Z","iopub.execute_input":"2025-10-07T14:11:48.742478Z","iopub.status.idle":"2025-10-07T14:11:51.944513Z","shell.execute_reply.started":"2025-10-07T14:11:48.742459Z","shell.execute_reply":"2025-10-07T14:11:51.943639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" # Draw Histograms For Each Chosen Feature\nplot_sample = df_train[chosen].sample(n=min(len(df_train), 200_000), random_state=42)\nfor c in chosen:\n    plt.figure()\n    plot_sample[c].dropna().hist(bins=50)\n    plt.title(f\"Histogram: {c}\")\n    plt.xlabel(c); plt.ylabel(\"count\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T14:11:51.945443Z","iopub.execute_input":"2025-10-07T14:11:51.946043Z","iopub.status.idle":"2025-10-07T14:11:52.735853Z","shell.execute_reply.started":"2025-10-07T14:11:51.946019Z","shell.execute_reply":"2025-10-07T14:11:52.734997Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train Models","metadata":{}},{"cell_type":"code","source":"# # ================================================================\n# # MODELLING: LR, RF, LightGBM with robust preprocessing\n# # Metrics: AUC, Gini, Gini Stability (OOF); Pick best and submit\n# # ================================================================\n# import os, gc, numpy as np, pandas as pd\n# from pandas.api.types import is_numeric_dtype, is_bool_dtype, is_categorical_dtype\n\n# from sklearn import set_config\n# from sklearn.model_selection import StratifiedKFold, StratifiedShuffleSplit, train_test_split\n# from sklearn.metrics import roc_auc_score\n# from sklearn.compose import ColumnTransformer\n# from sklearn.preprocessing import OneHotEncoder, OrdinalEncoder, MaxAbsScaler, FunctionTransformer\n# from sklearn.impute import SimpleImputer\n# from sklearn.pipeline import Pipeline, make_pipeline\n# from sklearn.linear_model import LogisticRegression\n# from sklearn.ensemble import RandomForestClassifier\n# from sklearn.base import clone\n# from scipy import sparse\n# from sklearn.preprocessing import StandardScaler\n\n# from lightgbm import LGBMClassifier\n# import lightgbm as lgb\n\n# # ---------------- thread caps to reduce RAM spikes ----------------\n# os.environ[\"OMP_NUM_THREADS\"] = \"2\"\n# os.environ[\"OPENBLAS_NUM_THREADS\"] = \"2\"\n# os.environ[\"MKL_NUM_THREADS\"] = \"2\"\n# os.environ[\"NUMEXPR_NUM_THREADS\"] = \"2\"\n# os.environ[\"VECLIB_MAXIMUM_THREADS\"] = \"2\"\n\n# # Ensure sklearn transformers return NumPy arrays by default (not pandas)\n# set_config(transform_output=\"default\")\n\n# # --------------------- competition stability metric ---------------------\n# def gini_stability(base, score_col=\"score\", w_fallingrate=88.0, w_resstd=-0.5):\n#     gini_in_time = (\n#         base.loc[:, [\"WEEK_NUM\", \"target\", score_col]]\n#             .sort_values(\"WEEK_NUM\")\n#             .groupby(\"WEEK_NUM\")[[\"target\", score_col]]\n#             .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[score_col]) - 1)\n#             .tolist()\n#     )\n#     x = np.arange(len(gini_in_time), dtype=float)\n#     y = np.array(gini_in_time, dtype=float)\n#     a, b = np.polyfit(x, y, 1)                    # trend\n#     res_std = np.std(y - (a*x + b))               # volatility\n#     return y.mean() + w_fallingrate*min(0.0, a) + w_resstd*res_std\n\n# # --------------------- Optional quick subsample for speed ---------------------\n# FAST_SAMPLE = True\n# SAMPLE_ROWS = 450_000\n\n# df = df_train.replace([np.inf, -np.inf], np.nan).copy()\n# if FAST_SAMPLE and len(df) > SAMPLE_ROWS:\n#     idx = np.arange(len(df))\n#     _, idx_small = train_test_split(idx, train_size=SAMPLE_ROWS,\n#                                     stratify=df[\"target\"], random_state=42)\n#     df = df.iloc[idx_small].copy()\n\n# TARGET_COL = \"target\"\n# WEEK_COL   = \"WEEK_NUM\"\n# drop_cols  = {TARGET_COL, WEEK_COL, \"case_id\"}\n\n# # --------------------- Robust feature split & dtype hygiene ---------------------\n# X = df.drop(columns=list(drop_cols)).copy()\n# y = df[TARGET_COL].astype(int)\n# week_num = df[WEEK_COL].values\n\n# # initial numeric guess by dtype (exclude bools)\n# num_cols = [c for c in X.columns if is_numeric_dtype(X[c]) and not is_bool_dtype(X[c])]\n# cat_cols_all = [c for c in X.columns if c not in num_cols]\n\n# # demote any \"numeric\" columns that actually contain non-numeric tokens\n# bad_num = []\n# for c in list(num_cols):\n#     coerced = pd.to_numeric(X[c], errors=\"coerce\")\n#     frac_bad = ((~X[c].isna()) & (coerced.isna())).mean()\n#     if frac_bad > 0.0:\n#         bad_num.append(c)\n# for c in bad_num:\n#     num_cols.remove(c)\n#     if c not in cat_cols_all:\n#         cat_cols_all.append(c)\n\n# # set categoricals to category dtype; downcast numerics for memory\n# for c in cat_cols_all:\n#     if not is_categorical_dtype(X[c]):\n#         X[c] = X[c].astype(\"category\")\n# for c in num_cols:\n#     if X[c].dtype == \"float64\":\n#         X[c] = X[c].astype(\"float32\")\n#     elif X[c].dtype == \"int64\":\n#         X[c] = X[c].astype(\"int32\")\n\n# # low-card subset for LR one-hot (keeps LR sparse matrix small)\n# low_card_cats = [c for c in cat_cols_all if X[c].nunique(dropna=True) <= 30]\n\n# # helpers: keep matrices memory-friendly\n# to_csr32 = FunctionTransformer(\n#     lambda A: sparse.csr_matrix(A, dtype=np.float32) if not sparse.issparse(A) else A.astype(np.float32),\n#     accept_sparse=True\n# )\n# to_numpy32 = FunctionTransformer(\n#     lambda A: (A.to_numpy(dtype=np.float32) if isinstance(A, pd.DataFrame) else A.astype(np.float32, copy=False)),\n#     accept_sparse=False\n# )\n\n# # helper: convert any DataFrame/ndarray to a numpy object array of strings\n# cat_to_object = FunctionTransformer(\n#     lambda A: (\n#         A.astype(str).to_numpy(object) if isinstance(A, pd.DataFrame)\n#         else np.asarray(A).astype(str)\n#     ),\n#     accept_sparse=False, feature_names_out=\"one-to-one\"\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:34:58.0813Z","iopub.execute_input":"2025-10-20T13:34:58.082207Z","iopub.status.idle":"2025-10-20T13:35:12.098324Z","shell.execute_reply.started":"2025-10-20T13:34:58.082185Z","shell.execute_reply":"2025-10-20T13:35:12.097422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # =============================== 1) LOGISTIC REGRESSION ===============================\n# to_csr_strict = FunctionTransformer(\n#     lambda A: (A.tocsr().astype(np.float32) if sparse.issparse(A)\n#                else sparse.csr_matrix(A, dtype=np.float32)),\n#     accept_sparse=True\n# )\n\n# pre_lr = ColumnTransformer(\n#     transformers=[\n#         # Numeric: impute -> force CSR -> Standardize (sparse-safe with_mean=False)\n#         (\"num\", make_pipeline(SimpleImputer(strategy=\"median\"),\n#                               to_csr_strict,\n#                               StandardScaler(with_mean=False)),\n#          num_cols),\n\n#         # Categorical (low-card only for LR): OHE sparse\n#         (\"cat_low\", OneHotEncoder(handle_unknown=\"ignore\", sparse=True), low_card_cats),\n#     ],\n#     remainder=\"drop\",\n#     sparse_threshold=1.0   # keep the combined matrix sparse\n# )\n\n# lr_pipe = Pipeline([\n#     (\"prep\", pre_lr),\n#     (\"clf\", LogisticRegression(\n#         solver=\"saga\", C=1.0, tol=1e-3, max_iter=150,\n#         n_jobs=-1, random_state=42\n#     )),\n# ])\n\n# # =============================== 2) RANDOM FOREST ===============================\n# pre_tree = ColumnTransformer(\n#     transformers=[\n#         (\"num\", SimpleImputer(strategy=\"median\"), num_cols),\n#         (\"cat\", make_pipeline(\n#             cat_to_object,  # <--- NEW: force string/object dtype so imputer won't try to float-cast\n#             SimpleImputer(strategy=\"most_frequent\"),\n#             OrdinalEncoder(handle_unknown=\"use_encoded_value\", unknown_value=-1)\n#         ), cat_cols_all),\n#     ],\n#     remainder=\"drop\",\n#     n_jobs=1\n# )\n\n# rf_pipe = Pipeline([\n#     (\"prep\", pre_tree),\n#     (\"to32\", to_numpy32),\n#     (\"clf\", RandomForestClassifier(\n#         n_estimators=160, max_depth=12, max_features=\"sqrt\",\n#         max_samples=0.6, bootstrap=True,\n#         n_jobs=2, random_state=42, class_weight=\"balanced_subsample\"\n#     )),\n# ])\n\n# # =============================== 3) LIGHTGBM ===============================\n# try:\n#     lgb_est = LGBMClassifier(\n#         n_estimators=600, learning_rate=0.06, num_leaves=64,\n#         max_depth=-1, subsample=0.8, colsample_bytree=0.8,\n#         reg_lambda=1.0, n_jobs=2, objective=\"binary\", metric=\"auc\",\n#         device_type=\"gpu\"   # falls back to CPU if no GPU\n#     )\n# except TypeError:\n#     lgb_est = LGBMClassifier(\n#         n_estimators=600, learning_rate=0.06, num_leaves=64,\n#         max_depth=-1, subsample=0.8, colsample_bytree=0.8,\n#         reg_lambda=1.0, n_jobs=2, objective=\"binary\", metric=\"auc\",\n#         device=\"gpu\"\n#     )\n\n# lgb_pipe = Pipeline([\n#     (\"prep\", pre_tree),\n#     (\"to32\", to_numpy32),\n#     (\"clf\", lgb_est),\n# ])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T15:08:35.206843Z","iopub.execute_input":"2025-10-07T15:08:35.207193Z","iopub.status.idle":"2025-10-07T15:08:35.21608Z","shell.execute_reply.started":"2025-10-07T15:08:35.207169Z","shell.execute_reply":"2025-10-07T15:08:35.215188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # =============================== CV runner (OOF + GS) ===============================\n# def cv_oof_report(name, pipeline, X, y, week_num, n_splits=3, random_state=42):\n#     skf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=random_state)\n#     oof = np.zeros(len(y), dtype=np.float32)\n\n#     for fold, (tr_idx, va_idx) in enumerate(skf.split(X, y), 1):\n#         X_tr, X_va = X.iloc[tr_idx], X.iloc[va_idx]\n#         y_tr, y_va = y.iloc[tr_idx], y.iloc[va_idx]\n\n#         model = clone(pipeline)\n#         fit_params = {}\n#         if name.lower().startswith(\"lgb\"):\n#             fit_params = {\n#                 \"clf__eval_set\": [(X_va, y_va)],\n#                 # Use callbacks for early stopping & silence logs\n#                 \"clf__callbacks\": [lgb.early_stopping(50, verbose=False),\n#                                    lgb.log_evaluation(period=-1)]\n#             }\n\n#         model.fit(X_tr, y_tr, **fit_params)\n#         oof[va_idx] = model.predict_proba(X_va)[:, 1].astype(np.float32)\n\n#         print(f\"[{name}] Fold {fold}/{n_splits} AUC={roc_auc_score(y_va, oof[va_idx]):.5f}\")\n\n#         del X_tr, X_va, y_tr, y_va, model\n#         gc.collect()\n\n#     auc  = roc_auc_score(y, oof)\n#     gini = 2*auc - 1\n#     base = pd.DataFrame({WEEK_COL: week_num, \"target\": y.values, \"score\": oof})\n#     gs   = gini_stability(base)\n\n#     print(f\"\\n[{name}] OOF AUC={auc:.5f} | Gini={gini:.5f} | GiniStability={gs:.5f}\\n\")\n#     return {\"model\": name, \"oof_auc\": auc, \"oof_gini\": gini, \"gini_stability\": gs, \"oof\": oof, \"pipe\": pipeline}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T15:34:41.826744Z","iopub.execute_input":"2025-10-07T15:34:41.827367Z","iopub.status.idle":"2025-10-07T15:34:41.834318Z","shell.execute_reply.started":"2025-10-07T15:34:41.827341Z","shell.execute_reply":"2025-10-07T15:34:41.833587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # =============================== Run CV for all models ===============================\n# out_lr  = cv_oof_report(\"LogReg_saga\",  lr_pipe, X, y, week_num)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T14:38:45.220964Z","iopub.execute_input":"2025-10-07T14:38:45.22123Z","iopub.status.idle":"2025-10-07T14:57:23.101942Z","shell.execute_reply.started":"2025-10-07T14:38:45.221211Z","shell.execute_reply":"2025-10-07T14:57:23.101251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# out_rf  = cv_oof_report(\"RandomForest\", rf_pipe, X, y, week_num)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T15:08:41.340318Z","iopub.execute_input":"2025-10-07T15:08:41.340561Z","iopub.status.idle":"2025-10-07T15:25:11.900999Z","shell.execute_reply.started":"2025-10-07T15:08:41.340546Z","shell.execute_reply":"2025-10-07T15:25:11.900357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# out_lgb = cv_oof_report(\"LGB_fast\",     lgb_pipe, X, y, week_num)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T15:35:52.722181Z","iopub.execute_input":"2025-10-07T15:35:52.722459Z","iopub.status.idle":"2025-10-07T15:40:00.686658Z","shell.execute_reply.started":"2025-10-07T15:35:52.722438Z","shell.execute_reply":"2025-10-07T15:40:00.685714Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# summary = pd.DataFrame([out_lr, out_rf, out_lgb]).drop(columns=[\"oof\", \"pipe\"]).sort_values(\"oof_auc\", ascending=False)\n# print(\"=== CV Summary ===\")\n# print(summary.to_string(index=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T15:40:30.478893Z","iopub.execute_input":"2025-10-07T15:40:30.479526Z","iopub.status.idle":"2025-10-07T15:40:30.487856Z","shell.execute_reply.started":"2025-10-07T15:40:30.479501Z","shell.execute_reply":"2025-10-07T15:40:30.48718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # =============================== Pick best, refit, predict test, submit ===============================\n# SELECT_BY = \"gini_stability\"   # or \"oof_auc\"\n# best = max([out_lr, out_rf, out_lgb], key=lambda d: d[SELECT_BY])\n# print(f\"[BEST MODEL] {best['model']} by {SELECT_BY}: {best[SELECT_BY]:.5f}\")\n\n# best_pipe = clone(best[\"pipe\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T15:40:41.44742Z","iopub.execute_input":"2025-10-07T15:40:41.447697Z","iopub.status.idle":"2025-10-07T15:40:41.453517Z","shell.execute_reply.started":"2025-10-07T15:40:41.447679Z","shell.execute_reply":"2025-10-07T15:40:41.452779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Refit: LGB uses a tiny validation split for early stopping; others fit on full data\n# if best[\"model\"].lower().startswith(\"lgb\"):\n#     sss = StratifiedShuffleSplit(n_splits=1, test_size=0.10, random_state=42)\n#     tr_idx, va_idx = next(sss.split(X, y))\n#     X_tr, y_tr = X.iloc[tr_idx], y.iloc[tr_idx]\n#     X_va, y_va = X.iloc[va_idx], y.iloc[va_idx]\n\n#     best_pipe.fit(\n#         X_tr, y_tr,\n#         clf__eval_set=[(X_va, y_va)],\n#         clf__callbacks=[lgb.early_stopping(100, verbose=False),\n#                         lgb.log_evaluation(period=-1)]\n#     )\n\n#     del X_tr, X_va, y_tr, y_va\n#     gc.collect()\n# else:\n#     best_pipe.fit(X, y)\n\n\n# # --------------------- Build X_test aligned to training columns ---------------------\n# TEST_DROP = [\"case_id\", \"WEEK_NUM\"]\n# case_ids = df_test[\"case_id\"].values\n# X_test = df_test.drop(columns=[c for c in TEST_DROP if c in df_test.columns]).copy()\n\n# # add any train-only columns missing in test as NaN\n# missing_in_test = [c for c in X.columns if c not in X_test.columns]\n# for c in missing_in_test:\n#     X_test[c] = np.nan\n\n# # drop extras and match column order exactly\n# X_test = X_test[X.columns]\n\n# # enforce dtypes consistent with training split\n# for c in X_test.columns:\n#     if c in cat_cols_all:\n#         if not is_categorical_dtype(X_test[c]):\n#             X_test[c] = X_test[c].astype(\"category\")\n#     else:\n#         X_test[c] = pd.to_numeric(X_test[c], errors=\"coerce\").astype(\"float32\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T15:40:52.92546Z","iopub.execute_input":"2025-10-07T15:40:52.926175Z","iopub.status.idle":"2025-10-07T15:50:08.240223Z","shell.execute_reply.started":"2025-10-07T15:40:52.926137Z","shell.execute_reply":"2025-10-07T15:50:08.239326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # --------------------- Predict and write submission ---------------------\n# test_scores = best_pipe.predict_proba(X_test)[:, 1]\n# pred_series = pd.Series(test_scores, index=case_ids, name=\"score\")\n\n# try:\n#     df_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\n# except Exception:\n#     df_subm = pd.read_csv(\"sample_submission.csv\")\n\n# df_subm = df_subm.set_index(\"case_id\")\n# df_subm[\"score\"] = pred_series.reindex(df_subm.index).fillna(0.0)\n# df_subm.to_csv(\"submission.csv\")\n\n# print(\"[SUBMISSION] Saved submission.csv\")\n# print(df_subm.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T15:53:20.836909Z","iopub.execute_input":"2025-10-07T15:53:20.837476Z","iopub.status.idle":"2025-10-07T15:53:20.897304Z","shell.execute_reply.started":"2025-10-07T15:53:20.837451Z","shell.execute_reply":"2025-10-07T15:53:20.896756Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# LGB For Submission","metadata":{}},{"cell_type":"code","source":"# ---------------- Custom Voting Model ----------------\n# It averages predictions from multiple trained estimators (e.g., LGBM models).\nclass VotingModel(BaseEstimator, RegressorMixin):\n    def __init__(self, estimators):\n        super().__init__()\n        self.estimators = estimators\n\n    def fit(self, X, y=None):\n        # No fitting needed here; models are already trained.\n        return self\n\n    def predict(self, X):\n        # Average the predicted values from all models.\n        y_preds = [estimator.predict(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)\n\n    def predict_proba(self, X):\n        # Average the predicted probabilities from all models.\n        y_preds = [estimator.predict_proba(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:46:41.632933Z","iopub.execute_input":"2025-10-20T13:46:41.633665Z","iopub.status.idle":"2025-10-20T13:46:41.63889Z","shell.execute_reply.started":"2025-10-20T13:46:41.633642Z","shell.execute_reply":"2025-10-20T13:46:41.638129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ---------------- Data Preparation ----------------\n# Define training features (X), target labels (y), and group info (weeks)\nX = df_train.drop(columns=[\"target\", \"case_id\", \"WEEK_NUM\"])\ny = df_train[\"target\"]\nweeks = df_train[\"WEEK_NUM\"]\n\n# ---------------- Cross-validation setup ----------------\n# StratifiedGroupKFold ensures balanced label distribution and that each week stays within a single fold (to prevent leakage)\ncv = StratifiedGroupKFold(n_splits=5, shuffle=False)\n\n# ---------------- LightGBM model parameters ----------------\n# LGBM model parameters after the cross-validation process\nparams = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"max_depth\": 10,\n    \"learning_rate\": 0.05,\n    \"max_bin\": 255,\n    \"n_estimators\": 1200,\n    \"colsample_bytree\": 0.8,\n    \"colsample_bynode\": 0.8,\n    \"verbose\": -1,\n    \"random_state\": 42,\n    \"reg_alpha\": 0.1,\n    \"reg_lambda\": 10,\n    \"extra_trees\": True,\n    \"num_leaves\": 64,\n    \"device\": \"gpu\", \n}\n\n# ---------------- Model training loop ----------------\nfitted_models = []\ncv_scores = []\n\n# Loop through each fold for training and validation\nfor idx_train, idx_valid in cv.split(X, y, groups=weeks):\n    X_train, y_train = X.iloc[idx_train], y.iloc[idx_train]\n    X_valid, y_valid = X.iloc[idx_valid], y.iloc[idx_valid]\n\n    print(\"Validation week range: \", (weeks.iloc[idx_valid].min(), weeks.iloc[idx_valid].max()))\n\n    # Initialize and train the LightGBM model\n    model = lgb.LGBMClassifier(**params)\n    model.fit(\n        X_train, y_train,\n        eval_set=[(X_valid, y_valid)],\n        callbacks=[lgb.log_evaluation(50), lgb.early_stopping(50)]  # Early stop if no improvement\n    )\n\n    # Save the trained model\n    fitted_models.append(model)\n\n    # Evaluate AUC on the validation fold\n    y_pred_valid = model.predict_proba(X_valid)[:, 1]\n    auc_score = roc_auc_score(y_valid, y_pred_valid)\n    cv_scores.append(auc_score)\n\n# ---------------- Ensemble & Results ----------------\n# Combine all fold models into a simple average voting model\nmodel = VotingModel(fitted_models)\n\n# Print all CV fold scores and their mean\nprint(\"CV AUC scores: \", cv_scores)\nprint(\"Average CV AUC score: \", sum(cv_scores) / len(cv_scores))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:46:44.353722Z","iopub.execute_input":"2025-10-20T13:46:44.354021Z","iopub.status.idle":"2025-10-20T14:03:15.344743Z","shell.execute_reply.started":"2025-10-20T13:46:44.353999Z","shell.execute_reply":"2025-10-20T14:03:15.344056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = df_test.drop(columns=[\"WEEK_NUM\"])\nX_test = X_test.set_index(\"case_id\")\n\nlgb_pred = pd.Series(model.predict_proba(X_test)[:, 1], index=X_test.index)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T14:03:25.583346Z","iopub.execute_input":"2025-10-20T14:03:25.583665Z","iopub.status.idle":"2025-10-20T14:03:25.76001Z","shell.execute_reply.started":"2025-10-20T14:03:25.58364Z","shell.execute_reply":"2025-10-20T14:03:25.759358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prepare for submission\ndf_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\n\ndf_subm[\"score\"] = lgb_pred\n\nprint(\"Check null: \", df_subm[\"score\"].isnull().any())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T14:03:28.536075Z","iopub.execute_input":"2025-10-20T14:03:28.537239Z","iopub.status.idle":"2025-10-20T14:03:28.560267Z","shell.execute_reply.started":"2025-10-20T14:03:28.537213Z","shell.execute_reply":"2025-10-20T14:03:28.559424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_subm.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T14:03:30.056119Z","iopub.execute_input":"2025-10-20T14:03:30.056749Z","iopub.status.idle":"2025-10-20T14:03:30.064452Z","shell.execute_reply.started":"2025-10-20T14:03:30.056715Z","shell.execute_reply":"2025-10-20T14:03:30.063595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_subm.to_csv(\"submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T14:03:33.950236Z","iopub.execute_input":"2025-10-20T14:03:33.950962Z","iopub.status.idle":"2025-10-20T14:03:33.959436Z","shell.execute_reply.started":"2025-10-20T14:03:33.950936Z","shell.execute_reply":"2025-10-20T14:03:33.958724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}