{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":119083,"databundleVersionId":15997195}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Cyber-Physical Anomaly Detection for DER Systems\n\nKaggle competition solution for detecting cyber-physical anomalies in \nDistributed Energy Resource (DER) telemetry. The competition metric is \n**F2 (beta=2)**, which weights recall twice as heavily as precision.\n\n**Final results:**\n- Combined OOF F2: **0.91179**\n- Small system model: AUC **0.99923**, F2 **0.99180**\n- Full model OOF AUC: **0.87705**\n\n**Approach:**  \nLightGBM-only solution with WMax-based subgroup routing — small systems \n(WMax ≤ 12k) get a dedicated Optuna-tuned model; Group A and string-rule \nslices are handled deterministically; Group 20k and Group B are scored via \na full-data model. 28 physics-aligned engineered features centered on \ncapacity utilization ratios and IEEE 1547 compliance residuals.","metadata":{}},{"cell_type":"markdown","source":"## 1. Setup: Imports, Paths & Helpers","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport polars as pl\nimport lightgbm as lgb\nimport re, gc\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import fbeta_score, roc_auc_score\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nTRAIN_PATH = \"/kaggle/input/competitions/cyber-physical-anomaly-detection-for-der-systems/train.csv\"\nTEST_PATH  = \"/kaggle/input/competitions/cyber-physical-anomaly-detection-for-der-systems/test.csv\"\nOUT_DIR    = \"/kaggle/working/\"\n\ndef f2_score(y_true, y_prob, threshold=0.5):\n    return fbeta_score(y_true, (y_prob >= threshold).astype(int),\n                       beta=2, zero_division=0)\n\ndef find_best_threshold(y_true, y_prob, low=0.05, high=0.95, step=0.005):\n    thresholds = np.arange(low, high, step)\n    scores     = [f2_score(y_true, y_prob, t) for t in thresholds]\n    best_idx   = np.argmax(scores)\n    return thresholds[best_idx], scores[best_idx]\n\nprint(f\"LightGBM version: {lgb.__version__}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T20:51:11.769685Z","iopub.execute_input":"2026-04-05T20:51:11.770020Z","iopub.status.idle":"2026-04-05T20:51:20.527156Z","shell.execute_reply.started":"2026-04-05T20:51:11.769996Z","shell.execute_reply":"2026-04-05T20:51:20.526378Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Load Training Data, Clean & Build Group Masks","metadata":{}},{"cell_type":"code","source":"# ── Single CSV read — collect into RAM ────────────────────────\nlazy = pl.scan_csv(TRAIN_PATH)\nschema = lazy.collect_schema()\nrename_map = {c: re.sub(r\"[\\[\\]<>]\", \"_\", c) for c in schema.names()}\nlazy = lazy.rename(rename_map)\ntrain_pl = lazy.collect()\nprint(f\"Raw shape: {train_pl.shape}\")\n\n# ── Drop 100% null columns ────────────────────────────────────\nnull_counts = train_pl.null_count()\nnull_cols   = [c for c in null_counts.columns\n               if null_counts[c][0] == train_pl.height]\ntrain_pl    = train_pl.drop(null_cols)\nprint(f\"Dropped {len(null_cols)} fully null columns\")\n\n# ── Drop string columns ───────────────────────────────────────\nstring_cols = [c for c, dt in train_pl.schema.items() if dt == pl.String]\ntrain_pl    = train_pl.drop(string_cols)\nprint(f\"Dropped {len(string_cols)} string columns\")\n\n# ── Drop dead (constant) columns ──────────────────────────────\nnum_cols  = [c for c, dt in train_pl.schema.items()\n             if dt in (pl.Float32, pl.Float64, pl.Int32, pl.Int64,\n                       pl.Int8, pl.Int16)\n             and c not in (\"Label\", \"Id\")]\nstd_vals  = train_pl.select(num_cols).select(pl.all().std()).row(0)\ndead_cols = [c for c, s in zip(num_cols, std_vals) if (s or 0.0) < 1e-6]\ntrain_pl  = train_pl.drop(dead_cols)\nprint(f\"Dropped {len(dead_cols)} constant columns\")\n\nfeat_cols = [c for c in train_pl.columns if c not in (\"Label\", \"Id\")]\nprint(f\"Final feature count: {len(feat_cols)}\")\nprint(f\"Final train shape  : {train_pl.shape}\")\n\n# ── Extract key arrays ────────────────────────────────────────\ny_all    = train_pl[\"Label\"].to_numpy().astype(np.int8)\nwmax_all = train_pl[\"DERCapacity_0_.WMax\"].fill_null(0).to_numpy()\n\nprint(f\"\\nTotal rows   : {len(y_all):,}\")\nprint(f\"Anomaly rate : {y_all.mean():.3f}\")\n\n# ── Load string columns for rule-based masks ──────────────────\nstr_df = pl.scan_csv(TRAIN_PATH).select([\n    \"Id\", \"common[0].Md\", \"common[0].SN\",\n    \"DERCapacity[0].WMax\", \"Label\"\n]).collect()\n\nsn_train = str_df[\"common[0].SN\"].to_numpy().astype(str)\nmd_train = str_df[\"common[0].Md\"].to_numpy().astype(str)\nwmax_str = str_df[\"DERCapacity[0].WMax\"].fill_null(0).to_numpy()\n\n# ── Build all group masks ─────────────────────────────────────\nGROUP_A_WMAX   = {14000.0, 16000.0, 18000.0}\nGROUP_20K_WMAX = {20000.0}\n\nsmall_mask       = wmax_all <= 12000\nlarge_mask       = ~small_mask\ngroup_a_mask     = np.isin(wmax_all, list(GROUP_A_WMAX))\ngroup_20k_mask   = np.isin(wmax_all, list(GROUP_20K_WMAX))\nrule_sn_mask     = (sn_train == \"1100058974.0\") | (sn_train == \"None\")\nrule_md_mask     = (md_train == \"DER Simulator\") & (wmax_str > 12000)\nrule_md_null     = (md_train == \"None\") & (wmax_str > 12000)\nstring_rule_mask = rule_sn_mask | rule_md_mask | rule_md_null\ngroup_b_mask     = (large_mask & ~group_a_mask\n                    & ~group_20k_mask & ~string_rule_mask)\n\nprint(\"\\n=== Group sizes and anomaly rates ===\")\nfor name, mask in [\n    (\"Small (WMax ≤ 12k)\",       small_mask),\n    (\"Group A (14/16/18k)\",      group_a_mask),\n    (\"Group 20k\",                group_20k_mask),\n    (\"String rules (100% anom)\", string_rule_mask),\n    (\"Group B (remainder)\",      group_b_mask),\n]:\n    n    = mask.sum()\n    rate = y_all[mask].mean() if n > 0 else 0\n    print(f\"  {name:<30} {n:>10,}   anomaly_rate={rate:.3f}\")\n\ndel str_df\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T20:51:25.074608Z","iopub.execute_input":"2026-04-05T20:51:25.075069Z","iopub.status.idle":"2026-04-05T20:52:53.315610Z","shell.execute_reply.started":"2026-04-05T20:51:25.075035Z","shell.execute_reply":"2026-04-05T20:52:53.314980Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Feature Engineering","metadata":{}},{"cell_type":"code","source":"def add_features(df):\n    wmax  = df[\"DERCapacity_0_.WMax\"].fillna(0)\n    vamax = df[\"DERCapacity_0_.VAMax\"].fillna(1).clip(lower=1) \\\n            if \"DERCapacity_0_.VAMax\" in df.columns \\\n            else pd.Series(1, index=df.index)\n    w     = df[\"DERMeasureAC_0_.W\"].fillna(0) \\\n            if \"DERMeasureAC_0_.W\" in df.columns \\\n            else pd.Series(0, index=df.index)\n    va    = df[\"DERMeasureAC_0_.VA\"].fillna(0) \\\n            if \"DERMeasureAC_0_.VA\" in df.columns \\\n            else pd.Series(0, index=df.index)\n    var_  = df[\"DERMeasureAC_0_.Var\"].fillna(0) \\\n            if \"DERMeasureAC_0_.Var\" in df.columns \\\n            else pd.Series(0, index=df.index)\n    varmax_abs = df[\"DERCapacity_0_.VarMaxAbs\"].fillna(1).clip(lower=1) \\\n                 if \"DERCapacity_0_.VarMaxAbs\" in df.columns \\\n                 else pd.Series(1, index=df.index)\n    varmax_inj = df[\"DERCapacity_0_.VarMaxInj\"].fillna(1).clip(lower=1) \\\n                 if \"DERCapacity_0_.VarMaxInj\" in df.columns \\\n                 else pd.Series(1, index=df.index)\n\n    # Power ratios\n    df[\"W_over_WMax\"]        = (w / wmax.clip(lower=1)).astype(np.float32)\n    df[\"VA_over_VAMax\"]      = (va / vamax).astype(np.float32)\n    df[\"Var_over_VarMaxAbs\"] = (var_ / varmax_abs).astype(np.float32)\n    df[\"Var_over_VarMaxInj\"] = (var_ / varmax_inj).astype(np.float32)\n    df[\"W_over_VA\"]          = (w / va.clip(lower=1)).astype(np.float32)\n    df[\"WMax_over_VAMax\"]    = (wmax / vamax).astype(np.float32)\n    df[\"W_minus_WMax\"]       = (wmax - w).astype(np.float32)\n    df[\"W_minus_WMax_pct\"]   = ((w - wmax) / wmax.clip(lower=1)).astype(np.float32)\n    df[\"W_exceeds_WMax\"]     = (w > wmax).astype(np.float32)\n    df[\"VA_exceeds_VAMax\"]   = (va > vamax).astype(np.float32)\n\n    # Log scale\n    df[\"log_WMax\"]  = np.log1p(wmax).astype(np.float32)\n    df[\"log_W\"]     = np.log1p(w.clip(lower=0)).astype(np.float32)\n    df[\"log_VAMax\"] = np.log1p(vamax).astype(np.float32)\n    df[\"log_VA\"]    = np.log1p(va.clip(lower=0)).astype(np.float32)\n\n    # IEEE 1547 compliance deviations\n    def dev(col, default):\n        return (df[col].fillna(default) - default).abs() \\\n               if col in df.columns else pd.Series(0, index=df.index)\n\n    dev_hz_hi   = dev(\"DEREnterService_0_.ESHzHi\",  65.0)\n    dev_hz_lo   = dev(\"DEREnterService_0_.ESHzLo\",  55.0)\n    dev_v_hi    = dev(\"DEREnterService_0_.ESVHi\",  112.0)\n    dev_v_lo    = dev(\"DEREnterService_0_.ESVLo\",   88.0)\n    dev_dly     = dev(\"DEREnterService_0_.ESDlyTms\",  2.0)\n    dev_pf_ovr  = dev(\"DERCapacity_0_.WOvrExtPF\",    0.8)\n    dev_pf_und  = dev(\"DERCapacity_0_.WUndExtPF\",    0.8)\n    dev_wmx_ovr = (df[\"DERCapacity_0_.WMaxOvrExt\"].fillna(0) - wmax).abs() \\\n                  if \"DERCapacity_0_.WMaxOvrExt\" in df.columns \\\n                  else pd.Series(0, index=df.index)\n    dev_wmx_und = (df[\"DERCapacity_0_.WMaxUndExt\"].fillna(0) - wmax).abs() \\\n                  if \"DERCapacity_0_.WMaxUndExt\" in df.columns \\\n                  else pd.Series(0, index=df.index)\n\n    df[\"dev_ESHzHi\"]     = dev_hz_hi.astype(np.float32)\n    df[\"dev_ESHzLo\"]     = dev_hz_lo.astype(np.float32)\n    df[\"dev_ESVHi\"]      = dev_v_hi.astype(np.float32)\n    df[\"dev_ESVLo\"]      = dev_v_lo.astype(np.float32)\n    df[\"dev_ESDlyTms\"]   = dev_dly.astype(np.float32)\n    df[\"dev_WOvrExtPF\"]  = dev_pf_ovr.astype(np.float32)\n    df[\"dev_WUndExtPF\"]  = dev_pf_und.astype(np.float32)\n    df[\"dev_WMaxOvrExt\"] = dev_wmx_ovr.astype(np.float32)\n    df[\"dev_WMaxUndExt\"] = dev_wmx_und.astype(np.float32)\n\n    df[\"compliance_score\"] = (\n        dev_hz_hi  / 0.637  +\n        dev_hz_lo  / 0.585  +\n        dev_v_hi   / 0.916  +\n        dev_v_lo   / 0.498  +\n        dev_dly    / 27.39  +\n        dev_pf_ovr / 0.0065 +\n        dev_pf_und / 0.0065\n    ).astype(np.float32)\n\n    df[\"n_deviant\"] = (\n        (dev_hz_hi  > 0.01).astype(int) +\n        (dev_hz_lo  > 0.01).astype(int) +\n        (dev_v_hi   > 0.01).astype(int) +\n        (dev_v_lo   > 0.01).astype(int) +\n        (dev_dly    > 0.01).astype(int) +\n        (dev_pf_ovr > 0.001).astype(int) +\n        (dev_pf_und > 0.001).astype(int)\n    ).astype(np.float32)\n\n    df[\"is_small_system\"] = (wmax <= 12000).astype(np.float32)\n    df[\"is_large_system\"] = (wmax >  55000).astype(np.float32)\n    df[\"W_over_VAMax\"]    = (w / vamax).astype(np.float32)\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T20:52:53.316800Z","iopub.execute_input":"2026-04-05T20:52:53.317038Z","iopub.status.idle":"2026-04-05T20:52:53.331074Z","shell.execute_reply.started":"2026-04-05T20:52:53.317019Z","shell.execute_reply":"2026-04-05T20:52:53.330551Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Small System Model — 5-Fold LightGBM (GPU, WMax ≤ 12k)","metadata":{}},{"cell_type":"code","source":"# Slice small-system rows and engineer features\ntrain_small   = train_pl.filter(pl.Series(small_mask)).to_pandas()\ny_small       = y_all[small_mask]\ntrain_small   = add_features(train_small)\nall_feat_cols = [c for c in train_small.columns if c not in (\"Label\", \"Id\")]\n\nX_small = train_small[all_feat_cols].values.astype(np.float32)\ndel train_small\ngc.collect()\n\nprint(f\"X_small shape : {X_small.shape}\")\nprint(f\"y_small shape : {y_small.shape}\")\nprint(f\"Anomaly rate  : {y_small.mean():.3f}\")\n\n# Best params from Optuna search (hardcoded for reproducibility)\nlgb_params_small = {\n    \"objective\"        : \"binary\",\n    \"metric\"           : \"auc\",\n    \"boosting_type\"    : \"gbdt\",\n    \"device\"           : \"gpu\",\n    \"gpu_platform_id\"  : 0,\n    \"gpu_device_id\"    : 0,\n    \"max_bin\"          : 255,\n    \"gpu_use_dp\"       : True,\n    \"num_leaves\"       : 192,\n    \"max_depth\"        : 8,\n    \"min_data_in_leaf\" : 59,\n    \"learning_rate\"    : 0.06925131019797677,\n    \"feature_fraction\" : 0.6045823462172908,\n    \"bagging_fraction\" : 0.9526908224423354,\n    \"bagging_freq\"     : 5,\n    \"lambda_l1\"        : 1.9790787014322153,\n    \"lambda_l2\"        : 1.8769523515026605,\n    \"verbose\"          : -1,\n    \"n_jobs\"           : -1,\n    \"seed\"             : 42,\n}\n\nskf          = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\noof_small    = np.zeros(len(y_small))\nsmall_models = []\n\nprint(f\"\\nTraining 5-fold | {X_small.shape[0]:,} rows | {X_small.shape[1]} features\\n\")\n\nfor fold, (tr_idx, val_idx) in enumerate(skf.split(X_small, y_small)):\n    dtrain = lgb.Dataset(X_small[tr_idx], y_small[tr_idx],\n                         feature_name=all_feat_cols, free_raw_data=True)\n    dval   = lgb.Dataset(X_small[val_idx], y_small[val_idx],\n                         reference=dtrain, free_raw_data=True)\n    m = lgb.train(\n        lgb_params_small, dtrain,\n        num_boost_round=3000,\n        valid_sets=[dval],\n        callbacks=[\n            lgb.early_stopping(100, verbose=True),\n            lgb.log_evaluation(200),\n        ]\n    )\n    oof_small[val_idx] = m.predict(X_small[val_idx],\n                                    num_iteration=m.best_iteration)\n    small_models.append(m)\n\n    auc_f      = roc_auc_score(y_small[val_idx], oof_small[val_idx])\n    bt_f, f2_f = find_best_threshold(y_small[val_idx], oof_small[val_idx])\n    print(f\"Fold {fold+1} | iter={m.best_iteration:,} | \"\n          f\"AUC={auc_f:.5f} | F2={f2_f:.5f} @ t={bt_f:.3f}\")\n    del dtrain, dval\n    gc.collect()\n\nauc_small          = roc_auc_score(y_small, oof_small)\nbt_small, f2_small = find_best_threshold(y_small, oof_small)\n\nprint(f\"\\nSmall system OOF | AUC={auc_small:.5f} | \"\n      f\"F2={f2_small:.5f} @ threshold={bt_small:.3f}\")\n\ndel X_small\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T20:53:59.117483Z","iopub.execute_input":"2026-04-05T20:53:59.118244Z","iopub.status.idle":"2026-04-05T21:06:36.883707Z","shell.execute_reply.started":"2026-04-05T20:53:59.118216Z","shell.execute_reply":"2026-04-05T21:06:36.882699Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Full Model — 5-Fold LightGBM (CPU, All Groups)","metadata":{}},{"cell_type":"code","source":"# Engineer features on full training set\ntrain_full = train_pl.to_pandas()\ntrain_full = add_features(train_full)\n\nX_full = train_full[all_feat_cols].values.astype(np.float32)\ndel train_full\ngc.collect()\n\n# Free train_pl — no longer needed after this point\ndel train_pl\ngc.collect()\n\nprint(f\"X_full shape: {X_full.shape}\")\n\n# Manually tuned params (CPU model — broader regularisation for full dataset)\nlgb_params_full = {\n    \"objective\"        : \"binary\",\n    \"metric\"           : \"auc\",\n    \"boosting_type\"    : \"gbdt\",\n    \"device\"           : \"cpu\",\n    \"num_leaves\"       : 63,\n    \"max_depth\"        : 6,\n    \"min_data_in_leaf\" : 100,\n    \"learning_rate\"    : 0.05,\n    \"feature_fraction\" : 0.7,\n    \"bagging_fraction\" : 0.8,\n    \"bagging_freq\"     : 5,\n    \"lambda_l1\"        : 1.0,\n    \"lambda_l2\"        : 5.0,\n    \"verbose\"          : -1,\n    \"n_jobs\"           : -1,\n    \"seed\"             : 42,\n}\n\nskf_full    = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\noof_full    = np.zeros(len(y_all))\nfull_models = []\n\nprint(f\"\\nTraining 5-fold | {X_full.shape[0]:,} rows | {X_full.shape[1]} features\\n\")\n\nfor fold, (tr_idx, val_idx) in enumerate(skf_full.split(X_full, y_all)):\n    dtrain = lgb.Dataset(X_full[tr_idx], y_all[tr_idx],\n                         feature_name=all_feat_cols, free_raw_data=True)\n    dval   = lgb.Dataset(X_full[val_idx], y_all[val_idx],\n                         reference=dtrain, free_raw_data=True)\n    m = lgb.train(\n        lgb_params_full, dtrain,\n        num_boost_round=3000,\n        valid_sets=[dval],\n        callbacks=[\n            lgb.early_stopping(150, verbose=True),\n            lgb.log_evaluation(200),\n        ]\n    )\n    oof_full[val_idx] = m.predict(X_full[val_idx],\n                                   num_iteration=m.best_iteration)\n    full_models.append(m)\n\n    auc_f = roc_auc_score(y_all[val_idx], oof_full[val_idx])\n    print(f\"Fold {fold+1} | iter={m.best_iteration:,} | AUC={auc_f:.5f}\")\n    del dtrain, dval\n    gc.collect()\n\nauc_full = roc_auc_score(y_all, oof_full)\nprint(f\"\\nFull model OOF AUC: {auc_full:.5f}\")\n\ndel X_full\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T21:11:00.825244Z","iopub.execute_input":"2026-04-05T21:11:00.825641Z","execution_failed":"2026-04-05T21:15:25.379Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6. Combine OOF Predictions & Verify F2","metadata":{}},{"cell_type":"code","source":"small_idx_arr = np.where(small_mask)[0]\nlarge_idx_arr = np.where(large_mask)[0]\n\nTHRESHOLD_SMALL = bt_small  # best threshold from Cell 4 Optuna-tuned model\nTHRESHOLD_LARGE = 0.05\n\n# Build combined OOF probabilities\noof_combined = np.zeros(len(y_all))\noof_combined[small_idx_arr]    = oof_small\noof_combined[large_idx_arr]    = oof_full[large_idx_arr]\noof_combined[group_a_mask]     = 1.0\noof_combined[string_rule_mask] = 1.0\n\n# Build combined hard predictions\ny_pred = np.zeros(len(y_all), dtype=int)\n\n# Small system — model threshold\ny_pred[small_idx_arr] = (oof_small >= THRESHOLD_SMALL).astype(int)\n\n# Group A — deterministic rule\ny_pred[group_a_mask] = 1\n\n# String rules — deterministic rule\ny_pred[string_rule_mask] = 1\n\n# Group 20k — full model (exclude rows already set by string rules)\nmask_20k_only = group_20k_mask & ~string_rule_mask\ny_pred[mask_20k_only] = (oof_full[mask_20k_only] >= THRESHOLD_LARGE).astype(int)\n\n# Group B — full model\ny_pred[group_b_mask] = (oof_full[group_b_mask] >= THRESHOLD_LARGE).astype(int)\n\n# ── Metrics ───────────────────────────────────────────────────\ntp   = ((y_pred == 1) & (y_all == 1)).sum()\nfp   = ((y_pred == 1) & (y_all == 0)).sum()\nfn   = ((y_pred == 0) & (y_all == 1)).sum()\nprec = tp / (tp + fp)\nrec  = tp / (tp + fn)\nf2   = 5 * prec * rec / (4 * prec + rec)\n\nprint(f\"Combined OOF results:\")\nprint(f\"  F2        : {f2:.5f}  (expected ~0.91179)\")\nprint(f\"  Precision : {prec:.4f}\")\nprint(f\"  Recall    : {rec:.4f}\")\nprint(f\"  TP={tp:,}  FP={fp:,}  FN={fn:,}\")\nprint(f\"\\n  Threshold small : {THRESHOLD_SMALL:.3f}\")\nprint(f\"  Threshold large : {THRESHOLD_LARGE:.3f}\")","metadata":{"trusted":true,"execution":{"execution_failed":"2026-04-05T21:15:25.380Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 7. Generate Test Predictions & Save Submission","metadata":{}},{"cell_type":"code","source":"# ── Load and clean test ───────────────────────────────────────\nlazy_test   = pl.scan_csv(TEST_PATH)\nschema_test = lazy_test.collect_schema()\nrename_test = {c: re.sub(r\"[\\[\\]<>]\", \"_\", c) for c in schema_test.names()}\nlazy_test   = lazy_test.rename(rename_test)\ntest_pl     = lazy_test.collect()\nprint(f\"Raw test shape: {test_pl.shape}\")\n\n# Drop same columns as train\nrenamed_null      = [rename_test.get(c, c) for c in null_cols]\nrenamed_string    = [rename_test.get(c, c) for c in string_cols]\nrenamed_dead      = [rename_test.get(c, c) for c in dead_cols]\ncols_to_drop_test = [c for c in set(renamed_null + renamed_string + renamed_dead)\n                     if c in test_pl.columns]\ntest_pl = test_pl.drop(cols_to_drop_test)\nprint(f\"Test shape after cleaning: {test_pl.shape}\")\n\n# ── Load test string columns for rules ───────────────────────\nstr_test      = pl.scan_csv(TEST_PATH).select([\n    \"Id\", \"common[0].Md\", \"common[0].SN\", \"DERCapacity[0].WMax\"\n]).collect()\n\ntest_ids      = str_test[\"Id\"].to_numpy()\nsn_test       = str_test[\"common[0].SN\"].to_numpy().astype(str)\nmd_test       = str_test[\"common[0].Md\"].to_numpy().astype(str)\nwmax_test_raw = str_test[\"DERCapacity[0].WMax\"].fill_null(0).to_numpy()\n\n# ── Build test group masks ────────────────────────────────────\nwmax_test_num = test_pl[\"DERCapacity_0_.WMax\"].fill_null(0).to_numpy() \\\n                if \"DERCapacity_0_.WMax\" in test_pl.columns \\\n                else wmax_test_raw\n\nsmall_test       = wmax_test_num <= 12000\nlarge_test       = ~small_test\ngroup_a_test     = np.isin(wmax_test_num, list(GROUP_A_WMAX))\ngroup_20k_test   = np.isin(wmax_test_num, list(GROUP_20K_WMAX))\nrule_sn_test     = (sn_test == \"1100058974.0\") | (sn_test == \"None\")\nrule_md_test     = (md_test == \"DER Simulator\") & (wmax_test_raw > 12000)\nrule_md_null_t   = (md_test == \"None\") & (wmax_test_raw > 12000)\nstring_rule_test = rule_sn_test | rule_md_test | rule_md_null_t\ngroup_b_test     = (large_test & ~group_a_test\n                    & ~group_20k_test & ~string_rule_test)\n\nprint(f\"\\nTest group sizes:\")\nprint(f\"  Small (WMax ≤ 12k)    : {small_test.sum():,}\")\nprint(f\"  Group A (rule=1)      : {group_a_test.sum():,}\")\nprint(f\"  Group 20k             : {group_20k_test.sum():,}\")\nprint(f\"  String rules (rule=1) : {string_rule_test.sum():,}\")\nprint(f\"  Group B               : {group_b_test.sum():,}\")\ntotal_check = (small_test.sum() + group_a_test.sum() + group_20k_test.sum()\n               + string_rule_test.sum() + group_b_test.sum())\nprint(f\"  Total (check)         : {total_check:,}  (expected {len(test_ids):,})\")\n\n# ── Engineer features and predict ────────────────────────────\ntest_pd   = test_pl.to_pandas()\ntest_pd   = add_features(test_pd)\ntest_cols = [c for c in all_feat_cols if c in test_pd.columns]\nX_test    = test_pd[test_cols].values.astype(np.float32)\ndel test_pd, test_pl, str_test\ngc.collect()\n\nprint(f\"\\nX_test shape: {X_test.shape}\")\n\n# Small system — average of 5 small models\nsmall_test_idx = np.where(small_test)[0]\npreds_small    = np.mean(\n    [m.predict(X_test[small_test], num_iteration=m.best_iteration)\n     for m in small_models], axis=0\n)\n\n# Large system — average of 5 full models\nlarge_test_idx = np.where(large_test)[0]\npreds_large    = np.mean(\n    [m.predict(X_test[large_test], num_iteration=m.best_iteration)\n     for m in full_models], axis=0\n)\n\ndel X_test\ngc.collect()\n\n# ── Assemble final predictions ────────────────────────────────\nfinal_labels = np.zeros(len(test_ids), dtype=int)\n\n# Small system — model threshold\nfinal_labels[small_test_idx] = (preds_small >= THRESHOLD_SMALL).astype(int)\n\n# Group A — deterministic rule (100% anomaly in train)\nfinal_labels[group_a_test] = 1\n\n# String rules — deterministic rule (100% anomaly in train)\nfinal_labels[string_rule_test] = 1\n\n# Group 20k and Group B — large model predictions\nlarge_20k_in_large = group_20k_test[large_test] & ~string_rule_test[large_test]\nlarge_b_in_large   = group_b_test[large_test]\n\nfinal_labels[large_test_idx[large_20k_in_large]] = (\n    preds_large[large_20k_in_large] >= THRESHOLD_LARGE\n).astype(int)\n\nfinal_labels[large_test_idx[large_b_in_large]] = (\n    preds_large[large_b_in_large] >= THRESHOLD_LARGE\n).astype(int)\n\n# ── Save submission ───────────────────────────────────────────\nsubmission = pd.DataFrame({\"Id\": test_ids, \"Label\": final_labels})\nsubmission.to_csv(OUT_DIR + \"submission.csv\", index=False)\n\n# ── Verify ────────────────────────────────────────────────────\nsub_check = pd.read_csv(OUT_DIR + \"submission.csv\")\nprint(f\"\\nTest stats:\")\nprint(f\"  Total rows         : {len(final_labels):,}\")\nprint(f\"  Predicted anomalies: {final_labels.sum():,}  ({final_labels.mean():.4f})\")\nprint(f\"\\nVerification:\")\nprint(f\"  Rows    : {len(sub_check):,}  (expected 1,012,785)\")\nprint(f\"  Columns : {list(sub_check.columns)}\")\nprint(f\"  Labels  : {sorted(sub_check['Label'].unique().tolist())}\")\nprint(f\"  No nulls: {sub_check['Label'].isnull().sum() == 0}\")\n\nprint(f\"\\n{'='*60}\")\nprint(f\"  FINAL RESULTS\")\nprint(f\"{'='*60}\")\nprint(f\"  Small system : AUC={auc_small:.5f}  F2={f2_small:.5f}\")\nprint(f\"  Full model   : AUC={auc_full:.5f}\")\nprint(f\"  Combined OOF : F2={f2:.5f}\")\nprint(f\"{'='*60}\")\nprint(f\"\\nSubmission saved to: {OUT_DIR}submission.csv\")","metadata":{"trusted":true,"execution":{"execution_failed":"2026-04-05T21:15:25.380Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}