{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":119083,"databundleVersionId":15997195,"isSourceIdPinned":false}],"dockerImageVersionId":31328,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Cyber-Physical Anomaly Detection for DER Systems\n## Solution: CatBoost + LightGBM Ensemble with Per-Device Z-Score Normalization\n\n**Private LB F2: 0.91190**\n\n### Key ideas\n1. Aggressive column pruning (725 → ~260 → top 120 by gain)\n2. Domain-specific feature engineering: capacity utilization, frequency deviation, per-device z-scores\n3. Multi-seed LightGBM (2 seeds × 5 folds) + CatBoost ensemble\n4. Optimized LGB/CB blend weight and F2 threshold","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import fbeta_score, precision_score, recall_score\nfrom sklearn.preprocessing import LabelEncoder\nimport warnings\nimport gc\nimport re\nimport time\n\nwarnings.filterwarnings('ignore')\n\nSEED = 42\nN_FOLDS = 5\nEARLY_STOPPING = 200\nSAMPLE_ROWS = 50000\n\nINPUT_DIR = '/kaggle/input/competitions/cyber-physical-anomaly-detection-for-der-systems'\nOUTPUT_DIR = '/kaggle/working'\n\nnp.random.seed(SEED)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-05T21:36:28.412293Z","iopub.execute_input":"2026-04-05T21:36:28.413577Z","iopub.status.idle":"2026-04-05T21:36:28.422693Z","shell.execute_reply.started":"2026-04-05T21:36:28.413534Z","shell.execute_reply":"2026-04-05T21:36:28.421723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def _downcast_df(df):\n    for col in df.select_dtypes(include=['float64']).columns:\n        df[col] = df[col].astype(np.float32)\n    for col in df.select_dtypes(include=['int64']).columns:\n        df[col] = pd.to_numeric(df[col], downcast='integer')\n    return df\n\n\ndef _safe_div(a, b, fill=0.0):\n    return np.where(np.abs(b) > 1e-10, a / b, fill).astype(np.float32)\n\n\ndef _profile_columns(csv_path, sample_rows=SAMPLE_ROWS):\n    sample = pd.read_csv(csv_path, nrows=sample_rows)\n    info = {}\n    for col in sample.columns:\n        null_frac = sample[col].isnull().mean()\n        nuniq = sample[col].nunique(dropna=True)\n        dtype = sample[col].dtype\n        info[col] = {'null_frac': null_frac, 'nunique': nuniq, 'dtype': dtype}\n    del sample; gc.collect()\n    return info","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T21:36:31.172277Z","iopub.execute_input":"2026-04-05T21:36:31.172646Z","iopub.status.idle":"2026-04-05T21:36:31.181823Z","shell.execute_reply.started":"2026-04-05T21:36:31.172597Z","shell.execute_reply":"2026-04-05T21:36:31.180618Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1. Profile & Load Data","metadata":{}},{"cell_type":"code","source":"print(\"[1/8] Loading data...\")\nt0 = time.time()\n\ntrain_profile = _profile_columns(f'{INPUT_DIR}/train.csv')\nall_cols = list(train_profile.keys())\n\nskip_cols = set()\nfor col in all_cols:\n    if col in ('Id', 'Label'):\n        continue\n    tp = train_profile[col]\n    if tp['null_frac'] == 1.0 or (tp['nunique'] <= 1 and tp['null_frac'] < 1.0) or tp['null_frac'] > 0.95:\n        skip_cols.add(col)\n\nuse_cols_train = [c for c in all_cols if c not in skip_cols]\nuse_cols_test = [c for c in use_cols_train if c != 'Label']\n\ndtype_map = {}\nfor col in use_cols_train:\n    tp = train_profile[col]\n    if tp['dtype'] == np.float64:\n        dtype_map[col] = np.float32\n    elif tp['dtype'] == np.int64:\n        dtype_map[col] = np.int32\n\ntrain = pd.read_csv(f'{INPUT_DIR}/train.csv', usecols=use_cols_train, dtype=dtype_map)\ngc.collect()\ntrain_ids = train['Id'].values\ny = train['Label'].values\ntrain.drop(['Id', 'Label'], axis=1, inplace=True)\ngc.collect()\n\ntest = pd.read_csv(f'{INPUT_DIR}/test.csv', usecols=use_cols_test, dtype=dtype_map)\ngc.collect()\ntest_ids = test['Id'].values\ntest.drop('Id', axis=1, inplace=True)\ngc.collect()\n\nprint(f\"  Train: {train.shape}, Test: {test.shape}\")\nprint(f\"  Loaded in {time.time()-t0:.1f}s\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T21:36:35.357375Z","iopub.execute_input":"2026-04-05T21:36:35.357778Z","iopub.status.idle":"2026-04-05T21:39:36.554725Z","shell.execute_reply.started":"2026-04-05T21:36:35.357744Z","shell.execute_reply":"2026-04-05T21:39:36.553593Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Preprocessing","metadata":{}},{"cell_type":"code","source":"print(\"[2/8] Preprocessing...\")\nt0 = time.time()\n\ncommon_cols = [c for c in train.columns if c in test.columns]\ntrain = train[common_cols]\ntest = test[common_cols]\n\n# Keep track of device SN before encoding\nsn_col = 'common[0].SN'\ntrain_sn_raw = train[sn_col].fillna('__MISSING__').values.astype(str)\ntest_sn_raw = test[sn_col].fillna('__MISSING__').values.astype(str)\n\n# Label encode object columns\nobj_cols = train.select_dtypes(include=['object']).columns.tolist()\nfor col in obj_cols:\n    le = LabelEncoder()\n    train[col] = train[col].fillna('__MISSING__')\n    le.fit(train[col])\n    train[col] = le.transform(train[col]).astype(np.int32)\n    test[col] = test[col].fillna('__MISSING__')\n    test_vals = test[col].values\n    known = set(le.classes_)\n    test_vals = np.array([v if v in known else '__MISSING__' for v in test_vals])\n    test[col] = le.transform(test_vals).astype(np.int32)\n\ntrain = _downcast_df(train)\ntest = _downcast_df(test)\ngc.collect()\n\n# Drop duplicate columns via hash comparison\nsample_idx = np.random.choice(len(train), min(50000, len(train)), replace=False)\ncol_hashes = {}\ndup_cols = []\nfor col in train.columns:\n    h = hash(train[col].values[sample_idx].tobytes())\n    if h in col_hashes:\n        dup_cols.append(col)\n    else:\n        col_hashes[h] = col\nif dup_cols:\n    print(f\"  Dropping {len(dup_cols)} duplicate columns\")\n    train.drop(dup_cols, axis=1, inplace=True)\n    test.drop(dup_cols, axis=1, inplace=True)\ndel col_hashes; gc.collect()\n\nprint(f\"  Features: {train.shape[1]}\")\nprint(f\"  Preprocessed in {time.time()-t0:.1f}s\")\n\nfeature_names = train.columns.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T21:58:16.148170Z","iopub.execute_input":"2026-04-05T21:58:16.149224Z","iopub.status.idle":"2026-04-05T21:58:25.336806Z","shell.execute_reply.started":"2026-04-05T21:58:16.149181Z","shell.execute_reply":"2026-04-05T21:58:25.335780Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Feature Engineering","metadata":{}},{"cell_type":"code","source":"print(\"[3/8] Feature engineering...\")\nt0 = time.time()\n\n\ndef _add_lean_features(df, feat_names, sn_raw):\n    cols = set(feat_names)\n\n    df['null_count'] = df[feat_names].isnull().sum(axis=1).astype(np.float32)\n\n    # AC aggregations\n    ac_cols = [c for c in feat_names if 'DERMeasureAC[0]' in c\n               and df[c].dtype in [np.float32, np.float64, np.int32, np.int64]]\n    if ac_cols:\n        ac_data = df[ac_cols]\n        df['ac_mean'] = ac_data.mean(axis=1).astype(np.float32)\n        df['ac_std'] = ac_data.std(axis=1).astype(np.float32)\n        df['ac_min'] = ac_data.min(axis=1).astype(np.float32)\n        df['ac_max'] = ac_data.max(axis=1).astype(np.float32)\n        df['ac_range'] = (df['ac_max'] - df['ac_min']).astype(np.float32)\n        df['ac_skew'] = ac_data.skew(axis=1).astype(np.float32)\n        del ac_data\n\n    # DC aggregations\n    dc_cols = [c for c in feat_names if 'DERMeasureDC[0]' in c\n               and df[c].dtype in [np.float32, np.float64, np.int32, np.int64]]\n    if dc_cols:\n        dc_data = df[dc_cols]\n        df['dc_mean'] = dc_data.mean(axis=1).astype(np.float32)\n        df['dc_std'] = dc_data.std(axis=1).astype(np.float32)\n        del dc_data\n\n    # Ratios\n    if 'DERMeasureAC[0].W' in cols and 'DERMeasureAC[0].VA' in cols:\n        df['W_VA_ratio'] = _safe_div(df['DERMeasureAC[0].W'].values, df['DERMeasureAC[0].VA'].values)\n    if 'DERMeasureAC[0].W' in cols and 'DERMeasureAC[0].Var' in cols:\n        df['W_Var_ratio'] = _safe_div(df['DERMeasureAC[0].W'].values, df['DERMeasureAC[0].Var'].values)\n\n    # Capacity utilization\n    cap_pairs = [\n        ('DERMeasureAC[0].W', 'DERCapacity[0].WMax', 'W_utilization'),\n        ('DERMeasureAC[0].W', 'DERCapacity[0].WMaxRtg', 'W_utilization_rtg'),\n        ('DERMeasureAC[0].VA', 'DERCapacity[0].VAMax', 'VA_utilization'),\n        ('DERMeasureAC[0].Var', 'DERCapacity[0].VarMaxInj', 'Var_util_inj'),\n        ('DERMeasureAC[0].Var', 'DERCapacity[0].VarMaxAbs', 'Var_util_abs'),\n    ]\n    for actual, capacity, name in cap_pairs:\n        if actual in cols and capacity in cols:\n            df[name] = _safe_div(df[actual].values, df[capacity].values)\n\n    # Capacity operational vs rated\n    cap_ratio_pairs = [\n        ('DERCapacity[0].WMax', 'DERCapacity[0].WMaxRtg', 'W_cap_ratio'),\n        ('DERCapacity[0].VAMax', 'DERCapacity[0].VAMaxRtg', 'VA_cap_ratio'),\n        ('DERCapacity[0].VarMaxInj', 'DERCapacity[0].VarMaxInjRtg', 'VarInj_cap_ratio'),\n        ('DERCapacity[0].VarMaxAbs', 'DERCapacity[0].VarMaxAbsRtg', 'VarAbs_cap_ratio'),\n    ]\n    for op, rated, name in cap_ratio_pairs:\n        if op in cols and rated in cols:\n            df[name] = _safe_div(df[op].values, df[rated].values)\n            df[name + '_diff'] = (df[op].values - df[rated].values).astype(np.float32)\n\n    # Frequency deviation\n    if 'DERMeasureAC[0].Hz' in cols:\n        hz = df['DERMeasureAC[0].Hz'].values\n        df['hz_dev_60'] = (hz - 60.0).astype(np.float32)\n        df['hz_dev_60_abs'] = np.abs(hz - 60.0).astype(np.float32)\n\n    # Control aggregation\n    ctl_cols = [c for c in feat_names if 'DERCtlAC[0]' in c\n                and df[c].dtype in [np.float32, np.float64, np.int32, np.int64]]\n    if ctl_cols:\n        ctl_data = df[ctl_cols]\n        df['ctl_mean'] = ctl_data.mean(axis=1).astype(np.float32)\n        df['ctl_std'] = ctl_data.std(axis=1).astype(np.float32)\n        del ctl_data\n\n    # Per-device normalized features (the two devices have VERY different scales)\n    sn_series = pd.Series(sn_raw, index=df.index)\n    key_meas = ['DERMeasureAC[0].W', 'DERMeasureAC[0].VA', 'DERMeasureAC[0].Var',\n                'DERMeasureAC[0].PF', 'DERMeasureAC[0].Hz',\n                'DERCapacity[0].WMax', 'DERCapacity[0].VAMax']\n    for c in key_meas:\n        if c in cols:\n            grouped_mean = sn_series.map(df.groupby(sn_series)[c].transform('mean'))\n            grouped_std = sn_series.map(df.groupby(sn_series)[c].transform('std'))\n            safe_std = grouped_std.replace(0, 1)\n            df[c.split('.')[-1] + '_dev_zscore'] = ((df[c] - grouped_mean) / safe_std).astype(np.float32)\n\n    gc.collect()\n    return df\n\n\nX_train = _add_lean_features(train, feature_names, train_sn_raw)\ndel train; gc.collect()\nX_test = _add_lean_features(test, feature_names, test_sn_raw)\ndel test; gc.collect()\n\nnew_feats = [c for c in X_train.columns if c not in feature_names]\nprint(f\"  Created {len(new_feats)} features\")\n\n# Sanitize column names\nclean_names = {c: re.sub(r'[\\[\\]{}\"\\']', '_', c) for c in X_train.columns}\nX_train.rename(columns=clean_names, inplace=True)\nX_test.rename(columns=clean_names, inplace=True)\n\nprint(f\"  X_train: {X_train.shape}, X_test: {X_test.shape}\")\nprint(f\"  Done in {time.time()-t0:.1f}s\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T21:58:36.963863Z","iopub.execute_input":"2026-04-05T21:58:36.964261Z","iopub.status.idle":"2026-04-05T21:59:14.976544Z","shell.execute_reply.started":"2026-04-05T21:58:36.964226Z","shell.execute_reply":"2026-04-05T21:59:14.975569Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Feature Selection","metadata":{}},{"cell_type":"code","source":"print(\"[4/8] Feature selection...\")\nt0 = time.time()\n\nquick_params = {\n    'objective': 'binary', 'metric': 'binary_logloss', 'boosting_type': 'gbdt',\n    'learning_rate': 0.1, 'num_leaves': 127, 'min_child_samples': 100,\n    'feature_fraction': 0.7, 'bagging_fraction': 0.8, 'bagging_freq': 5,\n    'reg_alpha': 0.1, 'reg_lambda': 1.0, 'n_jobs': -1, 'verbose': -1,\n    'seed': 42, 'is_unbalance': True,\n}\n\nidx = np.arange(len(X_train))\nnp.random.seed(42)\nnp.random.shuffle(idx)\nsplit = int(0.8 * len(idx))\n\ndtrain = lgb.Dataset(X_train.iloc[idx[:split]], label=y[idx[:split]])\ndvalid = lgb.Dataset(X_train.iloc[idx[split:]], label=y[idx[split:]])\n\nqm = lgb.train(quick_params, dtrain, num_boost_round=500,\n                valid_sets=[dvalid], callbacks=[lgb.early_stopping(50), lgb.log_evaluation(500)])\n\nimportance = qm.feature_importance(importance_type='gain')\nfeat_imp_df = pd.DataFrame({'feature': X_train.columns, 'importance': importance})\nfeat_imp_df = feat_imp_df.sort_values('importance', ascending=False)\n\ntop_features = feat_imp_df[feat_imp_df['importance'] > 0].head(120)['feature'].tolist()\nprint(f\"  Selected {len(top_features)} features\")\n\nX_train = X_train[top_features]; X_test = X_test[top_features]\ndel dtrain, dvalid, qm; gc.collect()\nprint(f\"  Done in {time.time()-t0:.1f}s\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T21:59:18.614346Z","iopub.execute_input":"2026-04-05T21:59:18.614755Z","iopub.status.idle":"2026-04-05T22:01:13.911341Z","shell.execute_reply.started":"2026-04-05T21:59:18.614721Z","shell.execute_reply":"2026-04-05T22:01:13.909867Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Train LightGBM (multi-seed)","metadata":{}},{"cell_type":"code","source":"print(\"[5/8] Training LightGBM (multi-seed)...\")\nt0 = time.time()\n\nlgb_params = {\n    'objective': 'binary', 'metric': 'binary_logloss', 'boosting_type': 'gbdt',\n    'learning_rate': 0.03, 'num_leaves': 255, 'max_depth': -1,\n    'min_child_samples': 100, 'feature_fraction': 0.6, 'bagging_fraction': 0.75,\n    'bagging_freq': 5, 'reg_alpha': 0.5, 'reg_lambda': 3.0,\n    'min_gain_to_split': 0.02, 'path_smooth': 1.0,\n    'n_jobs': -1, 'verbose': -1, 'is_unbalance': True,\n}\n\nSEEDS_LGB = [42, 2024]\noof_lgb = np.zeros(len(X_train))\ntest_lgb = np.zeros(len(X_test))\n\nfor seed_i, seed in enumerate(SEEDS_LGB):\n    lgb_params['seed'] = seed\n    oof_seed = np.zeros(len(X_train))\n    test_seed = np.zeros(len(X_test))\n    scores = []\n\n    skf = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=seed)\n    for fold_n, (tr_idx, val_idx) in enumerate(skf.split(X_train, y)):\n        dt = lgb.Dataset(X_train.iloc[tr_idx], label=y[tr_idx])\n        dv = lgb.Dataset(X_train.iloc[val_idx], label=y[val_idx], reference=dt)\n\n        model = lgb.train(lgb_params, dt, num_boost_round=10000,\n                          valid_sets=[dv],\n                          callbacks=[lgb.early_stopping(EARLY_STOPPING), lgb.log_evaluation(500)])\n\n        oof_seed[val_idx] = model.predict(X_train.iloc[val_idx])\n        test_seed += model.predict(X_test) / N_FOLDS\n\n        best_f2 = max(fbeta_score(y[val_idx], (oof_seed[val_idx] > t).astype(int), beta=2)\n                      for t in np.arange(0.10, 0.60, 0.01))\n        scores.append(best_f2)\n\n        del dt, dv, model; gc.collect()\n\n    print(f\"  Seed {seed}: mean F2 = {np.mean(scores):.6f}\")\n    oof_lgb += oof_seed / len(SEEDS_LGB)\n    test_lgb += test_seed / len(SEEDS_LGB)\n\nprint(f\"  LGB done in {time.time()-t0:.1f}s\")\n\n# OOF F2 for LGB\nbest_lgb_f2 = 0\nfor t in np.arange(0.05, 0.60, 0.005):\n    f2 = fbeta_score(y, (oof_lgb > t).astype(int), beta=2)\n    if f2 > best_lgb_f2:\n        best_lgb_f2 = f2\nprint(f\"  LGB OOF F2: {best_lgb_f2:.6f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T22:02:41.199929Z","iopub.execute_input":"2026-04-05T22:02:41.200302Z","iopub.status.idle":"2026-04-05T22:57:33.475943Z","shell.execute_reply.started":"2026-04-05T22:02:41.200273Z","shell.execute_reply":"2026-04-05T22:57:33.474577Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6. Train CatBoost","metadata":{}},{"cell_type":"code","source":"print(\"[6/8] Training CatBoost...\")\nt0 = time.time()\n\ntry:\n    from catboost import CatBoostClassifier, Pool\n    HAS_CB = True\nexcept ImportError:\n    print(\"  CatBoost not available, will use LGB only\")\n    HAS_CB = False\n\nif HAS_CB:\n    oof_cb = np.zeros(len(X_train))\n    test_cb = np.zeros(len(X_test))\n    cb_scores = []\n\n    cat_feat_indices = [i for i, c in enumerate(X_train.columns)\n                        if X_train[c].dtype in [np.int8, np.int16, np.int32]\n                        and X_train[c].nunique() < 20]\n    print(f\"  Cat features for CatBoost: {len(cat_feat_indices)}\")\n\n    skf = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=SEED)\n    for fold_n, (tr_idx, val_idx) in enumerate(skf.split(X_train, y)):\n        print(f\"  --- CB Fold {fold_n+1}/{N_FOLDS} ---\")\n\n        cb_model = CatBoostClassifier(\n            iterations=5000,\n            learning_rate=0.05,\n            depth=8,\n            l2_leaf_reg=5.0,\n            random_strength=1.0,\n            bagging_temperature=0.8,\n            border_count=254,\n            auto_class_weights='Balanced',\n            eval_metric='Logloss',\n            random_seed=SEED,\n            verbose=500,\n            early_stopping_rounds=200,\n            task_type='CPU',\n        )\n\n        train_pool = Pool(X_train.iloc[tr_idx], y[tr_idx], cat_features=cat_feat_indices)\n        val_pool = Pool(X_train.iloc[val_idx], y[val_idx], cat_features=cat_feat_indices)\n\n        cb_model.fit(train_pool, eval_set=val_pool, use_best_model=True)\n\n        oof_cb[val_idx] = cb_model.predict_proba(X_train.iloc[val_idx])[:, 1]\n        test_cb += cb_model.predict_proba(X_test)[:, 1] / N_FOLDS\n\n        best_f2 = max(fbeta_score(y[val_idx], (oof_cb[val_idx] > t).astype(int), beta=2)\n                      for t in np.arange(0.10, 0.60, 0.01))\n        cb_scores.append(best_f2)\n        print(f\"    CB Fold {fold_n+1} F2: {best_f2:.6f}\")\n\n        del cb_model, train_pool, val_pool; gc.collect()\n\n    print(f\"  CB done in {time.time()-t0:.1f}s\")\n    print(f\"  CB mean fold F2: {np.mean(cb_scores):.6f}\")\n\n    best_cb_f2 = 0\n    for t in np.arange(0.05, 0.60, 0.005):\n        f2 = fbeta_score(y, (oof_cb > t).astype(int), beta=2)\n        if f2 > best_cb_f2:\n            best_cb_f2 = f2\n    print(f\"  CB OOF F2: {best_cb_f2:.6f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T21:33:29.212189Z","iopub.status.idle":"2026-04-05T21:33:29.212553Z","shell.execute_reply.started":"2026-04-05T21:33:29.212395Z","shell.execute_reply":"2026-04-05T21:33:29.212418Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 7. Ensemble Optimization","metadata":{}},{"cell_type":"code","source":"print(\"[7/8] Optimizing ensemble...\")\n\nif HAS_CB:\n    best_blend_f2 = 0\n    best_w = 1.0\n    best_thresh = 0.5\n\n    for w in np.arange(0.40, 1.01, 0.05):\n        oof_blend = w * oof_lgb + (1 - w) * oof_cb\n        for thresh in np.arange(0.05, 0.60, 0.005):\n            f2 = fbeta_score(y, (oof_blend > thresh).astype(int), beta=2)\n            if f2 > best_blend_f2:\n                best_blend_f2 = f2\n                best_w = w\n                best_thresh = thresh\n\n    print(f\"  Best: LGB={best_w:.2f}, CB={1-best_w:.2f}, F2={best_blend_f2:.6f}, thresh={best_thresh:.3f}\")\n\n    if best_lgb_f2 >= best_blend_f2:\n        print(f\"  LGB-only ({best_lgb_f2:.6f}) >= Blend ({best_blend_f2:.6f}), using LGB only\")\n        test_preds = test_lgb\n        best_f2 = 0\n        best_threshold = 0.5\n        for t in np.arange(0.05, 0.60, 0.001):\n            f2 = fbeta_score(y, (oof_lgb > t).astype(int), beta=2)\n            if f2 > best_f2:\n                best_f2 = f2\n                best_threshold = t\n    else:\n        print(f\"  Blend ({best_blend_f2:.6f}) > LGB ({best_lgb_f2:.6f}), using blend\")\n        test_preds = best_w * test_lgb + (1 - best_w) * test_cb\n        best_f2 = best_blend_f2\n        best_threshold = best_thresh\nelse:\n    test_preds = test_lgb\n    best_f2 = 0\n    best_threshold = 0.5\n    for t in np.arange(0.05, 0.60, 0.001):\n        f2 = fbeta_score(y, (oof_lgb > t).astype(int), beta=2)\n        if f2 > best_f2:\n            best_f2 = f2\n            best_threshold = t\n\nprint(f\"\\n  Final OOF F2: {best_f2:.6f}\")\nprint(f\"  Threshold: {best_threshold:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T21:33:29.213618Z","iopub.status.idle":"2026-04-05T21:33:29.213995Z","shell.execute_reply.started":"2026-04-05T21:33:29.213829Z","shell.execute_reply":"2026-04-05T21:33:29.213853Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 8. Generate Submission","metadata":{}},{"cell_type":"code","source":"print(\"[8/8] Generating submission...\")\n\nbinary_preds = (test_preds > best_threshold).astype(int)\nsubmission = pd.DataFrame({'Id': test_ids, 'Label': binary_preds})\nsubmission.to_csv(f'{OUTPUT_DIR}/submission.csv', index=False)\n\nprint(f\"  submission.csv saved to {OUTPUT_DIR}/\")\nprint(f\"  Threshold: {best_threshold:.4f}\")\nprint(f\"  OOF F2: {best_f2:.6f}\")\nprint(f\"  Anomaly fraction: {binary_preds.mean():.4f}\")\nprint(f\"  Total predictions: {len(binary_preds)}\")\nsubmission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T21:33:29.215860Z","iopub.status.idle":"2026-04-05T21:33:29.216362Z","shell.execute_reply.started":"2026-04-05T21:33:29.216098Z","shell.execute_reply":"2026-04-05T21:33:29.216138Z"}},"outputs":[],"execution_count":null}]}