{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ╔══════════════════════════════════════════════════════════════╗\n# ║  MULTI-VIEW STACKING — MALWARE CLASSIFICATION BIG2015    ║\n# ║                                                          ║\n# ║  Kiến trúc V3 (7 inputs → 9 Models):                     ║\n# ║    View 1: XGB  trên [107 tabular]                       ║\n# ║    View 2: LGB  trên [3000 byte n-gram]                  ║\n# ║    View 3: XGB  trên [2000 opcode n-gram]                ║\n# ║    View 4: XGB  trên [1024 pixel density]                ║\n# ║    View 5: XGB  trên [173 API/DLL counts]                ║\n# ║    View 6: XGB  trên [27 Section stats]                  ║\n# ║    View 7: LGB  trên [2000 Strings chi2]                 ║\n# ║    View 8: XGB  trên [ALL features]                      ║\n# ║    View 9: LGB  trên [ALL features]                      ║\n# ║         ↓ 81 OOF proba (9 views × 9 classes)             ║\n# ║    Meta-Feature Engineering: OOF proba + Entropy + Std   ║\n# ║         ↓ 126 Features (Level-1 Smart Input)             ║\n# ║    Level-1: Proper Nested CV (LR + XGB)                  ║\n# ║         ↓ SLSQP Optimize 4 Weights                       ║\n# ║    Final Submission Ensemble                             ║\n# ╚══════════════════════════════════════════════════════════════╝\n# ║                                                          ║\n# ║                                                          ║\n# ║    1. OOF-Calibrated Pseudo-Label Threshold              ║\n# ║    2. Class Weighting toàn pipeline (sample_weight XGB)  ║\n# ║    3. Cost Evaluation Report (Training/Inference/Memory) ║\n# ╚══════════════════════════════════════════════════════════════╝","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 1 — IMPORTS\n# ============================================================\nimport os\nimport gc\nimport time\nimport pickle\nimport joblib\nimport numpy as np\nimport scipy.sparse as sp\nimport lightgbm as lgb\nimport xgboost as xgb\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.ensemble import ExtraTreesClassifier\nfrom sklearn.metrics import log_loss, accuracy_score, precision_score, recall_score, f1_score, matthews_corrcoef, classification_report, confusion_matrix\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom sklearn.utils.class_weight import compute_sample_weight\n\nfrom scipy.optimize import minimize\n\nprint(\"Import xong!\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 2 — CẤU HÌNH ĐƯỜNG DẪN\n# ============================================================\n\nDATA_IN_DIR   = '/kaggle/input/datasets/trankimhuu/data-ml-big-2015-80-20'\n\nWORKING_DIR   = '/kaggle/working'\nCKPT_DIR      = os.path.join(DATA_IN_DIR, 'checkpoints')\n\nNUM_CLASSES    = 9\nN_FOLDS        = 5\nRANDOM_STATE   = 42\n\n# Tỷ lệ train trên toàn bộ dataset; tối đa 0.80 vì test holdout cố định 20%.\nMAX_TRAIN_RATIO = 0.80\nratio_train = 0.70  # Chỉnh nhanh: 0.80, 0.60, 0.40 hoặc 0.20\nTRAIN_SUBSET_RANDOM_STATE = RANDOM_STATE\n\nif not 0.0 < ratio_train <= MAX_TRAIN_RATIO:\n    raise ValueError(f\"ratio_train phải thuộc (0, {MAX_TRAIN_RATIO}], nhận được {ratio_train}\")\n\nratio_percent_tag = f\"{ratio_train * 100:.6f}\".rstrip('0').rstrip('.').replace('.', 'p')\nRUN_TAG = f\"train_{ratio_percent_tag}_test_20\"\nMODEL_DIR = os.path.join(WORKING_DIR, 'models', RUN_TAG)\n\nos.makedirs(MODEL_DIR, exist_ok=True)\nos.makedirs(CKPT_DIR,  exist_ok=True)\n\nprint(f\"Input : {DATA_IN_DIR}\")\nprint(f\"Run   : {RUN_TAG} (ratio_train={ratio_train:.2f}, fixed test=0.20)\")\nprint(f\"Models: {MODEL_DIR}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 3 — LOAD CÁC VIEW VÀO MEMORY\n# ============================================================\n\nprint(\"\\n\" + \"=\" * 55)\nprint(\"BƯỚC 1: LOAD CÁC VIEWS (Tổng cộng 9 models)\")\nprint(\"=\" * 55)\n\nVIEWS = {\n    \"tab\":      {\"file\": \"v_tab\",      \"desc\": \"Tabular (meta features)\"},\n    \"byteng\":   {\"file\": \"v_byteng\",   \"desc\": \"Byte N-gram (3000 chi2)\"},\n    \"opng\":     {\"file\": \"v_opng\",     \"desc\": \"Opcode N-gram (2000 chi2)\"},\n    \"pixel\":    {\"file\": \"v_pixel\",    \"desc\": \"Pixel density (32x32)\"},\n    \"api\":      {\"file\": \"v_api\",      \"desc\": \"API Calls + DLLs\"},\n    \"sections\": {\"file\": \"v_sections\", \"desc\": \"Section Statistics\"},\n    \"strings\":  {\"file\": \"v_strings\",  \"desc\": \"Printable Strings HashVec\"},\n    \"all\":      {\"file\": \"v_all\",      \"desc\": \"ALL features combined (XGB)\"},\n    \"all_lgb\":  {\"file\": \"v_all\",      \"desc\": \"ALL features combined (LGB)\"},\n}\n\nVIEW_FEATURE_MAP = {\n    \"tab\": \"tab\", \"byteng\": \"byteng\", \"opng\": \"opng\",\n    \"pixel\": \"pixel\", \"api\": \"api\", \"sections\": \"sections\", \"strings\": \"strings\",\n    \"all\": \"all\", \"all_lgb\": \"all\",\n}\n\nX_train_views = {}\nX_test_views  = {}\n\n# ── Load labels trước để tạo một bộ index chung cho mọi view ──\ny_train = np.load(os.path.join(CKPT_DIR, \"y_train.npy\"))\ny_test  = np.load(os.path.join(CKPT_DIR, \"y_test.npy\"))\nprint(f\"  {'labels':8s} : y_train {y_train.shape}  y_test {y_test.shape}\")\n\n# ── Rút gọn training pool 80%, giữ nguyên test holdout 20% ──\ndef select_nested_stratified_indices(labels, pool_fraction, random_state):\n    \"\"\"Chọn prefix cố định trong từng lớp để các tập train nhỏ lồng nhau.\"\"\"\n    if not 0.0 < pool_fraction <= 1.0:\n        raise ValueError(f\"pool_fraction phải thuộc (0, 1], nhận được {pool_fraction}\")\n\n    class_ids, class_counts = np.unique(labels, return_counts=True)\n    too_small = class_ids[class_counts < N_FOLDS]\n    if len(too_small) > 0:\n        raise ValueError(\n            f\"Các lớp {too_small.tolist()} không đủ mẫu cho {N_FOLDS}-fold CV.\"\n        )\n    if np.isclose(pool_fraction, 1.0):\n        return np.arange(len(labels), dtype=np.int64)\n\n    rng = np.random.default_rng(random_state)\n    selected_parts = []\n    for class_id in np.unique(labels):\n        class_indices = np.flatnonzero(labels == class_id)\n        shuffled_indices = rng.permutation(class_indices)\n        n_keep = int(np.round(len(class_indices) * pool_fraction))\n        if n_keep < N_FOLDS:\n            raise ValueError(\n                f\"Lớp {class_id} chỉ còn {n_keep} mẫu, không đủ cho {N_FOLDS}-fold CV.\"\n            )\n        selected_parts.append(shuffled_indices[:n_keep])\n\n    return np.sort(np.concatenate(selected_parts)).astype(np.int64)\n\nn_train_pool = len(y_train)\nn_test_holdout = len(y_test)\nn_total_holdout_dataset = n_train_pool + n_test_holdout\noriginal_train_ratio = n_train_pool / n_total_holdout_dataset\nif not np.isclose(original_train_ratio, MAX_TRAIN_RATIO, atol=0.01):\n    raise ValueError(\n        f\"Checkpoint không phải split 80/20: train ratio thực tế={original_train_ratio:.4f}.\"\n    )\n\ntrain_pool_fraction = ratio_train / MAX_TRAIN_RATIO\nselected_train_idx = select_nested_stratified_indices(\n    y_train, train_pool_fraction, TRAIN_SUBSET_RANDOM_STATE\n)\nuse_full_train_pool = len(selected_train_idx) == n_train_pool\n\n# Load từng view và cắt ngay để không giữ đồng thời toàn bộ các train views trong RAM.\nfor vname in set(VIEW_FEATURE_MAP.values()):  # set để không load 'all' 2 lần\n    X_tr = np.load(os.path.join(CKPT_DIR, f\"X_train_v_{vname}.npy\"))\n    X_te = np.load(os.path.join(CKPT_DIR, f\"X_test_v_{vname}.npy\"))\n    if len(X_tr) != n_train_pool:\n        raise ValueError(\n            f\"View {vname} có {len(X_tr)} mẫu train, khác y_train={n_train_pool}.\"\n        )\n    if len(X_te) != n_test_holdout:\n        raise ValueError(\n            f\"View {vname} có {len(X_te)} mẫu test, khác y_test={n_test_holdout}.\"\n        )\n    X_train_views[vname] = X_tr if use_full_train_pool else X_tr[selected_train_idx]\n    X_test_views[vname] = X_te\n    print(f\"  {vname:8s} : train {X_train_views[vname].shape}  test {X_te.shape}\")\n    del X_tr, X_te\n\nif not use_full_train_pool:\n    y_train = y_train[selected_train_idx]\nactual_train_ratio = len(y_train) / n_total_holdout_dataset\n\nfor vname, X_view in X_train_views.items():\n    assert len(X_view) == len(y_train), f\"Sai lệch số mẫu ở view {vname}\"\n\nclass_ids, class_counts = np.unique(y_train, return_counts=True)\nprint(\"\\n  TRAIN SUBSET\")\nprint(f\"  Requested ratio : {ratio_train:.2f} of full dataset\")\nprint(f\"  Pool fraction   : {train_pool_fraction:.4f} of the original 80% train pool\")\nprint(f\"  Samples         : {len(y_train):,} / {n_train_pool:,}\")\nprint(f\"  Actual ratio    : {actual_train_ratio:.4f} of full dataset\")\nprint(f\"  Fixed test      : {n_test_holdout:,} samples (unchanged)\")\nprint(f\"  Class counts    : {dict(zip(class_ids.tolist(), class_counts.tolist()))}\")\n\nselected_idx_path = os.path.join(WORKING_DIR, f\"selected_train_idx_{RUN_TAG}.npy\")\nnp.save(selected_idx_path, selected_train_idx)\nprint(f\"  Saved indices   : {selected_idx_path}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 4 — CẤU HÌNH VÀ TRAIN MULTI-VIEW MODELS\n# ============================================================\n\nprint(\"\\n\" + \"=\" * 55)\nprint(\"BƯỚC 2: CẤU HÌNH 9 MULTI-VIEW MODELS\")\nprint(\"=\" * 55)\n\nclasses = np.unique(y_train)\nweights = compute_class_weight(\"balanced\", classes=classes, y=y_train)\nclass_weights_dict = dict(zip(classes.astype(int), weights))\n\nVIEW_MODELS = {\n    \"tab\": xgb.XGBClassifier(\n        objective='multi:softprob', num_class=NUM_CLASSES,\n        max_depth=8, n_estimators=500, learning_rate=0.05,\n        subsample=0.8, colsample_bytree=0.9, min_child_weight=3,\n        tree_method='hist', device='cuda', eval_metric='mlogloss', random_state=RANDOM_STATE,\n    ),\n    \"byteng\": lgb.LGBMClassifier(\n        objective='multiclass', num_class=NUM_CLASSES, class_weight=class_weights_dict,\n        n_estimators=800, learning_rate=0.05, num_leaves=127, min_child_samples=10,\n        feature_fraction=0.5, bagging_fraction=0.8, bagging_freq=5,\n        lambda_l1=0.1, lambda_l2=0.1, device='gpu', random_state=RANDOM_STATE, verbose=-1,\n    ),\n    \"opng\": xgb.XGBClassifier(\n        objective='multi:softprob', num_class=NUM_CLASSES,\n        max_depth=7, n_estimators=500, learning_rate=0.05,\n        subsample=0.8, colsample_bytree=0.6, min_child_weight=3,\n        tree_method='hist', device='cuda', eval_metric='mlogloss', random_state=RANDOM_STATE,\n    ),\n    \"pixel\": xgb.XGBClassifier(\n        objective='multi:softprob', num_class=NUM_CLASSES,\n        max_depth=7, n_estimators=500, learning_rate=0.05,\n        subsample=0.8, colsample_bytree=0.7, min_child_weight=3,\n        tree_method='hist', device='cuda', eval_metric='mlogloss', random_state=RANDOM_STATE,\n    ),\n    \"api\": xgb.XGBClassifier(\n        objective='multi:softprob', num_class=NUM_CLASSES,\n        max_depth=7, n_estimators=600, learning_rate=0.05,\n        subsample=0.8, colsample_bytree=0.8, min_child_weight=3,\n        tree_method='hist', device='cuda', eval_metric='mlogloss', random_state=RANDOM_STATE,\n    ),\n    \"sections\": xgb.XGBClassifier(\n        objective='multi:softprob', num_class=NUM_CLASSES,\n        max_depth=6, n_estimators=400, learning_rate=0.05,\n        subsample=0.8, colsample_bytree=0.9, min_child_weight=3,\n        tree_method='hist', device='cuda', eval_metric='mlogloss', random_state=RANDOM_STATE,\n    ),\n    \"strings\": lgb.LGBMClassifier(\n        objective='multiclass', num_class=NUM_CLASSES, class_weight=class_weights_dict,\n        n_estimators=800, learning_rate=0.05, num_leaves=127, min_child_samples=10,\n        feature_fraction=0.5, bagging_fraction=0.8, bagging_freq=5,\n        lambda_l1=0.1, lambda_l2=0.1, device='gpu', random_state=RANDOM_STATE, verbose=-1,\n    ),\n    \"all\": xgb.XGBClassifier(\n        objective='multi:softprob', num_class=NUM_CLASSES,\n        max_depth=7, n_estimators=500, learning_rate=0.05,\n        subsample=0.8, colsample_bytree=0.7, min_child_weight=3,\n        tree_method='hist', device='cuda', eval_metric='mlogloss', random_state=RANDOM_STATE,\n    ),\n    \"all_lgb\": lgb.LGBMClassifier(\n        objective='multiclass', num_class=NUM_CLASSES, class_weight=class_weights_dict,\n        n_estimators=1000, learning_rate=0.05, num_leaves=127, min_child_samples=10,\n        feature_fraction=0.7, bagging_fraction=0.8, bagging_freq=5,\n        lambda_l1=0.1, lambda_l2=0.1, device='gpu', random_state=RANDOM_STATE, verbose=-1,\n    ),\n}\n\nskf = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=2026)\nN_VIEWS = len(VIEW_MODELS)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 5 — LEVEL-0 TRAINING (OOF META-FEATURES)\n# ============================================================\n\nprint(\"\\n\" + \"=\" * 55)\nprint(\"BƯỚC 3: MULTI-VIEW LEVEL-0 (TẠO OOF)\")\nprint(\"=\" * 55)\n\nimport glob\nold_ckpts = glob.glob(os.path.join(CKPT_DIR, \"oof_train_*.npy\")) + glob.glob(os.path.join(CKPT_DIR, \"oof_test_*.npy\"))\nfor f in old_ckpts:\n    os.remove(f)\n\nmeta_train_list = []\nmeta_test_list  = []\nn_train = len(y_train)\nn_test  = X_test_views[\"tab\"].shape[0]\n\n# === Lưu thời gian training từng view để báo cáo ===\nlevel0_times = {}\n\nfor view_name, model in VIEW_MODELS.items():\n    feat_key = VIEW_FEATURE_MAP[view_name]\n    X_tr = X_train_views[feat_key]\n    X_te = X_test_views[feat_key]\n    \n    print(f\"\\n{'─'*50}\")\n    print(f\"[{view_name.upper()}] features={X_tr.shape[1]}\")\n\n    oof_train    = np.zeros((n_train, NUM_CLASSES), dtype=np.float32)\n    oof_test_avg = np.zeros((n_test,  NUM_CLASSES), dtype=np.float32)\n\n    view_t0 = time.time()\n    for fold, (tr_idx, val_idx) in enumerate(skf.split(X_tr, y_train)):\n        print(f\"  Fold {fold+1}/{N_FOLDS}...\", end=\"\", flush=True)\n        t0 = time.time()\n\n        # === [CLASS IMBALANCE] sample_weight cho XGB ===\n        if isinstance(model, xgb.XGBClassifier):\n            sw = compute_sample_weight(\"balanced\", y_train[tr_idx])\n            model.fit(X_tr[tr_idx], y_train[tr_idx], sample_weight=sw)\n        else:\n            # LGB đã có class_weight trong config\n            model.fit(X_tr[tr_idx], y_train[tr_idx])\n\n        oof_train[val_idx] = model.predict_proba(X_tr[val_idx])\n        oof_test_avg += model.predict_proba(X_te) / N_FOLDS\n        print(f\" ll={log_loss(y_train[val_idx], oof_train[val_idx]):.5f}  ({time.time()-t0:.1f}s)\")\n        gc.collect()\n\n    print(f\"   OOF logloss: {log_loss(y_train, oof_train):.5f}\")\n    \n    meta_train_list.append(oof_train)\n    meta_test_list.append(oof_test_avg)\n\n    # Full fit\n    print(f\"  Full model...\", end=\"\", flush=True)\n    t0 = time.time()\n    if isinstance(model, xgb.XGBClassifier):\n        sw_full = compute_sample_weight(\"balanced\", y_train)\n        model.fit(X_tr, y_train, sample_weight=sw_full)\n    else:\n        model.fit(X_tr, y_train)\n    joblib.dump(model, os.path.join(MODEL_DIR, f\"{view_name}_full.pkl\"))\n    print(f\" ({time.time()-t0:.1f}s)\")\n    level0_times[view_name] = time.time() - view_t0\n\nX_train_L1 = np.hstack(meta_train_list).astype(np.float32)\nX_test_L1  = np.hstack(meta_test_list).astype(np.float32)\n\nprint(f\"\\n Level-0 done! X_train_L1: {X_train_L1.shape} (9 models × 9 classes)\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 6 — PSEUDO-LABELING (OOF-CALIBRATED THRESHOLD)\n# ============================================================\n\nprint(\"\\n\" + \"=\" * 55)\nprint(\"BƯỚC 4: PSEUDO-LABELING (OOF-Calibrated Threshold)\")\nprint(\"=\" * 55)\n\n# === Dùng OOF predictions (có ground truth) để calibrate threshold ===\ndef oof_calibrated_threshold(oof_proba, y_true, base_percentile=95, min_threshold=0.90):\n    \"\"\"\n    Xác định threshold cho từng class dựa trên phân phối confidence\n    của các OOF predictions ĐÚNG (có ground truth).\n    \n    Logic:\n      - Với mỗi class c, lấy tất cả mẫu OOF mà model predict đúng class c\n      - Tính percentile thứ `base_percentile` của confidence khi predict đúng\n      - Class nào model mạnh (confidence cao khi đúng) → threshold tự cao\n      - Class nào model yếu (confidence thấp khi đúng) → threshold tự thấp\n    \n    Hoàn toàn data-driven, không có magic numbers.\n    \"\"\"\n    num_classes = oof_proba.shape[1]\n    thresholds = {}\n    \n    for c in range(num_classes):\n        # Mẫu mà model predict đúng class c\n        pred_c = oof_proba.argmax(axis=1) == c\n        true_c = y_true == c\n        correct_mask = pred_c & true_c\n        \n        if correct_mask.sum() == 0:\n            thresholds[c] = 0.99  # fallback nếu không có mẫu đúng\n            continue\n        \n        # Confidence khi predict đúng class c\n        correct_confs = oof_proba[correct_mask, c]\n        \n        # Threshold = percentile của confidence khi predict đúng\n        thresholds[c] = max(min_threshold, np.percentile(correct_confs, base_percentile))\n    \n    return thresholds\n\n# Tính OOF average từ X_train_L1 (9 models × 9 classes → average across models)\noof_avg = X_train_L1.reshape(len(y_train), N_VIEWS, NUM_CLASSES).mean(axis=1)\n\n# Calibrate threshold từ OOF (Dùng cho Microsoft Test later)\nca_thresholds = oof_calibrated_threshold(oof_avg, y_train, base_percentile=90)\nprint(f\"\\n  📊 OOF-CALIBRATED THRESHOLDS (base_percentile=90):\")\nprint(f\"  {'Class':<8} {'Train':>8} {'OOF Acc':>10} {'Threshold':>10}\")\nprint(f\"  {'─'*8} {'─'*8} {'─'*10} {'─'*10}\")\nfor c in range(NUM_CLASSES):\n    n_train = np.sum(y_train == c)\n    true_c = y_train == c\n    pred_c = oof_avg.argmax(axis=1) == c\n    oof_acc = np.sum(true_c & pred_c) / max(np.sum(true_c), 1)\n    print(f\"  {c:<8} {n_train:>8} {oof_acc:>10.4f} {ca_thresholds[c]:>10.4f}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 7 — LEVEL-1: NESTED CV + GIA TĂNG META-FEATURES \n# ============================================================\n\nprint(\"\\n\" + \"=\" * 55)\nprint(\"BƯỚC 5: LEVEL-1 (META-FEATURE ENGINEERING & TỐI ƯU TRỌNG SỐ)\")\nprint(\"=\" * 55)\n\ndef compute_meta_features(X_L1, num_models=9, num_classes=9):\n    N = X_L1.shape[0]\n    proba_3d = X_L1.reshape(N, num_models, num_classes)\n    \n    feat_mean = np.mean(proba_3d, axis=1)\n    feat_max  = np.max(proba_3d, axis=1)\n    feat_min  = np.min(proba_3d, axis=1)\n    feat_std  = np.std(proba_3d, axis=1)\n    \n    eps = 1e-15\n    proba_clip = np.clip(proba_3d, eps, 1 - eps)\n    entropy_per_model = -np.sum(proba_clip * np.log(proba_clip), axis=2)\n    \n    X_ext = np.hstack([X_L1, feat_mean, feat_max, feat_min, feat_std, entropy_per_model])\n    return X_ext.astype(np.float32)\n\nprint(\"  Đang tính toán Bất đồng (Std) và Vùng hỗn loạn (Entropy)...\", end=\"\", flush=True)\nX_train_L1_smart = compute_meta_features(X_train_L1, num_models=N_VIEWS)\nX_test_L1_smart  = compute_meta_features(X_test_L1,  num_models=N_VIEWS)\nprint(f\" ✓ ({X_train_L1.shape[1]} -> {X_train_L1_smart.shape[1]} đặc trưng)\")\n\noof_lr  = np.zeros((len(y_train), NUM_CLASSES), dtype=np.float32)\noof_xgb = np.zeros((len(y_train), NUM_CLASSES), dtype=np.float32)\n\nskf_nested = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=2026)\n\nlevel1_t0 = time.time()\nprint(f\"\\n── Tạo PROPER OOF Level-1 (nested {N_FOLDS}-fold) ──\")\nfor fold, (tr_idx, val_idx) in enumerate(skf_nested.split(X_train_L1_smart, y_train)):\n    # 1. Logistic Regression — thêm class_weight='balanced'\n    scaler_fold_l1 = StandardScaler()\n    X_tr_l1_s = scaler_fold_l1.fit_transform(X_train_L1_smart[tr_idx])\n    X_val_l1_s = scaler_fold_l1.transform(X_train_L1_smart[val_idx])\n    lr_fold = LogisticRegression(\n        C=0.1, solver='lbfgs', max_iter=2000,\n        multi_class='multinomial', class_weight='balanced',\n        random_state=RANDOM_STATE\n    )\n    lr_fold.fit(X_tr_l1_s, y_train[tr_idx])\n    oof_lr[val_idx] = lr_fold.predict_proba(X_val_l1_s)\n\n    # 2. XGB — Dùng 126 meta-features\n    xgb_fold = xgb.XGBClassifier(\n        objective='multi:softprob', num_class=NUM_CLASSES,\n        max_depth=3, n_estimators=200, learning_rate=0.05,\n        subsample=0.8, tree_method='hist', device='cuda', random_state=99\n    )\n    sw_l1 = compute_sample_weight(\"balanced\", y_train[tr_idx])\n    xgb_fold.fit(X_tr_l1_s, y_train[tr_idx], sample_weight=sw_l1)\n    oof_xgb[val_idx] = xgb_fold.predict_proba(X_val_l1_s)\n\n\n    \n    print(f\"  Fold {fold+1} ✓\")\n\nlevel1_cv_time = time.time() - level1_t0\n\nll_lr_proper  = log_loss(y_train, oof_lr)\nll_xgb_proper = log_loss(y_train, oof_xgb)\n\nprint(f\"   OOF LR   : {ll_lr_proper:.5f}\")\nprint(f\"   OOF XGB  : {ll_xgb_proper:.5f}\")\n\ndef ensemble_logloss(w):\n    blended = w[0] * oof_lr + w[1] * oof_xgb\n    blended = np.clip(blended, 1e-15, 1 - 1e-15)\n    blended = blended / blended.sum(axis=1, keepdims=True)\n    return log_loss(y_train, blended)\n\ncons = ({'type': 'eq', 'fun': lambda w: 1 - sum(w)})\nbounds = [(0.0, 1.0)] * 2\nresult = minimize(ensemble_logloss, x0=[0.5, 0.5], bounds=bounds, constraints=cons, method='SLSQP')\nw_lr_opt, w_xgb_opt = result.x\nll_opt = result.fun\n\nprint(f\"\\n   Optimal weights: w_LR={w_lr_opt:.4f}, w_XGB={w_xgb_opt:.4f} → Logloss: {ll_opt:.6f}\")\n\n\n# ── FINAL FIT ──\nprint(f\"\\n  Fit Final Meta-Models on Train Ratio {ratio_train:.0%} (Fixed Test 20%)...\")\nscaler_l1 = StandardScaler()\nX_train_L1_s = scaler_l1.fit_transform(X_train_L1_smart)\nX_test_L1_s  = scaler_l1.transform(X_test_L1_smart)\nmeta_lr = LogisticRegression(\n    C=0.1, solver='lbfgs', max_iter=2000,\n    multi_class='multinomial', class_weight='balanced',\n    random_state=RANDOM_STATE\n)\nmeta_lr.fit(X_train_L1_s, y_train)\n\nsw_l1_orig = compute_sample_weight(\"balanced\", y_train)\nmeta_xgb = xgb.XGBClassifier(\n    objective='multi:softprob', num_class=NUM_CLASSES,\n    max_depth=3, n_estimators=200, learning_rate=0.05,\n    subsample=0.8, tree_method='hist', device='cuda', random_state=99\n)\nmeta_xgb.fit(X_train_L1_s, y_train, sample_weight=sw_l1_orig)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 8 — ĐÁNH GIÁ CHI PHÍ TRAINING & INFERENCE\n# ============================================================\n\nprint(\"\\n\" + \"=\" * 55)\nprint(\"BƯỚC 6: ĐÁNH GIÁ CHI PHÍ -> Xuất file report\")\nprint(\"=\" * 55)\n\n# ── 1. Model Size ──\nmodel_sizes = {}\ntotal_model_size_mb = 0\nfor view_name in VIEW_MODELS.keys():\n    model_path = os.path.join(MODEL_DIR, f\"{view_name}_full.pkl\")\n    if os.path.exists(model_path):\n        size_mb = os.path.getsize(model_path) / (1024 * 1024)\n        model_sizes[view_name] = size_mb\n        total_model_size_mb += size_mb\n\n# ── 2. Training Time ──\ntotal_l0_time = sum(level0_times.values())\ntotal_pipeline_time = total_l0_time + level1_cv_time\n\n# ── 3. Inference Time ──\nt_inf_start = time.time()\n\nt_l0_inf = time.time()\nfor view_name in VIEW_MODELS.keys():\n    feat_key = VIEW_FEATURE_MAP[view_name]\n    m = joblib.load(os.path.join(MODEL_DIR, f\"{view_name}_full.pkl\"))\n    _ = m.predict_proba(X_test_views[feat_key])\n    del m\nt_l0_inf = time.time() - t_l0_inf\n\nt_l1_inf = time.time()\n_ = meta_lr.predict_proba(X_test_L1_s)\n_ = meta_xgb.predict_proba(X_test_L1_s)\nt_l1_inf = time.time() - t_l1_inf\n\nt_inference = time.time() - t_inf_start\nn_test_samples = len(X_test_L1)\n\n# ── 4. Memory Usage ──\nmem_rss_gb = None\nmem_vms_gb = None\ntry:\n    import psutil\n    process = psutil.Process()\n    mem_info = process.memory_info()\n    mem_rss_gb = mem_info.rss / (1024**3)\n    mem_vms_gb = mem_info.vms / (1024**3)\nexcept ImportError:\n    pass\n\n# ══════════════════════════════════════════════════════════════\n# TẠO FILE REPORT\n# ══════════════════════════════════════════════════════════════\nreport_path = os.path.join(WORKING_DIR, f'cost_evaluation_report_{RUN_TAG}.md')\nlines = []\n\nlines.append(\"# Cost Evaluation Report — Multi-view Stacking V3\")\nlines.append(\"\")\nlines.append(f\"**Pipeline**: 9 Level-0 models (XGB+LGB) → Meta-Feature Engineering → 2 Level-1 models → SLSQP Ensemble\")\nlines.append(f\"**Dataset**: {len(y_train):,} train / {n_test_samples:,} test — {NUM_CLASSES} classes\")\nlines.append(f\"**Hardware**: Kaggle T4 GPU\")\nlines.append(\"\")\n\n# Model Sizes\nlines.append(\"## 1. Model Sizes (Level-0)\")\nlines.append(\"\")\nlines.append(\"| Model | Size (MB) |\")\nlines.append(\"|---|---:|\")\nfor name, size in model_sizes.items():\n    lines.append(f\"| {name} | {size:.2f} |\")\nlines.append(f\"| **TOTAL** | **{total_model_size_mb:.2f}** |\")\nlines.append(\"\")\n\n# Training Time\nlines.append(\"## 2. Training Time\")\nlines.append(\"\")\nlines.append(\"### Level-0 (9 models x 5-fold CV + full fit)\")\nlines.append(\"\")\nlines.append(\"| View | Time (s) | % of L-0 |\")\nlines.append(\"|---|---:|---:|\")\nfor name, t in level0_times.items():\n    lines.append(f\"| {name} | {t:.1f} | {100*t/total_l0_time:.1f}% |\")\nlines.append(f\"| **TOTAL Level-0** | **{total_l0_time:.1f}** | **{total_l0_time/60:.1f} min** |\")\nlines.append(\"\")\n\nlines.append(\"### Level-1 (nested 5-fold CV + weight optimization)\")\nlines.append(\"\")\nlines.append(f\"| Phase | Time (s) |\")\nlines.append(f\"|---|---:|\")\nlines.append(f\"| Level-1 CV + Optimize | {level1_cv_time:.1f} |\")\nlines.append(\"\")\n\nlines.append(\"### Total Pipeline\")\nlines.append(\"\")\nlines.append(\"| Phase | Time (s) | % |\")\nlines.append(\"|---|---:|---:|\")\nlines.append(f\"| Level-0 | {total_l0_time:.1f} | {100*total_l0_time/total_pipeline_time:.1f}% |\")\nlines.append(f\"| Level-1 | {level1_cv_time:.1f} | {100*level1_cv_time/total_pipeline_time:.1f}% |\")\nlines.append(f\"| **TOTAL** | **{total_pipeline_time:.1f}** | **{total_pipeline_time/60:.1f} min** |\")\nlines.append(\"\")\n\n# Inference Time\nlines.append(\"## 3. Inference Time\")\nlines.append(\"\")\nlines.append(f\"| Phase | Time (s) |\")\nlines.append(f\"|---|---:|\")\nlines.append(f\"| Level-0 (9 models) | {t_l0_inf:.2f} |\")\nlines.append(f\"| Level-1 (2 meta-models) | {t_l1_inf:.2f} |\")\nlines.append(f\"| **Total ({n_test_samples:,} samples)** | **{t_inference:.2f}** |\")\nlines.append(f\"| **Per sample** | **{1000*t_inference/n_test_samples:.3f} ms** |\")\nlines.append(\"\")\n\n# Memory\nlines.append(\"## 4. Memory Usage\")\nlines.append(\"\")\nif mem_rss_gb is not None:\n    lines.append(f\"| Metric | Value |\")\n    lines.append(f\"|---|---:|\")\n    lines.append(f\"| RSS (Resident) | {mem_rss_gb:.2f} GB |\")\n    lines.append(f\"| VMS (Virtual) | {mem_vms_gb:.2f} GB |\")\nelse:\n    lines.append(\"*psutil not available — memory report skipped*\")\nlines.append(\"\")\n\n# Feature Dimensions\nlines.append(\"## 5. Feature Dimensions\")\nlines.append(\"\")\nlines.append(\"| View | Features | Model Type |\")\nlines.append(\"|---|---:|---|\")\nview_info = [\n    (\"tab\", 107, \"XGB\"), (\"byteng\", 3000, \"LGB\"), (\"opng\", 2000, \"XGB\"),\n    (\"pixel\", 1024, \"XGB\"), (\"api\", 173, \"XGB\"), (\"sections\", 27, \"XGB\"),\n    (\"strings\", 2000, \"LGB\"), (\"all (XGB)\", 8331, \"XGB\"), (\"all (LGB)\", 8331, \"LGB\"),\n]\nfor name, dim, model_type in view_info:\n    lines.append(f\"| {name} | {dim:,} | {model_type} |\")\nlines.append(f\"| **Unique total** | **8,331** | |\")\nlines.append(\"\")\n\n# DL Comparison\nlines.append(\"## 6. So sanh voi Deep Learning\")\nlines.append(\"\")\nlines.append(\"| Metric | Our ML Pipeline | Typical DL (CNN/LSTM) |\")\nlines.append(\"|---|---|---|\")\nlines.append(f\"| Training Time | {total_pipeline_time/60:.0f} min | 3-6 hours |\")\nlines.append(f\"| GPU Requirement | Optional (accelerates) | Mandatory |\")\nlines.append(f\"| Total Model Size | {total_model_size_mb:.0f} MB | 200-500 MB |\")\nlines.append(f\"| Inference (per sample) | {1000*t_inference/n_test_samples:.2f} ms | 10-50 ms |\")\nlines.append(f\"| Framework | sklearn + XGB/LGB | PyTorch/TensorFlow |\")\nlines.append(f\"| Interpretability | High (SHAP ready) | Low (black box) |\")\nlines.append(f\"| Hyperparameter Tuning | Moderate | Extensive |\")\nlines.append(\"\")\n\n# Write file\nreport_content = '\\n'.join(lines)\nwith open(report_path, 'w', encoding='utf-8') as f:\n    f.write(report_content)\n\nprint(f\"\\n  Report saved to: {report_path}\")\nprint(f\"  Model size total: {total_model_size_mb:.2f} MB\")\nprint(f\"  Training time:    {total_pipeline_time/60:.1f} min\")\nprint(f\"  Inference:        {1000*t_inference/n_test_samples:.3f} ms/sample\")\nif mem_rss_gb:\n    print(f\"  Memory RSS:       {mem_rss_gb:.2f} GB\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 9 — TẠO SUBMISSION\n# ============================================================\n\nprint(\"\\n\" + \"=\" * 55)\nprint(\"BƯỚC 7: TẠO SUBMISSIONS\")\nprint(\"=\" * 55)\n\nproba_lr_test  = meta_lr.predict_proba(X_test_L1_s)\nproba_xgb_test = meta_xgb.predict_proba(X_test_L1_s)\n\nproba_ens_opt = w_lr_opt * proba_lr_test + w_xgb_opt * proba_xgb_test\nproba_ens_opt = np.clip(proba_ens_opt, 1e-15, 1 - 1e-15)\nproba_ens_opt = proba_ens_opt / proba_ens_opt.sum(axis=1, keepdims=True)\n\n# ── Direct Level-0 average ───────────────────────────────────\nproba_direct_avg = X_test_L1.reshape(X_test_L1.shape[0], -1, NUM_CLASSES).mean(axis=1)\n\n# ── Đánh giá từng strategy ───────────────────────────────────\nstrategies = {\n    \"Ensemble Optimized\": proba_ens_opt,\n    \"LR Only\":            proba_lr_test,\n    \"XGB Only\":           proba_xgb_test,\n    \"Direct L0 Average\":  proba_direct_avg,\n}\n\nprint(f\"\\n{'Strategy':<22} {'Accuracy':>9} {'Weighted Precision':>20} {'Weighted Recall':>17} {'Weighted F1':>13} {'MCC':>7} {'LogLoss':>9}\")\nprint(\"─\" * 108)\n\nbest_strategy = None\nbest_weighted_f1 = -1\n\n# Chuẩn bị nội dung file markdown\neval_report_lines = [\n    \"# Model Evaluation Report\",\n    \"\",\n    \"## 1. Summary of Strategies\",\n    \"\",\n    \"| Strategy | Accuracy | Weighted Precision | Weighted Recall | Weighted F1 Score | MCC | Log Loss |\",\n    \"|---|---|---|---|---|---|---|\"\n]\n\nfor name, proba in strategies.items():\n    proba = np.clip(proba, 1e-15, 1 - 1e-15)\n    proba = proba / proba.sum(axis=1, keepdims=True)\n    y_pred = proba.argmax(axis=1)\n\n    acc  = accuracy_score(y_test, y_pred)\n    weighted_precision = precision_score(y_test, y_pred, average='weighted', zero_division=0)\n    weighted_recall = recall_score(y_test, y_pred, average='weighted', zero_division=0)\n    weighted_f1 = f1_score(y_test, y_pred, average='weighted', zero_division=0)\n    mcc  = matthews_corrcoef(y_test, y_pred)\n    ll   = log_loss(y_test, proba)\n\n    print(f\"{name:<22} {acc:>9.4f} {weighted_precision:>20.4f} {weighted_recall:>17.4f} {weighted_f1:>13.4f} {mcc:>7.4f} {ll:>9.5f}\")\n    \n    # Thêm vào report\n    eval_report_lines.append(f\"| {name} | {acc:.4f} | {weighted_precision:.4f} | {weighted_recall:.4f} | {weighted_f1:.4f} | {mcc:.4f} | {ll:.5f} |\")\n\n    if weighted_f1 > best_weighted_f1:\n        best_weighted_f1 = weighted_f1\n        best_strategy = name\n\n# ── Classification Report cho Ensemble ─────────────\nensemble_name = \"Ensemble Optimized\"\nensemble_proba = strategies[ensemble_name]\nensemble_proba = np.clip(ensemble_proba, 1e-15, 1 - 1e-15)\nensemble_proba = ensemble_proba / ensemble_proba.sum(axis=1, keepdims=True)\ny_pred_ensemble = ensemble_proba.argmax(axis=1)\n\nprint(f\"\\n{'='*55}\")\nprint(f\"CLASSIFICATION REPORT — {ensemble_name}\")\nprint(\"=\"*55)\nclass_report_str = classification_report(y_test, y_pred_ensemble, digits=4)\nprint(class_report_str)\n\nprint(\"CONFUSION MATRIX:\")\ncm = confusion_matrix(y_test, y_pred_ensemble)\nprint(cm)\n\n# Thêm chi tiết vào report\neval_report_lines.extend([\n    \"\",\n    f\"## 2. Detailed Report: {ensemble_name}\",\n    \"\",\n    \"### Classification Report\",\n    \"```text\",\n    class_report_str,\n    \"```\",\n    \"\",\n    \"### Confusion Matrix\",\n    \"```text\",\n    str(cm),\n    \"```\"\n])\n\n# Ghi file\neval_report_path = os.path.join(WORKING_DIR, f'evaluation_report_{RUN_TAG}.md')\nwith open(eval_report_path, 'w', encoding='utf-8') as f:\n    f.write('\\n'.join(eval_report_lines))\n\nprint(\"\\n\" + \"=\" * 55)\nprint(\" HOÀN TẤT PIPELINE EVALUATION!\")\nprint(\"=\" * 55)\nprint(f\"   Kết quả tốt nhất: {best_strategy} (Weighted F1={best_weighted_f1:.4f})\")\nprint(f\"   File report đã lưu tại: {eval_report_path}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 10 — TẠO SUBMISSION CHO MICROSOFT TEST SET\n# ============================================================\n\nprint(\"\\n\" + \"=\" * 55)\nprint(\"BƯỚC 8: TẠO SUBMISSION — MICROSOFT OFFICIAL TEST SET\")\nprint(\"=\" * 55)\n\nSUBMISSION_CSV = '/kaggle/input/competitions/malware-classification/sampleSubmission.csv'\n\n# ── Load Microsoft test views ──\nX_ms_views = {}\nms_available = True\nfor vname in set(VIEW_FEATURE_MAP.values()):\n    ms_path = os.path.join(CKPT_DIR, f\"X_mstest_v_{vname}.npy\")\n    if os.path.exists(ms_path):\n        X_ms_views[vname] = np.load(ms_path)\n        print(f\"  {vname:8s} : {X_ms_views[vname].shape}\")\n    else:\n        print(f\"  [WARNING] Missing {ms_path}\")\n        ms_available = False\n\nif ms_available and os.path.exists(SUBMISSION_CSV):\n    n_ms = X_ms_views[\"tab\"].shape[0]\n\n    # ── Level-0: Predict trên Microsoft test ──\n    ms_meta_list = []\n    for view_name in VIEW_MODELS.keys():\n        feat_key = VIEW_FEATURE_MAP[view_name]\n        m = joblib.load(os.path.join(MODEL_DIR, f\"{view_name}_full.pkl\"))\n        ms_meta_list.append(m.predict_proba(X_ms_views[feat_key]))\n        del m; gc.collect()\n\n    X_ms_L1 = np.hstack(ms_meta_list).astype(np.float32)\n    del ms_meta_list; gc.collect()\n\n    # ── Meta-features ──\n    X_ms_L1_smart = compute_meta_features(X_ms_L1, num_models=N_VIEWS)\n\n    # ── Pseudo-Labeling trên Microsoft Test Set ──\n    print(\"\\n  Applying Pseudo-Labeling on Microsoft Test Set...\")\n    ms_avg_proba = X_ms_L1.reshape(-1, N_VIEWS, NUM_CLASSES).mean(axis=1)\n    ms_pred_classes = ms_avg_proba.argmax(axis=1)\n    ms_max_conf = ms_avg_proba.max(axis=1)\n    \n    ms_confident_mask = np.zeros(len(ms_avg_proba), dtype=bool)\n    for c in range(NUM_CLASSES):\n        class_mask = (ms_pred_classes == c) & (ms_max_conf >= ca_thresholds[c])\n        ms_confident_mask |= class_mask\n\n    ms_confident_idx = np.where(ms_confident_mask)[0]\n    ms_pseudo_labels = ms_pred_classes[ms_confident_idx]\n    print(f\"  Found {len(ms_confident_idx):,} confident pseudo-labels ({100*len(ms_confident_idx)/n_ms:.1f}%)\")\n\n    # Mở rộng tập Train\n    X_train_L1_ms_ext = np.vstack([X_train_L1_smart, X_ms_L1_smart[ms_confident_idx]])\n    y_train_ms_ext = np.concatenate([y_train, ms_pseudo_labels])\n\n    # Retrain Meta-Models\n    print(\"  Retraining Meta-Models with Pseudo-Labels...\")\n    scaler_l1_ms = StandardScaler()\n    X_train_L1_ms_ext_s = scaler_l1_ms.fit_transform(X_train_L1_ms_ext)\n    meta_lr_ms = LogisticRegression(\n        C=0.1, solver='lbfgs', max_iter=2000, multi_class='multinomial', class_weight='balanced', random_state=RANDOM_STATE\n    )\n    meta_lr_ms.fit(X_train_L1_ms_ext_s, y_train_ms_ext)\n\n    sw_ms_ext = compute_sample_weight(\"balanced\", y_train_ms_ext)\n    meta_xgb_ms = xgb.XGBClassifier(\n        objective='multi:softprob', num_class=NUM_CLASSES, max_depth=3, n_estimators=200, \n        learning_rate=0.05, subsample=0.8, tree_method='hist', device='cuda', random_state=99\n    )\n    meta_xgb_ms.fit(X_train_L1_ms_ext_s, y_train_ms_ext, sample_weight=sw_ms_ext)\n\n    # ── Level-1: Predict trên MS Test ──\n    X_ms_L1_s = scaler_l1_ms.transform(X_ms_L1_smart)\n\n    proba_lr_ms  = meta_lr_ms.predict_proba(X_ms_L1_s)\n    proba_xgb_ms = meta_xgb_ms.predict_proba(X_ms_L1_s)\n\n    proba_ms_ens = w_lr_opt * proba_lr_ms + w_xgb_opt * proba_xgb_ms\n    proba_ms_ens = np.clip(proba_ms_ens, 1e-15, 1 - 1e-15)\n    proba_ms_ens = proba_ms_ens / proba_ms_ens.sum(axis=1, keepdims=True)\n\n    # ── Ghi submission.csv ──\n    # ── Ghi các file submission ──\n    import pandas as pd\n    df_sub = pd.read_csv(SUBMISSION_CSV)\n    has_pred_cols = any(c.startswith(\"Prediction\") for c in df_sub.columns)\n\n    def save_sub(proba, name):\n        df_out = df_sub[[\"Id\"]].copy()\n        if has_pred_cols:\n            for j in range(NUM_CLASSES):\n                df_out[f\"Prediction{j+1}\"] = proba[:, j]\n        else:\n            df_out[\"Class\"] = proba.argmax(axis=1) + 1\n        \n        path = os.path.join(WORKING_DIR, name)\n        df_out.to_csv(path, index=False)\n        return path\n\n    # 1. Ensemble Optimized\n    path_ens = save_sub(proba_ms_ens, \"submission.csv\")\n    print(f\"\\n   Ensemble Submission saved: {path_ens}\")\n    \n    # 2. LR Only\n    path_lr = save_sub(proba_lr_ms, \"submission_lr.csv\")\n    print(f\"   LR Only Submission saved: {path_lr}\")\n    \n    # 3. XGB Only\n    path_xgb = save_sub(proba_xgb_ms, \"submission_xgb.csv\")\n    print(f\"   XGB Only Submission saved: {path_xgb}\")\n    \n    print(f\"\\n     Total samples: {len(df_sub):,}\")\n    print(f\"     Weights applied: w_LR={w_lr_opt:.4f}, w_XGB={w_xgb_opt:.4f}\")\nelse:\n    if not ms_available:\n        print(\"   Microsoft test views not found. Skipping submission.\")\n    if not os.path.exists(SUBMISSION_CSV):\n        print(\"   sampleSubmission.csv not found. Skipping submission.\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null}]}