{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"databundleVersionId":46665,"sourceId":4117,"sourceType":"competition"},{"databundleVersionId":16640307,"datasetId":10057663,"sourceId":15701084,"sourceType":"datasetVersion"}],"dockerImageVersionId":31328,"isGpuEnabled":true,"isInternetEnabled":true,"language":"python","sourceType":"notebook"},"papermill":{"default_parameters":{},"duration":1284.674761,"end_time":"2026-04-13T09:14:30.286949+00:00","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-04-13T08:53:05.612188+00:00","version":"2.7.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"902e5f3f","cell_type":"markdown","source":"# Improved Training Notebook (v2 — Merged Data)\n\n## Input dataset\nNotebook này đọc **9 files đã gộp** từ `merge-parts.ipynb`:\n```\nX_train_tab.npy      X_test_tab.npy\nX_train_ng.npz       X_test_ng.npz\nX_train_opseq.pkl    X_test_opseq.pkl\nX_train_pixel.npy    X_test_pixel.npy\ny_train.npy\n```\n\n## Pipeline\n```\n[tab ~75] + [byte_ng 3K] + [opcode_ng 2K] + [pixel 1024] = ~6099 features\n        ↓\n  Level-0: XGBoost · LightGBM · ExtraTrees (5-fold OOF)\n        ↓  27 OOF proba cols  +  pseudo-labels\n  Level-1: Logistic Regression\n        ↓\n  (Optional) Extended Level-1: LR output + raw features → XGBoost shallow\n        ↓\n  submission.csv\n```","metadata":{"papermill":{"duration":0.002834,"end_time":"2026-04-13T08:53:08.223581+00:00","exception":false,"start_time":"2026-04-13T08:53:08.220747+00:00","status":"completed"},"tags":[]}},{"id":"a89fd9cd","cell_type":"code","source":"# ============================================================\n# CELL 1 — IMPORTS\n# ============================================================\nimport os\nimport gc\nimport time\nimport pickle\nimport joblib\nimport numpy as np\nimport pandas as pd\nimport scipy.sparse as sp\nimport lightgbm as lgb\nimport xgboost as xgb\nfrom sklearn.ensemble import ExtraTreesClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import log_loss, accuracy_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.feature_selection import SelectKBest, mutual_info_classif,chi2\n\nprint(\"Import xong!\")","metadata":{"execution":{"iopub.execute_input":"2026-04-13T08:53:08.229216Z","iopub.status.busy":"2026-04-13T08:53:08.228589Z","iopub.status.idle":"2026-04-13T08:53:16.355153Z","shell.execute_reply":"2026-04-13T08:53:16.354514Z"},"papermill":{"duration":8.131161,"end_time":"2026-04-13T08:53:16.356939+00:00","exception":false,"start_time":"2026-04-13T08:53:08.225778+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"7ff2eeb2","cell_type":"code","source":"# ============================================================\n# CELL 2 — CẤU HÌNH ĐƯỜNG DẪN\n# ============================================================\n\nDATA_IN_DIR   = '/kaggle/input/datasets/trankimhuu/data-ml-big-2015-ver-2'\n\nWORKING_DIR   = '/kaggle/working'\nMODEL_DIR     = os.path.join(WORKING_DIR, 'models')\nCKPT_DIR      = os.path.join(WORKING_DIR, 'checkpoints')\n\nos.makedirs(MODEL_DIR, exist_ok=True)\nos.makedirs(CKPT_DIR,  exist_ok=True)\n\nSUBMISSION_CSV = '/kaggle/input/competitions/malware-classification/sampleSubmission.csv'\nNUM_CLASSES    = 9\nN_FOLDS        = 5\nRANDOM_STATE   = 42\n\nprint(f\"📂 Input : {DATA_IN_DIR}\")\nprint(f\"📂 Models: {MODEL_DIR}\")","metadata":{"execution":{"iopub.execute_input":"2026-04-13T08:53:16.363390Z","iopub.status.busy":"2026-04-13T08:53:16.362780Z","iopub.status.idle":"2026-04-13T08:53:16.368824Z","shell.execute_reply":"2026-04-13T08:53:16.367955Z"},"papermill":{"duration":0.010561,"end_time":"2026-04-13T08:53:16.370130+00:00","exception":false,"start_time":"2026-04-13T08:53:16.359569+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"c1541426","cell_type":"code","source":"# ============================================================\n# CELL MỚI A — TÍNH CHI-SQUARE VÀ TRANSFORM BẰNG CHUNKING (CHỐNG OOM)\n# ============================================================\n\nprint(\"=\" * 55)\nprint(\"BƯỚC 1: CHUNKED FEATURE SELECTION TRÊN BYTE N-GRAM\")\nprint(\"=\" * 55)\n\nK_BYTE_NG = 3000\nCHUNK_SIZE = 1000  # Cắt nhỏ 1000 mẫu/lượt, tiêu tốn < 600MB RAM mỗi vòng\n\n# 1. Load data\nprint(\"Loading y_train và X_train_ng...\")\ny_train = np.load(os.path.join(DATA_IN_DIR, 'y_train.npy')).astype(int)\nX_train_ng = sp.load_npz(os.path.join(DATA_IN_DIR, 'X_train_ng.npz'))\n\nn_samples, n_features = X_train_ng.shape\nn_classes = len(np.unique(y_train))\nprint(f\"  Shape X_train_ng: {n_samples} samples, {n_features} features\")\n\n# 2. TÍNH TOÁN CHI-SQUARE THỦ CÔNG THEO CHUNK\nprint(f\"\\nTính toán Chi2 thủ công trên {n_features} features...\")\nt0 = time.time()\n\n# Tạo ma trận One-hot encoding cho y_train\nY = np.zeros((n_samples, n_classes))\nY[np.arange(n_samples), y_train] = 1\n\n# Các biến chứa kết quả thống kê\nobserved = np.zeros((n_classes, n_features), dtype=np.float64)\nfeature_count = np.zeros(n_features, dtype=np.float64)\n\n# Tính toán theo từng khối để tránh OOM\nfor start in range(0, n_samples, CHUNK_SIZE):\n    end = min(start + CHUNK_SIZE, n_samples)\n    X_chunk = X_train_ng[start:end]\n    Y_chunk = Y[start:end]\n    \n    # Cộng dồn Observed (X_chunk.T @ Y_chunk hiệu quả hơn với sparse matrix)\n    observed += (X_chunk.T @ Y_chunk).T\n    # Cộng dồn tổng xuất hiện của feature\n    feature_count += X_chunk.sum(axis=0).A1\n\n# Tính Expected theo công thức chuẩn của scikit-learn\nclass_prob = Y.mean(axis=0)\nexpected = np.outer(class_prob, feature_count)\nexpected = np.maximum(expected, 1e-9)  # Tránh lỗi chia cho 0\n\n# Tính điểm Chi-Square và lấy Top K\nchi2_scores = np.sum((observed - expected) ** 2 / expected, axis=0)\ntop_k_idx = np.argsort(chi2_scores)[-K_BYTE_NG:][::-1]\nprint(f\"  ✅ Tính xong Chi2 ({time.time()-t0:.1f}s)\")\n\n# 3. TRANSFORM X_TRAIN (Cũng bằng Chunking)\nprint(f\"\\nTrích xuất Top {K_BYTE_NG} features (Transform)...\")\nt0 = time.time()\nX_train_byteng = np.zeros((n_samples, K_BYTE_NG), dtype=np.float32)\n\nfor start in range(0, n_samples, CHUNK_SIZE):\n    end = min(start + CHUNK_SIZE, n_samples)\n    # Lấy chunk dạng sparse -> chuyển sang dense -> cắt cột top K\n    X_chunk_dense = X_train_ng[start:end].toarray()\n    X_train_byteng[start:end] = X_chunk_dense[:, top_k_idx]\n\n# CỰC KỲ QUAN TRỌNG: Giải phóng ma trận khổng lồ ngay lập tức\ndel X_train_ng\ngc.collect()\nprint(f\"  ✅ X_train_byteng xong: {X_train_byteng.shape} ({time.time()-t0:.1f}s)\")\n\n# 4. TRANSFORM X_TEST\nprint(\"\\nLoading và transforming X_test_ng...\")\nt0 = time.time()\nX_test_ng = sp.load_npz(os.path.join(DATA_IN_DIR, 'X_test_ng.npz'))\nn_samples_test = X_test_ng.shape[0]\n\nX_test_byteng = np.zeros((n_samples_test, K_BYTE_NG), dtype=np.float32)\n\nfor start in range(0, n_samples_test, CHUNK_SIZE):\n    end = min(start + CHUNK_SIZE, n_samples_test)\n    X_chunk_dense = X_test_ng[start:end].toarray()\n    X_test_byteng[start:end] = X_chunk_dense[:, top_k_idx]\n\ndel X_test_ng\ngc.collect()\nprint(f\"  ✅ X_test_byteng xong: {X_test_byteng.shape} ({time.time()-t0:.1f}s)\")\n\n# Lưu index thay vì lưu object selector của sklearn\njoblib.dump(top_k_idx, os.path.join(MODEL_DIR, 'top_k_byte_idx.pkl'))\nprint(\"Đã lưu top_k_byte_idx.pkl\")","metadata":{"execution":{"iopub.execute_input":"2026-04-13T08:53:16.375758Z","iopub.status.busy":"2026-04-13T08:53:16.375547Z","iopub.status.idle":"2026-04-13T08:57:05.407731Z","shell.execute_reply":"2026-04-13T08:57:05.406891Z"},"papermill":{"duration":229.039556,"end_time":"2026-04-13T08:57:05.411817+00:00","exception":false,"start_time":"2026-04-13T08:53:16.372261+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"876ee23d","cell_type":"code","source":"# ============================================================\n# CELL MỚI B — LOAD TABULAR & PIXEL RỒI GHÉP NỐI\n# ============================================================\nprint(\"=\" * 55)\nprint(\"BƯỚC 2: LOAD CÁC FEATURES CÒN LẠI VÀ GHÉP NỐI\")\nprint(\"=\" * 55)\n\nprint(\"Loading Tabular và Pixel features...\")\n# TRAIN\nX_train_tab = np.load(os.path.join(DATA_IN_DIR, 'X_train_tab.npy')).astype(np.float32)\nX_train_pixel = np.load(os.path.join(DATA_IN_DIR, 'X_train_pixel.npy')).astype(np.float32)\n\n# TEST\nX_test_tab = np.load(os.path.join(DATA_IN_DIR, 'X_test_tab.npy')).astype(np.float32)\nX_test_pixel = np.load(os.path.join(DATA_IN_DIR, 'X_test_pixel.npy')).astype(np.float32)\n\n# Ghép nối (Bỏ qua Opcode vì USE_OPCODE = False do lỗi data)\nprint(\"\\nĐang ghép features (hstack)...\")\nX_train = np.hstack([X_train_tab, X_train_byteng, X_train_pixel])\nX_test  = np.hstack([X_test_tab,  X_test_byteng,  X_test_pixel])\n\n# Giải phóng các mảng lẻ\ndel X_train_tab, X_train_byteng, X_train_pixel\ndel X_test_tab, X_test_byteng, X_test_pixel\ngc.collect()\n\nprint(f\"\\n  ✅ X_train Final : {X_train.shape}\")\nprint(f\"  ✅ X_test Final  : {X_test.shape}\")\n\n# Lưu checkpoint để meta-learner phía dưới (Cell 11) có thể tái sử dụng\nnp.save(os.path.join(CKPT_DIR, 'X_train.npy'), X_train)\nnp.save(os.path.join(CKPT_DIR, 'X_test.npy'),  X_test)\nnp.save(os.path.join(CKPT_DIR, 'y_train.npy'), y_train)\nprint(\"  💾 Các checkpoint đã được lưu an toàn!\")","metadata":{"execution":{"iopub.execute_input":"2026-04-13T08:57:05.418049Z","iopub.status.busy":"2026-04-13T08:57:05.417592Z","iopub.status.idle":"2026-04-13T08:57:06.612805Z","shell.execute_reply":"2026-04-13T08:57:06.612041Z"},"papermill":{"duration":1.200073,"end_time":"2026-04-13T08:57:06.614356+00:00","exception":false,"start_time":"2026-04-13T08:57:05.414283+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"c2807a80","cell_type":"code","source":"# ============================================================\n# CELL 7 — ĐỊNH NGHĨA BASE MODELS (LEVEL-0)\n# ============================================================\n\nprint(\"=\" * 55)\nprint(\"BƯỚC 5: CẤU HÌNH BASE MODELS\")\nprint(\"=\" * 55)\n\nclasses = np.unique(y_train)\nweights = compute_class_weight('balanced', classes=classes, y=y_train)\nclass_weights_dict = dict(zip(classes.astype(int), weights))\n\nprint(\"Class distribution & weights:\")\nfor c, w in class_weights_dict.items():\n    count = (y_train == c).sum()\n    print(f\"  Class {c+1}: {count:5d} mẫu  weight={w:.3f}\")\n\n# ── XGBoost ─────────────────────────────────────────────────\nxgb_model = xgb.XGBClassifier(\n    objective        = 'multi:softprob',\n    num_class        = NUM_CLASSES,\n    max_depth        = 7,\n    n_estimators     = 500,\n    learning_rate    = 0.05,\n    subsample        = 0.8,\n    colsample_bytree = 0.7,\n    min_child_weight = 3,\n    tree_method      = 'hist',\n    device           = 'cuda',\n    eval_metric      = 'mlogloss',\n    random_state     = RANDOM_STATE,\n)\n\n# ── LightGBM ────────────────────────────────────────────────\nlgb_model = lgb.LGBMClassifier(\n    objective         = 'multiclass',\n    num_class         = NUM_CLASSES,\n    class_weight      = class_weights_dict,\n    n_estimators      = 1000,\n    learning_rate     = 0.05,\n    num_leaves        = 127,\n    min_child_samples = 10,\n    feature_fraction  = 0.7,\n    bagging_fraction  = 0.8,\n    bagging_freq      = 5,\n    lambda_l1         = 0.1,\n    lambda_l2         = 0.1,\n    device            = 'gpu',\n    random_state      = RANDOM_STATE,\n    verbose           = -1,\n)\n\n# ── ExtraTrees ──────────────────────────────────────────────\net_model = ExtraTreesClassifier(\n    n_estimators = 500,\n    max_depth    = 20,\n    max_features = 'sqrt',\n    class_weight = 'balanced',\n    n_jobs       = -1,\n    random_state = RANDOM_STATE,\n)\n\nBASE_MODELS = {\n    'xgb':         xgb_model,\n    'lgb':         lgb_model,\n    'extra_trees': et_model,\n}\n\nskf = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=2026)\nprint(f\"\\n3 base models + {N_FOLDS}-fold CV sẵn sàng\")","metadata":{"execution":{"iopub.execute_input":"2026-04-13T08:57:06.621530Z","iopub.status.busy":"2026-04-13T08:57:06.621286Z","iopub.status.idle":"2026-04-13T08:57:06.635525Z","shell.execute_reply":"2026-04-13T08:57:06.634613Z"},"papermill":{"duration":0.019457,"end_time":"2026-04-13T08:57:06.636871+00:00","exception":false,"start_time":"2026-04-13T08:57:06.617414+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"0229438f","cell_type":"code","source":"# ============================================================\n# CELL 8 — LEVEL-0: OOF META-FEATURES\n# ============================================================\n# Mỗi model sinh:\n#   oof_train[valid_idx] = predict_proba(X_val)   → 9 cols\n#   oof_test             = mean(predict_proba(X_test) qua 5 folds) → 9 cols\n# Kết quả ghép: X_train_L1 (N_train × 27), X_test_L1 (N_test × 27)\n\nprint(\"=\" * 55)\nprint(\"BƯỚC 6: LEVEL-0 (OOF META-FEATURES)\")\nprint(\"=\" * 55)\n\nmeta_train_list = []\nmeta_test_list  = []\n\nfor model_name, model in BASE_MODELS.items():\n    print(f\"\\n{'─'*45}\")\n    print(f\"[{model_name.upper()}]\")\n\n    oof_train      = np.zeros((X_train.shape[0], NUM_CLASSES), dtype=np.float32)\n    oof_test_folds = np.zeros((N_FOLDS, X_test.shape[0], NUM_CLASSES), dtype=np.float32)\n\n    for fold, (tr_idx, val_idx) in enumerate(skf.split(X_train, y_train)):\n        print(f\"  Fold {fold+1}/{N_FOLDS}...\", end='', flush=True)\n        t0 = time.time()\n\n        model.fit(X_train[tr_idx], y_train[tr_idx])\n\n        oof_train[val_idx]   = model.predict_proba(X_train[val_idx])\n        oof_test_folds[fold] = model.predict_proba(X_test)\n\n        fold_ll = log_loss(y_train[val_idx], oof_train[val_idx])\n        print(f\" logloss={fold_ll:.5f}  ({time.time()-t0:.1f}s)\")\n\n    overall_ll = log_loss(y_train, oof_train)\n    print(f\"  OOF logloss: {overall_ll:.5f}\")\n\n    meta_train_list.append(oof_train)\n    meta_test_list.append(oof_test_folds.mean(axis=0))\n\n    # Full model (train trên 100% train data)\n    print(f\"  Training full model...\", end='', flush=True)\n    t0 = time.time()\n    model.fit(X_train, y_train)\n    joblib.dump(model, os.path.join(MODEL_DIR, f'{model_name}_full.pkl'))\n    print(f\" ✓ ({time.time()-t0:.1f}s)\")\n\nX_train_L1 = np.hstack(meta_train_list).astype(np.float32)  # (N_train, 27)\nX_test_L1  = np.hstack(meta_test_list).astype(np.float32)   # (N_test,  27)\n\nprint(f\"\\nLevel-0 hoàn tất!\")\nprint(f\"  X_train_L1: {X_train_L1.shape}\")\nprint(f\"  X_test_L1 : {X_test_L1.shape}\")\n\nnp.save(os.path.join(CKPT_DIR, 'X_train_L1.npy'), X_train_L1)\nnp.save(os.path.join(CKPT_DIR, 'X_test_L1.npy'),  X_test_L1)\nprint(\"Level-1 inputs saved\")","metadata":{"execution":{"iopub.execute_input":"2026-04-13T08:57:06.644986Z","iopub.status.busy":"2026-04-13T08:57:06.644732Z","iopub.status.idle":"2026-04-13T09:14:03.338334Z","shell.execute_reply":"2026-04-13T09:14:03.337507Z"},"papermill":{"duration":1016.703116,"end_time":"2026-04-13T09:14:03.344273+00:00","exception":false,"start_time":"2026-04-13T08:57:06.641157+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"bc8165d7","cell_type":"code","source":"# ============================================================\n# CELL 9 — PSEUDO-LABELING\n# ============================================================\n# Dùng average probability từ 3 full models (qua X_test_L1) với\n# threshold 0.999 để chỉ lấy predictions rất tự tin.\n# Tránh vòng lặp thông tin bằng cách không train model mới để tạo labels.\n\nprint(\"=\" * 55)\nprint(\"BƯỚC 7: PSEUDO-LABELING (threshold=0.999)\")\nprint(\"=\" * 55)\n\nCONFIDENCE_THRESHOLD = 0.999\n\n# X_test_L1 gồm [xgb_9cols | lgb_9cols | et_9cols]\n# Average probability qua 3 models\navg_test_proba = X_test_L1.reshape(-1, 3, NUM_CLASSES).mean(axis=1)  # (N_test, 9)\n\nmax_conf       = avg_test_proba.max(axis=1)\nconfident_mask = max_conf >= CONFIDENCE_THRESHOLD\nconfident_idx  = np.where(confident_mask)[0]\npseudo_labels  = avg_test_proba.argmax(axis=1)[confident_idx]\n\nprint(f\"  Tổng test    : {len(avg_test_proba):,}\")\nprint(f\"  Confident    : {len(confident_idx):,} ({100*len(confident_idx)/len(avg_test_proba):.1f}%)\")\nprint(\"\\n  Phân bố pseudo-labels:\")\nfor c in range(NUM_CLASSES):\n    cnt = (pseudo_labels == c).sum()\n    bar = '█' * (cnt // max(1, len(confident_idx)//40))\n    print(f\"    Class {c+1}: {cnt:5d}  {bar}\")\n\n# Ghép vào train set Level-1\nX_train_L1_ext = np.vstack([X_train_L1, X_test_L1[confident_idx]])\ny_train_ext    = np.concatenate([y_train, pseudo_labels])\nprint(f\"\\n  X_train_L1_ext: {X_train_L1_ext.shape}\")","metadata":{"execution":{"iopub.execute_input":"2026-04-13T09:14:03.354678Z","iopub.status.busy":"2026-04-13T09:14:03.354031Z","iopub.status.idle":"2026-04-13T09:14:03.365305Z","shell.execute_reply":"2026-04-13T09:14:03.364414Z"},"papermill":{"duration":0.01805,"end_time":"2026-04-13T09:14:03.366752+00:00","exception":false,"start_time":"2026-04-13T09:14:03.348702+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"0f26512e","cell_type":"code","source":"# ============================================================\n# CELL 10 — LEVEL-1: META-LEARNER (Logistic Regression)\n# ============================================================\n# LR với C=0.1 (regularization mạnh) trên 27 features:\n# - Tránh overfit hơn XGBoost 500 trees\n# - Đúng về lý thuyết: tổ hợp tuyến tính các probability predictions\n# - Nhanh hơn nhiều\n\nprint(\"=\" * 55)\nprint(\"BƯỚC 8: LEVEL-1 META-LEARNER (Logistic Regression)\")\nprint(\"=\" * 55)\n\n# StandardScaler vì LR nhạy cảm với scale\nscaler = StandardScaler()\nX_train_L1_ext_s = scaler.fit_transform(X_train_L1_ext)\nX_test_L1_s      = scaler.transform(X_test_L1)\njoblib.dump(scaler, os.path.join(MODEL_DIR, 'meta_scaler.pkl'))\n\nmeta_lr = LogisticRegression(\n    C            = 0.1,\n    solver       = 'lbfgs',\n    max_iter     = 2000,\n    multi_class  = 'multinomial',\n    class_weight = 'balanced',\n    random_state = RANDOM_STATE,\n)\n\nprint(\"Training...\", end='', flush=True)\nt0 = time.time()\nmeta_lr.fit(X_train_L1_ext_s, y_train_ext)\nprint(f\" ✓ ({time.time()-t0:.1f}s)\")\n\n# OOF eval (không có pseudo-labels để đánh giá sạch)\nX_train_L1_s = scaler.transform(X_train_L1)\npreds_lr     = meta_lr.predict_proba(X_train_L1_s)\nll_lr        = log_loss(y_train, preds_lr)\nacc_lr       = accuracy_score(y_train, preds_lr.argmax(axis=1))\nprint(f\"  OOF logloss : {ll_lr:.5f}\")\nprint(f\"  OOF accuracy: {acc_lr:.4f}\")\n\njoblib.dump(meta_lr, os.path.join(MODEL_DIR, 'meta_lr.pkl'))\nprint(\"meta_lr.pkl saved\")","metadata":{"execution":{"iopub.execute_input":"2026-04-13T09:14:03.377283Z","iopub.status.busy":"2026-04-13T09:14:03.376618Z","iopub.status.idle":"2026-04-13T09:14:03.756522Z","shell.execute_reply":"2026-04-13T09:14:03.755542Z"},"papermill":{"duration":0.388669,"end_time":"2026-04-13T09:14:03.759968+00:00","exception":false,"start_time":"2026-04-13T09:14:03.371299+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"8cd69338","cell_type":"code","source":"# ============================================================\n# CELL 11 — EXTENDED META-LEARNER (OOF + RAW FEATURES)\n# ============================================================\n# Bổ sung features gốc vào Level-1 để meta-learner có thêm context.\n# Dùng XGBoost shallow (depth=3) vì input giờ có nhiều chiều hơn.\n\nprint(\"=\" * 55)\nprint(\"BƯỚC 9: EXTENDED META-LEARNER (27 OOF + raw features)\")\nprint(\"=\" * 55)\n\n# Load lại X_train / X_test từ checkpoint\nX_train_raw = np.load(os.path.join(CKPT_DIR, 'X_train.npy'))\nX_test_raw  = np.load(os.path.join(CKPT_DIR, 'X_test.npy'))\nprint(f\"  X_train_raw: {X_train_raw.shape}\")\n\n# Ghép [27 OOF proba] + [~6099 raw features]\nX_train_L1_full = np.hstack([X_train_L1, X_train_raw]).astype(np.float32)\nX_test_L1_full  = np.hstack([X_test_L1,  X_test_raw]).astype(np.float32)\n\nX_train_L1_full_ext = np.vstack([\n    X_train_L1_full,\n    X_test_L1_full[confident_idx]\n])\n\nscaler_full = StandardScaler()\nX_train_L1_full_ext_s = scaler_full.fit_transform(X_train_L1_full_ext)\nX_test_L1_full_s      = scaler_full.transform(X_test_L1_full)\nX_train_L1_full_s     = scaler_full.transform(X_train_L1_full)\njoblib.dump(scaler_full, os.path.join(MODEL_DIR, 'meta_scaler_full.pkl'))\n\n# XGBoost shallow (depth=3 là intentional)\nmeta_xgb = xgb.XGBClassifier(\n    objective    = 'multi:softprob',\n    num_class    = NUM_CLASSES,\n    max_depth    = 3,    # shallow để tránh overfit\n    n_estimators = 200,  # ít hơn nhiều so với cũ (500)\n    learning_rate= 0.05,\n    subsample    = 0.8,\n    tree_method  = 'hist',\n    device       = 'cuda',\n    random_state = 99,\n)\n\nprint(\"Training extended XGB meta-learner...\", end='', flush=True)\nt0 = time.time()\nmeta_xgb.fit(X_train_L1_full_ext_s, y_train_ext)\nprint(f\" ✓ ({time.time()-t0:.1f}s)\")\n\npreds_xgb = meta_xgb.predict_proba(X_train_L1_full_s)\nll_xgb    = log_loss(y_train, preds_xgb)\nacc_xgb   = accuracy_score(y_train, preds_xgb.argmax(axis=1))\nprint(f\"  OOF logloss (extended): {ll_xgb:.5f}\")\nprint(f\"  OOF accuracy (extended): {acc_xgb:.4f}\")\n\njoblib.dump(meta_xgb, os.path.join(MODEL_DIR, 'meta_xgb_full.pkl'))\nprint(\"meta_xgb_full.pkl saved\")\n\ndel X_train_raw, X_test_raw\ngc.collect()","metadata":{"execution":{"iopub.execute_input":"2026-04-13T09:14:03.799633Z","iopub.status.busy":"2026-04-13T09:14:03.796642Z","iopub.status.idle":"2026-04-13T09:14:26.938580Z","shell.execute_reply":"2026-04-13T09:14:26.937768Z"},"papermill":{"duration":23.159295,"end_time":"2026-04-13T09:14:26.940098+00:00","exception":false,"start_time":"2026-04-13T09:14:03.780803+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"c6f4c342","cell_type":"code","source":"# ============================================================\n# CELL 12 — TẠO SUBMISSION (ĐÃ SỬA LỖI GHI ĐÈ FILE)\n# ============================================================\n\nprint(\"=\" * 55)\nprint(\"BƯỚC 10: TẠO SUBMISSION\")\nprint(\"=\" * 55)\n\ndf_sub = pd.read_csv(SUBMISSION_CSV)\n\n# ── Prediction từ 3 variants ─────────────────────────────────\nproba_lr  = meta_lr.predict_proba(X_test_L1_s)          # LR (27 features)\nproba_xgb = meta_xgb.predict_proba(X_test_L1_full_s)    # XGB (27 + raw)\n\n# Ensemble: average hai meta-learners\nproba_ens = (proba_lr + proba_xgb) / 2\n\n# ── Lưu 3 submission files ───────────────────────────────────\nresults = {\n    'lr_only' : proba_lr,\n    'xgb_full': proba_xgb,\n    'ensemble': proba_ens,\n}\n\nprint(\"\\n  OOF logloss summary:\")\n# In lại logloss nếu bạn muốn check\n# print(f\"    LR  (27 feat)   : {ll_lr:.5f}\")\n# print(f\"    XGB (27+raw)    : {ll_xgb:.5f}\")\n\n# Cột đích của Kaggle\npred_cols = [f'Prediction{i}' for i in range(1, 10)]\n\nfor name, proba in results.items():\n    df_out = df_sub.copy()\n    \n    # GHI ĐÈ TRỰC TIẾP ma trận xác suất (N x 9) vào 9 cột Prediction\n    df_out[pred_cols] = proba \n    \n    # (Tùy chọn) Xóa cột 'Class' nếu code cũ vô tình lưu vào df_sub\n    if 'Class' in df_out.columns:\n        df_out = df_out.drop(columns=['Class'])\n        \n    path = os.path.join(WORKING_DIR, f'submission_{name}.csv')\n    df_out.to_csv(path, index=False)\n    print(f\"\\n  Lưu thành công: {path}\")\n    \n    # In phân bố dự đoán ra console để kiểm tra xem model có thực sự học không\n    classes = proba.argmax(axis=1) + 1\n    for c in range(1, 10):\n        cnt = (classes == c).sum()\n        print(f\"     Dự đoán Class {c}: {cnt:5d} ({100*cnt/len(classes):.1f}%)\")\n\nprint(\"\\n✅ HOÀN TẤT! Recommend submit: submission_ensemble.csv\")","metadata":{"execution":{"iopub.execute_input":"2026-04-13T09:14:26.951761Z","iopub.status.busy":"2026-04-13T09:14:26.951067Z","iopub.status.idle":"2026-04-13T09:14:27.987706Z","shell.execute_reply":"2026-04-13T09:14:27.986374Z"},"papermill":{"duration":1.043966,"end_time":"2026-04-13T09:14:27.989256+00:00","exception":false,"start_time":"2026-04-13T09:14:26.945290+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"0f7f6859","cell_type":"code","source":"# # ============================================================\n# # CELL 13 — (TÙY CHỌN) OPTUNA TUNING CHO LightGBM\n# # ============================================================\n# # Bật ENABLE_OPTUNA=True nếu muốn tune LightGBM (~1-2h, 50 trials).\n# # Best params dùng để override lgb_model ở Cell 7 rồi chạy lại.\n\n# ENABLE_OPTUNA = False  # ← Đổi thành True để chạy\n\n# if ENABLE_OPTUNA:\n#     import optuna\n#     optuna.logging.set_verbosity(optuna.logging.WARNING)\n\n#     def lgb_objective(trial):\n#         params = {\n#             'objective'         : 'multiclass',\n#             'num_class'         : NUM_CLASSES,\n#             'metric'            : 'multi_logloss',\n#             'verbosity'         : -1,\n#             'device'            : 'gpu',\n#             'n_estimators'      : 1000,\n#             'learning_rate'     : trial.suggest_float('lr',         0.02, 0.1,  log=True),\n#             'num_leaves'        : trial.suggest_int('num_leaves',   63,   255),\n#             'min_child_samples' : trial.suggest_int('min_child',    5,    50),\n#             'feature_fraction'  : trial.suggest_float('feat_frac',  0.5,  1.0),\n#             'bagging_fraction'  : trial.suggest_float('bag_frac',   0.6,  1.0),\n#             'bagging_freq'      : trial.suggest_int('bag_freq',     1,    7),\n#             'lambda_l1'         : trial.suggest_float('l1',  1e-3, 10.0, log=True),\n#             'lambda_l2'         : trial.suggest_float('l2',  1e-3, 10.0, log=True),\n#             'max_depth'         : trial.suggest_int('max_depth',    6,    15),\n#         }\n#         dtrain = lgb.Dataset(X_train, label=y_train)\n#         cv_res = lgb.cv(\n#             params, dtrain,\n#             nfold=3, stratified=True,\n#             num_boost_round=500,\n#             callbacks=[lgb.early_stopping(30, verbose=False)],\n#         )\n#         return min(cv_res['valid multi_logloss-mean'])\n\n#     study = optuna.create_study(direction='minimize')\n#     study.optimize(lgb_objective, n_trials=50, show_progress_bar=True)\n\n#     print(f\"\\nBest logloss: {study.best_value:.5f}\")\n#     print(\"Best params (copy vào Cell 7):\")\n#     for k, v in study.best_params.items():\n#         print(f\"  '{k}': {v},\")\n\n#     joblib.dump(study, os.path.join(MODEL_DIR, 'optuna_study.pkl'))\n#     print(\"optuna_study.pkl saved\")\n# else:\n#     print(\"⏸  Optuna tắt. Đổi ENABLE_OPTUNA=True để chạy.\")","metadata":{"execution":{"iopub.execute_input":"2026-04-13T09:14:28.002197Z","iopub.status.busy":"2026-04-13T09:14:28.001364Z","iopub.status.idle":"2026-04-13T09:14:28.005951Z","shell.execute_reply":"2026-04-13T09:14:28.005380Z"},"papermill":{"duration":0.011858,"end_time":"2026-04-13T09:14:28.007346+00:00","exception":false,"start_time":"2026-04-13T09:14:27.995488+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null}]}