{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"databundleVersionId":46665,"sourceId":4117,"sourceType":"competition"},{"databundleVersionId":16663061,"datasetId":10073093,"sourceId":15722285,"sourceType":"datasetVersion"}],"dockerImageVersionId":31328,"isGpuEnabled":true,"isInternetEnabled":true,"language":"python","sourceType":"notebook"},"papermill":{"default_parameters":{},"duration":2410.121529,"end_time":"2026-04-14T17:28:26.281021+00:00","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-04-14T16:48:16.159492+00:00","version":"2.7.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"295973f6","cell_type":"code","source":"# ╔══════════════════════════════════════════════════════════════╗\n# ║  MULTI-VIEW STACKING — MALWARE CLASSIFICATION BIG2015      ║\n# ║                                                              ║\n# ║  Kiến trúc: Train model riêng cho từng nhóm feature,        ║\n# ║  rồi stack predictions để tạo meta-features.                ║\n# ║                                                              ║\n# ║  Pipeline:                                                   ║\n# ║    View 1: XGB  trên [107 tabular]                          ║\n# ║    View 2: LGB  trên [3000 byte n-gram]                     ║\n# ║    View 3: XGB  trên [2000 opcode n-gram]                   ║\n# ║    View 4: XGB  trên [1024 pixel density]                   ║\n# ║    View 5: XGB  trên [6131 ALL features]                    ║\n# ║    View 6: LGB  trên [6131 ALL features]                    ║\n# ║         ↓ 54 OOF proba (6 views × 9 classes)                ║\n# ║    Level-1: XGB shallow (54 OOF + 6131 raw)                 ║\n# ║         ↓                                                    ║\n# ║    Temperature Calibration → submission.csv                  ║\n# ╚══════════════════════════════════════════════════════════════╝","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2026-04-14T16:48:18.800986Z","iopub.status.busy":"2026-04-14T16:48:18.800134Z","iopub.status.idle":"2026-04-14T16:48:18.804762Z","shell.execute_reply":"2026-04-14T16:48:18.804209Z"},"papermill":{"duration":0.010513,"end_time":"2026-04-14T16:48:18.806234+00:00","exception":false,"start_time":"2026-04-14T16:48:18.795721+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"987940f0","cell_type":"code","source":"# ============================================================\n# CELL 1 — IMPORTS\n# ============================================================\nimport os\nimport gc\nimport time\nimport pickle\nimport joblib\nimport numpy as np\nimport pandas as pd\nimport scipy.sparse as sp\nimport lightgbm as lgb\nimport xgboost as xgb\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import log_loss, accuracy_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.feature_selection import SelectKBest, mutual_info_classif\nfrom scipy.optimize import minimize, minimize_scalar\n\nprint(\"✅ Import xong!\")","metadata":{"execution":{"iopub.execute_input":"2026-04-14T16:48:18.812912Z","iopub.status.busy":"2026-04-14T16:48:18.812387Z","iopub.status.idle":"2026-04-14T16:48:26.685345Z","shell.execute_reply":"2026-04-14T16:48:26.684386Z"},"papermill":{"duration":7.878238,"end_time":"2026-04-14T16:48:26.686939+00:00","exception":false,"start_time":"2026-04-14T16:48:18.808701+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"2ea95d4b","cell_type":"code","source":"# ============================================================\n# CELL 2 — CẤU HÌNH ĐƯỜNG DẪN\n# ============================================================\n\nDATA_IN_DIR   = '/kaggle/input/datasets/trankimhuu/data-ml-big-2015-ver-2'\n\nWORKING_DIR   = '/kaggle/working'\nMODEL_DIR     = os.path.join(WORKING_DIR, 'models')\nCKPT_DIR      = os.path.join(WORKING_DIR, 'checkpoints')\n\nos.makedirs(MODEL_DIR, exist_ok=True)\nos.makedirs(CKPT_DIR,  exist_ok=True)\n\nSUBMISSION_CSV = '/kaggle/input/competitions/malware-classification/sampleSubmission.csv'\nNUM_CLASSES    = 9\nN_FOLDS        = 5\nRANDOM_STATE   = 42\n\nprint(f\"📂 Input : {DATA_IN_DIR}\")\nprint(f\"📂 Models: {MODEL_DIR}\")\n","metadata":{"execution":{"iopub.execute_input":"2026-04-14T16:48:26.694496Z","iopub.status.busy":"2026-04-14T16:48:26.693223Z","iopub.status.idle":"2026-04-14T16:48:26.699326Z","shell.execute_reply":"2026-04-14T16:48:26.698549Z"},"papermill":{"duration":0.011084,"end_time":"2026-04-14T16:48:26.700730+00:00","exception":false,"start_time":"2026-04-14T16:48:26.689646+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"e3ded680","cell_type":"code","source":"# ============================================================\n# CELL 3 — LOAD DỮ LIỆU (train_ng riêng, test_ng sau)\n# ============================================================\n# Giống notebook gốc: load train_ng trước, xử lý xong del,\n# rồi mới load test_ng ở Cell 5 để tránh OOM.\n\nprint(\"=\" * 55)\nprint(\"BƯỚC 1: LOAD DỮ LIỆU\")\nprint(\"=\" * 55)\n\ndef load_file(filename):\n    fp = os.path.join(DATA_IN_DIR, filename)\n    size_mb = os.path.getsize(fp) / 1e6\n    print(f\"  {filename:<30} ({size_mb:.0f} MB)...\", end=\"\", flush=True)\n    return fp\n\n# ── TRAIN ──\nfp = load_file(\"X_train_tab.npy\");   X_train_tab   = np.load(fp);          print(f\" {X_train_tab.shape}\")\nfp = load_file(\"X_train_ng.npz\");    X_train_ng    = sp.load_npz(fp);       print(f\" {X_train_ng.shape}\")\nfp = load_file(\"X_train_pixel.npy\"); X_train_pixel = np.load(fp);           print(f\" {X_train_pixel.shape}\")\nfp = load_file(\"X_train_opseq.pkl\")\nwith open(fp, \"rb\") as f: train_opseq = pickle.load(f)\nprint(f\" n={len(train_opseq):,}\")\nfp = load_file(\"y_train.npy\");       y_train = np.load(fp);                 print(f\" {y_train.shape}\")\n\n# ── TEST (chỉ load non-ng) ──\nfp = load_file(\"X_test_tab.npy\");    X_test_tab    = np.load(fp);           print(f\" {X_test_tab.shape}\")\nfp = load_file(\"X_test_pixel.npy\");  X_test_pixel  = np.load(fp);           print(f\" {X_test_pixel.shape}\")\nfp = load_file(\"X_test_opseq.pkl\")\nwith open(fp, \"rb\") as f: test_opseq = pickle.load(f)\nprint(f\" n={len(test_opseq):,}\")\n\nprint(f\"\\nTrain={X_train_tab.shape[0]:,} | Test={X_test_tab.shape[0]:,}\")","metadata":{"execution":{"iopub.execute_input":"2026-04-14T16:48:26.706985Z","iopub.status.busy":"2026-04-14T16:48:26.706494Z","iopub.status.idle":"2026-04-14T16:49:31.945261Z","shell.execute_reply":"2026-04-14T16:49:31.944281Z"},"papermill":{"duration":65.243702,"end_time":"2026-04-14T16:49:31.946976+00:00","exception":false,"start_time":"2026-04-14T16:48:26.703274+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"07ee272a","cell_type":"code","source":"# ============================================================\n# CELL 4 — XỬ LÝ TRAIN: Opcode + Byte N-gram + LƯU RIÊNG TỪNG VIEW\n# ============================================================\n# KHÁC BIỆT SO VỚI NOTEBOOK GỐC:\n#   Gốc: ghép tất cả → 1 array X_train → xóa individual arrays\n#   Mới: lưu TỪNG view riêng biệt vào checkpoint → dùng cho multi-view\n#\n# Sau cell này:\n#   CKPT_DIR/X_train_v_tab.npy     (10868, 107)\n#   CKPT_DIR/X_train_v_byteng.npy  (10868, 3000)\n#   CKPT_DIR/X_train_v_opng.npy    (10868, 2000)\n#   CKPT_DIR/X_train_v_pixel.npy   (10868, 1024)\n#   CKPT_DIR/X_train_v_all.npy     (10868, 6131)\n\nUSE_OPCODE  = True\nK_BYTE_NG   = 3000\nK_OPCODE_NG = 2000\nCHUNK_SIZE  = 500\n\nprint(\"=\" * 55)\nprint(\"BƯỚC 2: XỬ LÝ TRAIN — Opcode + Byte N-gram\")\nprint(\"=\" * 55)\n\n# ── Opcode N-gram (TRAIN) ────────────────────────────────────\nif USE_OPCODE:\n    from sklearn.feature_extraction.text import HashingVectorizer as _HashVec\n    lengths = [len(s.split()) for s in train_opseq]\n    avg_len = sum(lengths) / max(len(lengths), 1)\n    print(f\"  Avg opcode tokens/seq: {avg_len:.0f}\")\n\n    opcode_hash_vec = _HashVec(\n        ngram_range    = (2, 3),\n        n_features     = 2**14,\n        analyzer       = \"word\",\n        norm           = None,\n        alternate_sign = False,\n    )\n    print(\"  Vectorizing train opcode...\", end=\"\", flush=True)\n    t0 = time.time()\n    X_train_opng = opcode_hash_vec.transform(train_opseq)\n    print(f\" ({time.time()-t0:.1f}s) {X_train_opng.shape}\")\n    joblib.dump(opcode_hash_vec, os.path.join(MODEL_DIR, \"opcode_hash_vec.pkl\"))\nelse:\n    X_train_opng = None\n\ndel train_opseq; gc.collect()\n\n# ── Opcode chi2 selection (TRAIN) ────────────────────────────\nif USE_OPCODE and X_train_opng is not None:\n    from sklearn.feature_selection import chi2, SelectKBest\n    k_op = min(K_OPCODE_NG, X_train_opng.shape[1])\n    print(f\"  Opcode chi2 {X_train_opng.shape[1]:,}→{k_op}...\", end=\"\", flush=True)\n    t0 = time.time()\n    sel_opcode = SelectKBest(chi2, k=k_op)\n    X_train_opng_sel = sel_opcode.fit_transform(X_train_opng, y_train).toarray().astype(np.float32)\n    del X_train_opng; gc.collect()\n    print(f\" ({time.time()-t0:.1f}s) {X_train_opng_sel.shape}\")\n    joblib.dump(sel_opcode, os.path.join(MODEL_DIR, \"selector_opcode.pkl\"))\nelse:\n    X_train_opng_sel = np.zeros((X_train_tab.shape[0], 0), dtype=np.float32)\n\n# ── Byte N-gram chi2 (TRAIN, chunking) ───────────────────────\nprint(f\"  Byte N-gram chi2 train {X_train_ng.shape[1]:,}→{K_BYTE_NG} (chunking)...\", end=\"\", flush=True)\nt0 = time.time()\nn_samples, n_features = X_train_ng.shape\nn_classes = len(np.unique(y_train))\n\nY = np.zeros((n_samples, n_classes), dtype=np.float32)\nY[np.arange(n_samples), y_train] = 1.0\n\nobserved     = np.zeros((n_classes, n_features), dtype=np.float64)\nfeature_sum  = np.zeros(n_features, dtype=np.float64)\n\nfor start in range(0, n_samples, CHUNK_SIZE):\n    end      = min(start + CHUNK_SIZE, n_samples)\n    X_chunk  = X_train_ng[start:end]\n    observed += (X_chunk.T.dot(Y[start:end])).T\n    feature_sum += np.asarray(X_chunk.sum(axis=0)).ravel()\n\nclass_prob = Y.mean(axis=0)\nexpected   = np.outer(class_prob, feature_sum)\nnp.clip(expected, 1e-9, None, out=expected)\nchi2_scores = np.sum((observed - expected) ** 2 / expected, axis=0)\ntop_k_idx   = np.argsort(chi2_scores)[-K_BYTE_NG:]\njoblib.dump(top_k_idx, os.path.join(MODEL_DIR, \"top_k_byte_idx.pkl\"))\n\nX_train_byteng = np.zeros((n_samples, K_BYTE_NG), dtype=np.float32)\nfor start in range(0, n_samples, CHUNK_SIZE):\n    end = min(start + CHUNK_SIZE, n_samples)\n    X_train_byteng[start:end] = X_train_ng[start:end].toarray()[:, top_k_idx]\n\ndel X_train_ng, Y, observed, feature_sum; gc.collect()\nprint(f\" ({time.time()-t0:.1f}s) {X_train_byteng.shape}\")\n\n# ── LƯU TỪNG VIEW RIÊNG + ALL ────────────────────────────────\nX_train_tab_f   = X_train_tab.astype(np.float32)\nX_train_pixel_f = X_train_pixel.astype(np.float32)\n\nnp.save(os.path.join(CKPT_DIR, \"X_train_v_tab.npy\"),    X_train_tab_f)\nnp.save(os.path.join(CKPT_DIR, \"X_train_v_byteng.npy\"), X_train_byteng)\nnp.save(os.path.join(CKPT_DIR, \"X_train_v_opng.npy\"),   X_train_opng_sel)\nnp.save(os.path.join(CKPT_DIR, \"X_train_v_pixel.npy\"),  X_train_pixel_f)\n\nX_train_all = np.hstack([X_train_tab_f, X_train_byteng, X_train_opng_sel, X_train_pixel_f]).astype(np.float32)\nnp.save(os.path.join(CKPT_DIR, \"X_train_v_all.npy\"), X_train_all)\n\nprint(f\"\\n✅ Saved 5 train views:\")\nprint(f\"  tab    : {X_train_tab_f.shape}\")\nprint(f\"  byteng : {X_train_byteng.shape}\")\nprint(f\"  opng   : {X_train_opng_sel.shape}\")\nprint(f\"  pixel  : {X_train_pixel_f.shape}\")\nprint(f\"  all    : {X_train_all.shape}\")\n\ndel X_train_tab, X_train_tab_f, X_train_byteng, X_train_opng_sel, X_train_pixel, X_train_pixel_f, X_train_all\ngc.collect()\nprint(\"X_train_ng đã del. Sẵn sàng load X_test_ng ở Cell 5.\")","metadata":{"execution":{"iopub.execute_input":"2026-04-14T16:49:31.954667Z","iopub.status.busy":"2026-04-14T16:49:31.954406Z","iopub.status.idle":"2026-04-14T16:51:33.152656Z","shell.execute_reply":"2026-04-14T16:51:33.151964Z"},"papermill":{"duration":121.204162,"end_time":"2026-04-14T16:51:33.154283+00:00","exception":false,"start_time":"2026-04-14T16:49:31.950121+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"0b10882b","cell_type":"code","source":"# ============================================================\n# CELL 5 — XỬ LÝ TEST: Load X_test_ng → select → LƯU TỪNG VIEW\n# ============================================================\n\nprint(\"=\" * 55)\nprint(\"BƯỚC 3: XỬ LÝ TEST\")\nprint(\"=\" * 55)\n\ntop_k_idx  = joblib.load(os.path.join(MODEL_DIR, \"top_k_byte_idx.pkl\"))\nsel_opcode = joblib.load(os.path.join(MODEL_DIR, \"selector_opcode.pkl\")) if USE_OPCODE else None\n\n# ── Opcode (TEST) ──\nif USE_OPCODE:\n    opcode_hash_vec = joblib.load(os.path.join(MODEL_DIR, \"opcode_hash_vec.pkl\"))\n    print(\"  Vectorizing test opcode...\", end=\"\", flush=True)\n    t0 = time.time()\n    X_test_opng = opcode_hash_vec.transform(test_opseq)\n    X_test_opng_sel = sel_opcode.transform(X_test_opng).toarray().astype(np.float32)\n    del X_test_opng; gc.collect()\n    print(f\" ({time.time()-t0:.1f}s) {X_test_opng_sel.shape}\")\nelse:\n    X_test_opng_sel = np.zeros((X_test_tab.shape[0], 0), dtype=np.float32)\ndel test_opseq; gc.collect()\n\n# ── Byte N-gram (TEST) ──\nfp = os.path.join(DATA_IN_DIR, \"X_test_ng.npz\")\nsize_mb = os.path.getsize(fp) / 1e6\nprint(f\"  Loading X_test_ng.npz ({size_mb:.0f} MB)...\", end=\"\", flush=True)\nX_test_ng = sp.load_npz(fp)\nprint(f\" {X_test_ng.shape}\")\n\nn_test = X_test_ng.shape[0]\nX_test_byteng = np.zeros((n_test, len(top_k_idx)), dtype=np.float32)\nprint(f\"  Densify top-{len(top_k_idx)} cols (chunking)...\", end=\"\", flush=True)\nt0 = time.time()\nfor start in range(0, n_test, CHUNK_SIZE):\n    end = min(start + CHUNK_SIZE, n_test)\n    X_test_byteng[start:end] = X_test_ng[start:end].toarray()[:, top_k_idx]\ndel X_test_ng; gc.collect()\nprint(f\" ({time.time()-t0:.1f}s) {X_test_byteng.shape}\")\n\n# ── LƯU TỪNG VIEW RIÊNG + ALL ──\nX_test_tab_f   = X_test_tab.astype(np.float32)\nX_test_pixel_f = X_test_pixel.astype(np.float32)\n\nnp.save(os.path.join(CKPT_DIR, \"X_test_v_tab.npy\"),    X_test_tab_f)\nnp.save(os.path.join(CKPT_DIR, \"X_test_v_byteng.npy\"), X_test_byteng)\nnp.save(os.path.join(CKPT_DIR, \"X_test_v_opng.npy\"),   X_test_opng_sel)\nnp.save(os.path.join(CKPT_DIR, \"X_test_v_pixel.npy\"),  X_test_pixel_f)\n\nX_test_all = np.hstack([X_test_tab_f, X_test_byteng, X_test_opng_sel, X_test_pixel_f]).astype(np.float32)\nnp.save(os.path.join(CKPT_DIR, \"X_test_v_all.npy\"), X_test_all)\n\nprint(f\"\\n✅ Saved 5 test views:\")\nprint(f\"  tab    : {X_test_tab_f.shape}\")\nprint(f\"  byteng : {X_test_byteng.shape}\")\nprint(f\"  opng   : {X_test_opng_sel.shape}\")\nprint(f\"  pixel  : {X_test_pixel_f.shape}\")\nprint(f\"  all    : {X_test_all.shape}\")\n\ndel X_test_tab, X_test_tab_f, X_test_byteng, X_test_opng_sel, X_test_pixel, X_test_pixel_f, X_test_all\ngc.collect()\nprint(\"RAM đã sạch.\")","metadata":{"execution":{"iopub.execute_input":"2026-04-14T16:51:33.161763Z","iopub.status.busy":"2026-04-14T16:51:33.161551Z","iopub.status.idle":"2026-04-14T16:54:07.328317Z","shell.execute_reply":"2026-04-14T16:54:07.327427Z"},"papermill":{"duration":154.172533,"end_time":"2026-04-14T16:54:07.330036+00:00","exception":false,"start_time":"2026-04-14T16:51:33.157503+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"420f73b1","cell_type":"code","source":"# ============================================================\n# CELL 6 — LOAD CÁC VIEW TỪ CHECKPOINT\n# ============================================================\n# Load lại tất cả views vào memory. Tổng ~1.1 GB — an toàn.\n\nprint(\"=\" * 55)\nprint(\"BƯỚC 4: LOAD VIEWS TỪ CHECKPOINT\")\nprint(\"=\" * 55)\n\nVIEWS = {\n    \"tab\":    {\"file\": \"v_tab\",    \"desc\": \"Tabular (meta features)\"},\n    \"byteng\": {\"file\": \"v_byteng\", \"desc\": \"Byte N-gram (3000 chi2)\"},\n    \"opng\":   {\"file\": \"v_opng\",   \"desc\": \"Opcode N-gram (2000 chi2)\"},\n    \"pixel\":  {\"file\": \"v_pixel\",  \"desc\": \"Pixel density (32×32)\"},\n    \"all\":    {\"file\": \"v_all\",    \"desc\": \"ALL features combined\"},\n}\n\nX_train_views = {}\nX_test_views  = {}\n\nfor vname, vinfo in VIEWS.items():\n    X_tr = np.load(os.path.join(CKPT_DIR, f\"X_train_{vinfo['file']}.npy\"))\n    X_te = np.load(os.path.join(CKPT_DIR, f\"X_test_{vinfo['file']}.npy\"))\n    X_train_views[vname] = X_tr\n    X_test_views[vname]  = X_te\n    print(f\"  {vname:8s} : train {X_tr.shape}  test {X_te.shape}  — {vinfo['desc']}\")\n\nprint(f\"\\nTổng RAM views: {sum(v.nbytes for v in X_train_views.values()) / 1e9:.2f} GB (train)\")","metadata":{"execution":{"iopub.execute_input":"2026-04-14T16:54:07.338116Z","iopub.status.busy":"2026-04-14T16:54:07.337887Z","iopub.status.idle":"2026-04-14T16:54:07.680840Z","shell.execute_reply":"2026-04-14T16:54:07.679854Z"},"papermill":{"duration":0.348966,"end_time":"2026-04-14T16:54:07.682498+00:00","exception":false,"start_time":"2026-04-14T16:54:07.333532+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"5f40ae5f","cell_type":"code","source":"# ============================================================\n# CELL 7 — ĐỊNH NGHĨA MULTI-VIEW MODELS\n# ============================================================\n# Mỗi view dùng model với hyperparams TỐI ƯU cho feature group đó:\n#\n# - Tab (107 dim):    Ít features → depth cao hơn, colsample cao\n# - Byte_ng (3000):   Nhiều features sparse → colsample thấp, regularize mạnh\n# - Opcode_ng (2000): Tương tự byte_ng\n# - Pixel (1024):     Features liên tục, spatial → depth vừa phải\n# - ALL (6131):       Giống notebook gốc\n#\n# DIVERSITY: Dùng cả XGB và LGB trên ALL features để thêm diversity\n#            (2 thuật toán khác nhau trên cùng data = predictions ít correlated)\n\nprint(\"=\" * 55)\nprint(\"BƯỚC 5: CẤU HÌNH MULTI-VIEW MODELS\")\nprint(\"=\" * 55)\n\nclasses = np.unique(y_train)\nweights = compute_class_weight(\"balanced\", classes=classes, y=y_train)\nclass_weights_dict = dict(zip(classes.astype(int), weights))\n\nprint(\"Class distribution:\")\nfor c, w in class_weights_dict.items():\n    cnt = (y_train == c).sum()\n    print(f\"  Class {c+1}: {cnt:5d}  weight={w:.3f}\")\n\n# ── Định nghĩa model cho từng view ──────────────────────────\nVIEW_MODELS = {\n    # View 1: Tabular — ít features, cho phép deep trees\n    \"tab\": xgb.XGBClassifier(\n        objective='multi:softprob', num_class=NUM_CLASSES,\n        max_depth=8, n_estimators=500, learning_rate=0.05,\n        subsample=0.8, colsample_bytree=0.9,  # colsample cao vì chỉ 107 features\n        min_child_weight=3,\n        tree_method='hist', device='cuda',\n        eval_metric='mlogloss', random_state=RANDOM_STATE,\n    ),\n\n    # View 2: Byte N-gram — nhiều features, cần regularize mạnh\n    \"byteng\": lgb.LGBMClassifier(\n        objective='multiclass', num_class=NUM_CLASSES,\n        class_weight=class_weights_dict,\n        n_estimators=800, learning_rate=0.05,\n        num_leaves=127, min_child_samples=10,\n        feature_fraction=0.5,  # chỉ 50% features/tree → reduce overfitting\n        bagging_fraction=0.8, bagging_freq=5,\n        lambda_l1=0.1, lambda_l2=0.1,\n        device='gpu', random_state=RANDOM_STATE, verbose=-1,\n    ),\n\n    # View 3: Opcode N-gram — tương tự byte nhưng ít features hơn\n    \"opng\": xgb.XGBClassifier(\n        objective='multi:softprob', num_class=NUM_CLASSES,\n        max_depth=7, n_estimators=500, learning_rate=0.05,\n        subsample=0.8, colsample_bytree=0.6,\n        min_child_weight=3,\n        tree_method='hist', device='cuda',\n        eval_metric='mlogloss', random_state=RANDOM_STATE,\n    ),\n\n    # View 4: Pixel density — 1024 spatial features\n    \"pixel\": xgb.XGBClassifier(\n        objective='multi:softprob', num_class=NUM_CLASSES,\n        max_depth=7, n_estimators=500, learning_rate=0.05,\n        subsample=0.8, colsample_bytree=0.7,\n        min_child_weight=3,\n        tree_method='hist', device='cuda',\n        eval_metric='mlogloss', random_state=RANDOM_STATE,\n    ),\n\n    # View 5: ALL features (XGB) — giống notebook gốc\n    \"all\": xgb.XGBClassifier(\n        objective='multi:softprob', num_class=NUM_CLASSES,\n        max_depth=7, n_estimators=500, learning_rate=0.05,\n        subsample=0.8, colsample_bytree=0.7,\n        min_child_weight=3,\n        tree_method='hist', device='cuda',\n        eval_metric='mlogloss', random_state=RANDOM_STATE,\n    ),\n\n    # View 6: ALL features (LGB) — cùng features, khác thuật toán\n    \"all_lgb\": lgb.LGBMClassifier(\n        objective='multiclass', num_class=NUM_CLASSES,\n        class_weight=class_weights_dict,\n        n_estimators=1000, learning_rate=0.05,\n        num_leaves=127, min_child_samples=10,\n        feature_fraction=0.7, bagging_fraction=0.8, bagging_freq=5,\n        lambda_l1=0.1, lambda_l2=0.1,\n        device='gpu', random_state=RANDOM_STATE, verbose=-1,\n    ),\n}\n\n# Map view_name → feature key (all_lgb dùng cùng features với all)\nVIEW_FEATURE_MAP = {\n    \"tab\": \"tab\", \"byteng\": \"byteng\", \"opng\": \"opng\",\n    \"pixel\": \"pixel\", \"all\": \"all\", \"all_lgb\": \"all\",\n}\n\nskf = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=2026)\nN_VIEWS = len(VIEW_MODELS)\nprint(f\"\\n{N_VIEWS} view models + {N_FOLDS}-fold CV sẵn sàng\")\nprint(f\"OOF meta-features: {N_VIEWS} × {NUM_CLASSES} = {N_VIEWS * NUM_CLASSES}\")","metadata":{"execution":{"iopub.execute_input":"2026-04-14T16:54:07.691114Z","iopub.status.busy":"2026-04-14T16:54:07.690839Z","iopub.status.idle":"2026-04-14T16:54:07.706237Z","shell.execute_reply":"2026-04-14T16:54:07.705334Z"},"papermill":{"duration":0.021517,"end_time":"2026-04-14T16:54:07.707757+00:00","exception":false,"start_time":"2026-04-14T16:54:07.686240+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"821b7be5","cell_type":"code","source":"# ============================================================\n# CELL 8 — MULTI-VIEW LEVEL-0: OOF META-FEATURES\n# ============================================================\n# Train từng view model trên feature group riêng → tạo OOF predictions.\n# Mỗi model chỉ thấy 1 \"góc nhìn\" của data.\n#\n# Checkpoint: lưu OOF ngay sau mỗi view → resume nếu kernel crash.\n\nprint(\"=\" * 55)\nprint(\"BƯỚC 6: MULTI-VIEW LEVEL-0 (OOF)\")\nprint(\"=\" * 55)\n\nn_train = len(y_train)\nn_test  = X_test_views[\"tab\"].shape[0]\n\nmeta_train_list = []\nmeta_test_list  = []\n\nfor view_name, model in VIEW_MODELS.items():\n    feat_key = VIEW_FEATURE_MAP[view_name]\n    X_tr = X_train_views[feat_key]\n    X_te = X_test_views[feat_key]\n\n    ckpt_train = os.path.join(CKPT_DIR, f\"oof_train_{view_name}.npy\")\n    ckpt_test  = os.path.join(CKPT_DIR, f\"oof_test_{view_name}.npy\")\n\n    # Resume nếu đã có checkpoint\n    if os.path.exists(ckpt_train) and os.path.exists(ckpt_test):\n        print(f\"\\n[{view_name.upper()}] Checkpoint found, loading...\")\n        meta_train_list.append(np.load(ckpt_train))\n        meta_test_list.append(np.load(ckpt_test))\n        continue\n\n    print(f\"\\n{'─'*50}\")\n    print(f\"[{view_name.upper()}] features={X_tr.shape[1]}\")\n\n    oof_train    = np.zeros((n_train, NUM_CLASSES), dtype=np.float32)\n    oof_test_avg = np.zeros((n_test,  NUM_CLASSES), dtype=np.float32)\n\n    for fold, (tr_idx, val_idx) in enumerate(skf.split(X_tr, y_train)):\n        print(f\"  Fold {fold+1}/{N_FOLDS}...\", end=\"\", flush=True)\n        t0 = time.time()\n\n        model.fit(X_tr[tr_idx], y_train[tr_idx])\n        oof_train[val_idx] = model.predict_proba(X_tr[val_idx])\n        oof_test_avg += model.predict_proba(X_te) / N_FOLDS\n\n        fold_ll = log_loss(y_train[val_idx], oof_train[val_idx])\n        print(f\" ll={fold_ll:.5f}  ({time.time()-t0:.1f}s)\")\n        gc.collect()\n\n    overall_ll = log_loss(y_train, oof_train)\n    print(f\"  ✅ OOF logloss: {overall_ll:.5f}\")\n\n    # Checkpoint\n    np.save(ckpt_train, oof_train)\n    np.save(ckpt_test,  oof_test_avg)\n    print(f\"  Checkpoint saved\")\n\n    meta_train_list.append(oof_train)\n    meta_test_list.append(oof_test_avg)\n\n    # Full model train + save\n    print(f\"  Full model...\", end=\"\", flush=True)\n    t0 = time.time()\n    model.fit(X_tr, y_train)\n    joblib.dump(model, os.path.join(MODEL_DIR, f\"{view_name}_full.pkl\"))\n    print(f\" ({time.time()-t0:.1f}s)\")\n\n    # Cleanup model RAM\n    VIEW_MODELS[view_name] = None\n    del model; gc.collect()\n\n# Stack all OOF predictions\nX_train_L1 = np.hstack(meta_train_list).astype(np.float32)\nX_test_L1  = np.hstack(meta_test_list).astype(np.float32)\n\nnp.save(os.path.join(CKPT_DIR, \"X_train_L1.npy\"), X_train_L1)\nnp.save(os.path.join(CKPT_DIR, \"X_test_L1.npy\"),  X_test_L1)\n\nprint(f\"\\n{'='*50}\")\nprint(f\"Level-0 done!\")\nprint(f\"  X_train_L1: {X_train_L1.shape}  ({N_VIEWS} views × {NUM_CLASSES} classes)\")\nprint(f\"  X_test_L1 : {X_test_L1.shape}\")\n\n# Báo cáo OOF từng view\nprint(f\"\\n  Per-view OOF logloss:\")\nview_names = list(VIEW_FEATURE_MAP.keys())\nfor i, vname in enumerate(view_names):\n    oof_v = X_train_L1[:, i*NUM_CLASSES:(i+1)*NUM_CLASSES]\n    ll_v = log_loss(y_train, oof_v)\n    print(f\"    {vname:10s}: {ll_v:.5f}\")","metadata":{"execution":{"iopub.execute_input":"2026-04-14T16:54:07.716811Z","iopub.status.busy":"2026-04-14T16:54:07.716447Z","iopub.status.idle":"2026-04-14T17:27:37.714768Z","shell.execute_reply":"2026-04-14T17:27:37.713987Z"},"papermill":{"duration":2010.011512,"end_time":"2026-04-14T17:27:37.723192+00:00","exception":false,"start_time":"2026-04-14T16:54:07.711680+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"ed09c94f","cell_type":"code","source":"# ============================================================\n# CELL 9 — PSEUDO-LABELING\n# ============================================================\n# Dùng average prediction từ TẤT CẢ full models để tạo pseudo-labels.\n# Confidence threshold = 0.99\n\nprint(\"=\" * 55)\nprint(\"BƯỚC 7: PSEUDO-LABELING (threshold=0.99)\")\nprint(\"=\" * 55)\n\nCONFIDENCE_THRESHOLD = 0.99\n\n# Load tất cả full models và predict trên X_test\nprint(\"  Loading full models...\")\nproba_list = []\nfor view_name in VIEW_FEATURE_MAP.keys():\n    feat_key = VIEW_FEATURE_MAP[view_name]\n    m = joblib.load(os.path.join(MODEL_DIR, f\"{view_name}_full.pkl\"))\n    proba_list.append(m.predict_proba(X_test_views[feat_key]))\n    del m; gc.collect()\n\navg_test_proba = np.mean(proba_list, axis=0)\ndel proba_list; gc.collect()\n\nmax_conf      = avg_test_proba.max(axis=1)\nconfident_idx = np.where(max_conf >= CONFIDENCE_THRESHOLD)[0]\npseudo_labels = avg_test_proba.argmax(axis=1)[confident_idx]\n\nprint(f\"  Tổng test  : {len(avg_test_proba):,}\")\nprint(f\"  Confident  : {len(confident_idx):,} ({100*len(confident_idx)/len(avg_test_proba):.1f}%)\")\nprint(\"\\n  Phân bố pseudo-labels:\")\nfor c in range(NUM_CLASSES):\n    cnt = (pseudo_labels == c).sum()\n    print(f\"    Class {c+1}: {cnt:5d}\")\n\nX_train_L1_ext = np.vstack([X_train_L1, X_test_L1[confident_idx]])\ny_train_ext    = np.concatenate([y_train, pseudo_labels])\nprint(f\"\\n  X_train_L1_ext: {X_train_L1_ext.shape}\")\n\n# Giải phóng individual views (không cần nữa, chỉ cần ALL cho Level-1)\nfor vname in [\"tab\", \"byteng\", \"opng\", \"pixel\"]:\n    if vname in X_train_views: del X_train_views[vname]\n    if vname in X_test_views:  del X_test_views[vname]\ngc.collect()\nprint(\"  Views riêng lẻ đã giải phóng (giữ lại ALL cho Level-1)\")","metadata":{"execution":{"iopub.execute_input":"2026-04-14T17:27:37.737107Z","iopub.status.busy":"2026-04-14T17:27:37.736474Z","iopub.status.idle":"2026-04-14T17:27:44.790854Z","shell.execute_reply":"2026-04-14T17:27:44.790241Z"},"papermill":{"duration":7.063022,"end_time":"2026-04-14T17:27:44.792509+00:00","exception":false,"start_time":"2026-04-14T17:27:37.729487+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"ed917b3c","cell_type":"code","source":"# ============================================================\n# CELL 10 — LEVEL-1: META-LEARNER (OOF + RAW FEATURES)\n# ============================================================\n# Kiến trúc giống notebook gốc (đã cho kết quả tốt nhất):\n#   Input: [54 OOF proba] + [6131 raw features] = 6185 dim\n#   Model: XGB shallow (depth=3) + pseudo-labels\n#\n# THÊM: Train cả LR trên OOF-only (54 dim) để có backup\n\nprint(\"=\" * 55)\nprint(\"BƯỚC 8: LEVEL-1 META-LEARNER\")\nprint(\"=\" * 55)\n\n# ── Load raw features ──\nX_train_raw = X_train_views[\"all\"]  # vẫn còn trong memory\nX_test_raw  = X_test_views[\"all\"]\nprint(f\"  X_train_raw: {X_train_raw.shape}\")\n\n# ── 8a. LR trên OOF-only (nhanh, backup) ─────────────────────\nprint(\"\\n── LR Meta-learner (54 OOF features) ──\")\nscaler_lr = StandardScaler()\nX_train_L1_ext_s = scaler_lr.fit_transform(X_train_L1_ext)\nX_test_L1_s      = scaler_lr.transform(X_test_L1)\nX_train_L1_s     = scaler_lr.transform(X_train_L1)\njoblib.dump(scaler_lr, os.path.join(MODEL_DIR, 'meta_scaler_lr.pkl'))\n\nmeta_lr = LogisticRegression(\n    C=0.1, solver='lbfgs', max_iter=2000,\n    multi_class='multinomial', random_state=RANDOM_STATE,\n)\nmeta_lr.fit(X_train_L1_ext_s, y_train_ext)\npreds_lr = meta_lr.predict_proba(X_train_L1_s)\nll_lr    = log_loss(y_train, preds_lr)\nprint(f\"  OOF logloss (LR): {ll_lr:.5f}\")\njoblib.dump(meta_lr, os.path.join(MODEL_DIR, 'meta_lr.pkl'))\n\n# ── 8b. XGB Extended (OOF + raw, mạnh nhất) ──────────────────\nprint(\"\\n── XGB Extended Meta-learner (54 OOF + 6131 raw) ──\")\nX_train_L1_full = np.hstack([X_train_L1, X_train_raw]).astype(np.float32)\nX_test_L1_full  = np.hstack([X_test_L1,  X_test_raw]).astype(np.float32)\n\nX_train_L1_full_ext = np.vstack([\n    X_train_L1_full,\n    X_test_L1_full[confident_idx]\n])\n\nscaler_full = StandardScaler()\nX_train_L1_full_ext_s = scaler_full.fit_transform(X_train_L1_full_ext)\nX_test_L1_full_s      = scaler_full.transform(X_test_L1_full)\nX_train_L1_full_s     = scaler_full.transform(X_train_L1_full)\njoblib.dump(scaler_full, os.path.join(MODEL_DIR, 'meta_scaler_full.pkl'))\n\nmeta_xgb = xgb.XGBClassifier(\n    objective='multi:softprob', num_class=NUM_CLASSES,\n    max_depth=3, n_estimators=200, learning_rate=0.05,\n    subsample=0.8, tree_method='hist', device='cuda',\n    random_state=99,\n)\nprint(\"  Training...\", end='', flush=True)\nt0 = time.time()\nmeta_xgb.fit(X_train_L1_full_ext_s, y_train_ext)\nprint(f\" ✓ ({time.time()-t0:.1f}s)\")\n\npreds_xgb = meta_xgb.predict_proba(X_train_L1_full_s)\nll_xgb    = log_loss(y_train, preds_xgb)\nprint(f\"  OOF logloss (XGB ext): {ll_xgb:.5f}\")\njoblib.dump(meta_xgb, os.path.join(MODEL_DIR, 'meta_xgb_full.pkl'))\n\ndel X_train_raw, X_test_raw, X_train_L1_full, X_train_L1_full_ext\ngc.collect()","metadata":{"execution":{"iopub.execute_input":"2026-04-14T17:27:44.806764Z","iopub.status.busy":"2026-04-14T17:27:44.806117Z","iopub.status.idle":"2026-04-14T17:28:21.697732Z","shell.execute_reply":"2026-04-14T17:28:21.697006Z"},"papermill":{"duration":36.899976,"end_time":"2026-04-14T17:28:21.699044+00:00","exception":false,"start_time":"2026-04-14T17:27:44.799068+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"bdc87964","cell_type":"code","source":"# ============================================================\n# CELL 11 — TEMPERATURE CALIBRATION + SUBMISSION\n# ============================================================\n# Bước cuối: calibrate probabilities để tối ưu logloss.\n# Temperature scaling: logits/T → softmax\n# Tìm T tối ưu trên OOF predictions, áp dụng lên test.\n\nprint(\"=\" * 55)\nprint(\"BƯỚC 9: CALIBRATION + SUBMISSION\")\nprint(\"=\" * 55)\n\n# ── Test predictions ──\nproba_lr_test  = meta_lr.predict_proba(X_test_L1_s)\nproba_xgb_test = meta_xgb.predict_proba(X_test_L1_full_s)\n\n# ── Temperature Scaling ──────────────────────────────────────\ndef apply_temperature(proba, T):\n    log_p = np.log(np.clip(proba, 1e-15, 1.0))\n    scaled = log_p / T\n    exp_s = np.exp(scaled - scaled.max(axis=1, keepdims=True))\n    return exp_s / exp_s.sum(axis=1, keepdims=True)\n\n# Calibrate XGB Extended (model tốt nhất)\ndef temp_loss_xgb(T):\n    return log_loss(y_train, apply_temperature(preds_xgb, T))\n\nresult_T = minimize_scalar(temp_loss_xgb, bounds=(0.5, 3.0), method='bounded')\nT_opt = result_T.x\nll_calibrated = result_T.fun\n\nprint(f\"\\n  Temperature Scaling (XGB Extended):\")\nprint(f\"    T optimal      : {T_opt:.4f}\")\nprint(f\"    Before calib   : {ll_xgb:.6f}\")\nprint(f\"    After calib    : {ll_calibrated:.6f}\")\nprint(f\"    Improvement    : {ll_xgb - ll_calibrated:.6f}\")\n\nproba_xgb_cal = apply_temperature(proba_xgb_test, T_opt)\n\n# ── Ensemble: tìm weight tối ưu LR vs XGB_calibrated ─────────\nprint(f\"\\n  Ensemble weight optimization:\")\n\ndef ens_loss(w):\n    blended = w[0] * preds_lr + (1 - w[0]) * apply_temperature(preds_xgb, T_opt)\n    blended = np.clip(blended, 1e-15, 1 - 1e-15)\n    blended = blended / blended.sum(axis=1, keepdims=True)\n    return log_loss(y_train, blended)\n\nresult_w = minimize(ens_loss, x0=[0.3], bounds=[(0.0, 1.0)], method='L-BFGS-B')\nw_lr = result_w.x[0]\nll_ens = result_w.fun\nprint(f\"    w_LR={w_lr:.4f}  w_XGB={1-w_lr:.4f}\")\nprint(f\"    Ensemble OOF   : {ll_ens:.6f}\")\n\nproba_ens = w_lr * proba_lr_test + (1 - w_lr) * proba_xgb_cal\nproba_ens = np.clip(proba_ens, 1e-15, 1 - 1e-15)\nproba_ens = proba_ens / proba_ens.sum(axis=1, keepdims=True)\n\n# ── Tạo submissions ──────────────────────────────────────────\ndf_sub = pd.read_csv(SUBMISSION_CSV)\nprint(f\"\\n  sampleSubmission: {list(df_sub.columns)}\")\nsub_cols = list(df_sub.columns)\nhas_pred_cols = any(c.startswith(\"Prediction\") for c in sub_cols)\n\nsubmissions = {\n    \"xgb_extended\":   proba_xgb_test,               # raw XGB Extended\n    \"xgb_calibrated\": proba_xgb_cal,                 # + temperature calibration\n    \"lr_only\":        proba_lr_test,                  # LR only (backup)\n    \"ensemble\":       proba_ens,                      # calibrated ensemble\n}\n\nfor name, proba in submissions.items():\n    df_out = df_sub[[\"Id\"]].copy()\n    if has_pred_cols:\n        for j in range(NUM_CLASSES):\n            df_out[f\"Prediction{j+1}\"] = proba[:, j]\n    else:\n        df_out[\"Class\"] = proba.argmax(axis=1) + 1\n\n    path = os.path.join(WORKING_DIR, f\"submission_{name}.csv\")\n    df_out.to_csv(path, index=False)\n\n    if name == \"xgb_extended\":   note = f\"raw (OOF: {ll_xgb:.5f})\"\n    elif name == \"xgb_calibrated\": note = f\"T={T_opt:.3f} (OOF: {ll_calibrated:.5f})\"\n    elif name == \"lr_only\":      note = f\"OOF: {ll_lr:.5f}\"\n    else:                        note = f\"w_lr={w_lr:.3f} (OOF: {ll_ens:.5f})\"\n\n    print(f\"\\n  {path}  [{note}]\")\n    pred_classes = proba.argmax(axis=1) + 1\n    for c in range(1, 10):\n        cnt = (pred_classes == c).sum()\n        print(f\"    Class {c}: {cnt:5d} ({100*cnt/len(pred_classes):.1f}%)\")\n\nprint(\"\\n\" + \"=\" * 55)\nprint(\"✅ HOÀN TẤT!\")\nprint(\"=\" * 55)\nprint(f\"\\n  Multi-view pipeline: {N_VIEWS} views × {N_FOLDS} folds\")\nprint(f\"  Temperature: T = {T_opt:.4f}\")\nprint(f\"  Ensemble weights: LR={w_lr:.3f}, XGB={1-w_lr:.3f}\")\nprint(f\"\\n  📋 Submissions tạo ra:\")\nprint(f\"     1. submission_xgb_calibrated.csv  — KHUYÊN DÙNG (calibrated)\")\nprint(f\"     2. submission_ensemble.csv         — thử nếu (1) không tốt\")\nprint(f\"     3. submission_xgb_extended.csv     — raw, không calibrate\")\nprint(f\"     4. submission_lr_only.csv          — backup\")","metadata":{"execution":{"iopub.execute_input":"2026-04-14T17:28:21.714495Z","iopub.status.busy":"2026-04-14T17:28:21.714259Z","iopub.status.idle":"2026-04-14T17:28:23.496912Z","shell.execute_reply":"2026-04-14T17:28:23.496102Z"},"papermill":{"duration":1.792089,"end_time":"2026-04-14T17:28:23.498555+00:00","exception":false,"start_time":"2026-04-14T17:28:21.706466+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null}]}