{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":4117,"databundleVersionId":46665},{"sourceType":"datasetVersion","sourceId":15592323,"datasetId":9960002,"databundleVersionId":16524836},{"sourceType":"kernelVersion","sourceId":309855409}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Phase 2+3 — CNN Feature Extraction + XGBoost Training\n\n**Input từ Phase 1:**\n```\n/kaggle/input/notebooks/nhatnguyenhcmusk24/nahhhhhhh/features/\n  ├── image_features.zip      (10,868 × 256×256×3 .npy)\n  ├── tabular_features.csv    (10,868 × 525 opcode features + Class)\n  └── vocabulary.pkl          (top_uni + top_bi)\n```\n\n**Pipeline Phase 2+3:**\n```\nimage_features.zip → extract → EfficientNet-B0 (frozen) → cnn_features (1280-d)\ntabular_features.csv                                     → ngram_features (525-d)\n7z l train.7z + test.7z                                  → metadata (4-d)\ntest.7z → batch extract → Markov + opcode               → test features\n                                                              ↓\nCONCAT → Feature Selection (RF) → XGBoost (Stratified 5-Fold + sample_weight)\n       → Pseudo-labeling → Ensemble → submission.csv\n```","metadata":{}},{"cell_type":"code","source":"import sys, os, gc, time, logging, zipfile, shutil, subprocess, pickle\nimport numpy as np\nimport pandas as pd\nimport scipy.sparse as sp\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\n\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.models as models\nimport torchvision.transforms as T\n\nimport xgboost as xgb\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.utils.class_weight import compute_sample_weight\nfrom sklearn.metrics import accuracy_score, log_loss\n\n# ── Utils ─────────────────────────────────────────────────────────────────────\nUTILS_PATH = Path('/kaggle/input/datasets/nhatnguyenhcmusk24/logic-feature')\nsys.path.append(str(UTILS_PATH))\nfrom malware_utils import Config, GPU, FileOps, ImageOps, OpcodeOps, CacheOps\n\n# ── Logging ───────────────────────────────────────────────────────────────────\nlogging.basicConfig(\n    level=logging.INFO,\n    format='%(asctime)s [%(levelname)s] %(message)s',\n    datefmt='%H:%M:%S',\n    handlers=[\n        logging.StreamHandler(sys.stdout),\n        logging.FileHandler('/kaggle/working/phase2.log')\n    ]\n)\nlog = logging.getLogger('phase2')\n\n# ── Paths ─────────────────────────────────────────────────────────────────────\nPHASE1_DIR  = Path('/kaggle/input/notebooks/nhatnguyenhcmusk24/data-process-for-microsoft-malware-detection-big/features')\nCOMP_DIR    = Path('/kaggle/input/competitions/malware-classification')\nWORK_DIR    = Path('/kaggle/working')\nIMG_DIR     = WORK_DIR / 'train_images'   # sau khi extract zip\nFEAT_DIR    = WORK_DIR / 'features'\nSCRATCH     = WORK_DIR / 'scratch'\n\nfor d in [IMG_DIR, FEAT_DIR, SCRATCH]:\n    d.mkdir(parents=True, exist_ok=True)\n\n# ── Config ────────────────────────────────────────────────────────────────────\ncfg = Config(base_dir=COMP_DIR)\nGPU.setup()\nDEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\nSEED     = 42\nN_FOLDS  = 5\nN_CLASSES= 9\n\nCLASS_NAMES = {\n    1:'Ramnit', 2:'Lollipop', 3:'Kelihos_ver3', 4:'Vundo', 5:'Simda',\n    6:'Tracur', 7:'Kelihos_ver1', 8:'Obfuscator.ACY', 9:'Gatak'\n}\n\nlog.info('Device: %s', DEVICE)\nlog.info('Phase1 dir: %s (exists=%s)', PHASE1_DIR, PHASE1_DIR.exists())\nlog.info('Disk free: %.1f GB', FileOps.get_free_gb())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T14:21:39.495284Z","iopub.execute_input":"2026-04-08T14:21:39.495583Z","iopub.status.idle":"2026-04-08T14:21:39.512252Z","shell.execute_reply.started":"2026-04-08T14:21:39.495561Z","shell.execute_reply":"2026-04-08T14:21:39.511448Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load Data process outputs","metadata":{}},{"cell_type":"code","source":"# ── Labels ────────────────────────────────────────────────────────────────────\ndf_labels = pd.read_csv(COMP_DIR / 'trainLabels.csv')\ndf_labels.columns = df_labels.columns.str.strip()\nfor c in df_labels.columns:\n    if c.lower() in ('id','name'): df_labels = df_labels.rename(columns={c:'Id'})\n    if c.lower() == 'class':       df_labels = df_labels.rename(columns={c:'Class'})\n\n# Label offset: XGBoost cần 0-indexed\ndf_labels['label'] = df_labels['Class'] - 1   # 1-9 → 0-8\nALL_IDS = df_labels['Id'].tolist()\nlog.info('Train samples: %d', len(ALL_IDS))\n\n# ── Tabular features từ Phase 1 ───────────────────────────────────────────────\ndf_tab = pd.read_csv(PHASE1_DIR / 'tabular_features.csv', index_col='Id')\nlog.info('Tabular shape: %s', df_tab.shape)\n# Meta data\nMETA_COLS    = ['bytes_size', 'asm_size', 'size_ratio']\ndf_meta  = df_tab[META_COLS].copy()\ndf_tab = df_tab.drop(columns=META_COLS)  \n# Tách Class ra\ny_tab = df_tab.pop('Class') if 'Class' in df_tab.columns else None\n\n# ── Vocabulary ────────────────────────────────────────────────────────────────\ntop_uni, top_bi = CacheOps.load_vocabulary(PHASE1_DIR / 'vocabulary.pkl')\nlog.info('Vocabulary: %d uni + %d bi', len(top_uni), len(top_bi))\n\n# ── Preview ───────────────────────────────────────────────────────────────────\ndisplay(df_tab.head(2))\nprint(f'Tabular shape  : {df_tab.shape}')\nprint(f'Non-zero cols  : {(df_tab > 0).sum(axis=1).describe().round(1)}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T14:21:39.513755Z","iopub.execute_input":"2026-04-08T14:21:39.514090Z","iopub.status.idle":"2026-04-08T14:21:40.350977Z","shell.execute_reply.started":"2026-04-08T14:21:39.514067Z","shell.execute_reply":"2026-04-08T14:21:40.350207Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Extract image_features.zip → train_images/","metadata":{}},{"cell_type":"code","source":"# ── Extract zip một lần, load nhanh về sau ────────────────────────────────────\nzip_path = PHASE1_DIR / 'image_features.zip'\nalready  = list(IMG_DIR.glob('*.npy'))\n\nif len(already) >= len(ALL_IDS):\n    log.info('Images đã extract: %d files — skip', len(already))\nelse:\n    log.info('Extracting %s → %s ...', zip_path.name, IMG_DIR)\n    t0 = time.time()\n    with zipfile.ZipFile(zip_path, 'r') as zf:\n        zf.extractall(IMG_DIR)\n    n = len(list(IMG_DIR.glob('*.npy')))\n    log.info('Extracted %d files in %.1fs', n, time.time()-t0)\n\nlog.info('Disk free after extract: %.1f GB', FileOps.get_free_gb())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T14:21:40.351875Z","iopub.execute_input":"2026-04-08T14:21:40.352273Z","iopub.status.idle":"2026-04-08T14:21:40.398358Z","shell.execute_reply.started":"2026-04-08T14:21:40.352246Z","shell.execute_reply":"2026-04-08T14:21:40.397763Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## CNN Feature Extraction (EfficientNet-B0, Frozen)","metadata":{}},{"cell_type":"code","source":"# ── Dataset ───────────────────────────────────────────────────────────────────\nclass MalwareImageDataset(Dataset):\n    \"\"\"\n    Load .npy tensors (256,256,3) uint8.\n    Normalize về ImageNet mean/std để dùng pretrained weights.\n    \"\"\"\n    def __init__(self, sample_ids: list, img_dir: Path):\n        self.ids     = sample_ids\n        self.img_dir = img_dir\n        self.tf = T.Compose([\n            T.ToTensor(),                          # HWC uint8 → CHW float32 [0,1]\n            T.Normalize(mean=[0.485, 0.456, 0.406],\n                        std =[0.229, 0.224, 0.225]),\n        ])\n\n    def __len__(self):  return len(self.ids)\n\n    def __getitem__(self, idx):\n        arr = np.load(self.img_dir / f'{self.ids[idx]}.npy')  # (256,256,3) uint8\n        return self.tf(arr), self.ids[idx]\n\n\n# ── Model: EfficientNet-B0 — bỏ head, giữ feature extractor ──────────────────\nclass CNNExtractor(nn.Module):\n    def __init__(self):\n        super().__init__()\n        backbone = models.efficientnet_b0(weights='IMAGENET1K_V1')\n        # Bỏ classifier head, giữ features + avgpool\n        self.features = backbone.features\n        self.avgpool  = backbone.avgpool   # AdaptiveAvgPool2d(1)\n        # Freeze toàn bộ\n        for p in self.parameters():\n            p.requires_grad = False\n\n    def forward(self, x):\n        x = self.features(x)\n        x = self.avgpool(x)\n        return x.flatten(1)   # → (batch, 1280)\n\n\n# ── Extract features ──────────────────────────────────────────────────────────\nCNN_FEAT_PATH = FEAT_DIR / 'cnn_features_train.npy'\nCNN_IDS_PATH  = FEAT_DIR / 'cnn_ids_train.pkl'\n\nif CNN_FEAT_PATH.exists():\n    log.info('CNN features đã có — skip extraction')\n    cnn_feat_train = np.load(CNN_FEAT_PATH)\n    with open(CNN_IDS_PATH, 'rb') as f:\n        cnn_ids_train = pickle.load(f)\nelse:\n    model = CNNExtractor().to(DEVICE).eval()\n    log.info('EfficientNet-B0 loaded (frozen) — output dim: 1280')\n\n    dataset = MalwareImageDataset(ALL_IDS, IMG_DIR)\n    loader  = DataLoader(dataset, batch_size=64, shuffle=False,\n                         num_workers=4, pin_memory=True)\n\n    all_feats = []\n    all_ids   = []\n    t0 = time.time()\n\n    with torch.no_grad():\n        for imgs, ids in tqdm(loader, desc='CNN extract', ncols=80):\n            feats = model(imgs.to(DEVICE))\n            all_feats.append(feats.cpu().numpy())\n            all_ids.extend(ids)\n\n    cnn_feat_train = np.vstack(all_feats)   # (N, 1280)\n    cnn_ids_train  = all_ids\n\n    np.save(CNN_FEAT_PATH, cnn_feat_train)\n    with open(CNN_IDS_PATH, 'wb') as f:\n        pickle.dump(cnn_ids_train, f, protocol=4)\n\n    elapsed = time.time() - t0\n    log.info('CNN extraction done: shape=%s in %.1f min',\n             cnn_feat_train.shape, elapsed/60)\n\n    del model\n    torch.cuda.empty_cache()\n    gc.collect()\n\nprint(f'CNN features shape: {cnn_feat_train.shape}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T14:21:40.399246Z","iopub.execute_input":"2026-04-08T14:21:40.399530Z","iopub.status.idle":"2026-04-08T14:21:40.434072Z","shell.execute_reply.started":"2026-04-08T14:21:40.399499Z","shell.execute_reply":"2026-04-08T14:21:40.433303Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Test set processing\n\nExtract test.7z theo batch → Markov image + Opcode features dùng vocabulary.pkl từ Phase 1","metadata":{}},{"cell_type":"code","source":"TEST_IMG_DIR  = WORK_DIR / 'test_images'\nTEST_CNN_PATH = FEAT_DIR / 'cnn_features_test.npy'\nTEST_TAB_PATH = FEAT_DIR / 'tabular_features_test.csv'\nTEST_META_PATH= FEAT_DIR / 'metadata_test.csv'\nTEST_IMG_DIR.mkdir(exist_ok=True)\n\ndf_sub_template = pd.read_csv(COMP_DIR / 'sampleSubmission.csv')\nTEST_IDS = df_sub_template['Id'].tolist()\nlog.info('Test samples: %d', len(TEST_IDS))\n\n\nfor p in [TEST_CNN_PATH, TEST_TAB_PATH, TEST_META_PATH]:\n    if p.exists():\n        p.unlink()\n        log.info('Deleted stale: %s', p.name)\n        \ntest_done = (TEST_CNN_PATH.exists() and\n             TEST_TAB_PATH.exists() and\n             TEST_META_PATH.exists())\nif test_done:\n    log.info('Test features đã có — skip')\nelse:\n    log.info('Processing test set...')\n\n    # FIX 2: Định nghĩa hàm trước khi dùng\n    def get_file_sizes_by_ext(archive: Path) -> tuple:\n        result = subprocess.run(['7z', 'l', '-ba', str(archive)],\n                                capture_output=True, text=True)\n        bytes_sizes, asm_sizes = {}, {}\n        for line in result.stdout.splitlines():\n            parts = line.strip().split()\n            if len(parts) >= 5:\n                try:\n                    size = int(parts[3])\n                    p    = Path(parts[-1])\n                    if p.suffix == '.bytes': bytes_sizes[p.stem] = size\n                    elif p.suffix == '.asm': asm_sizes[p.stem]   = size\n                except ValueError:\n                    pass\n        return bytes_sizes, asm_sizes\n\n    def process_test_batch(batch_ids, archive, img_dir, raw_records):\n        FileOps.extract_batch(archive, batch_ids, SCRATCH)\n        bytes_idx, asm_idx = {}, {}\n        for p in SCRATCH.rglob('*'):\n            if p.is_file():\n                if p.suffix == '.bytes': bytes_idx[p.stem] = p\n                elif p.suffix == '.asm': asm_idx[p.stem]   = p\n        for sid in batch_ids:\n            bp = bytes_idx.get(sid)\n            ap = asm_idx.get(sid)\n            img = ImageOps.process_image(bp, ap, cfg, exact_entropy=False)\n            np.save(img_dir / f'{sid}.npy', img)\n            counts = OpcodeOps.parse(ap) if ap else {'uni': {}, 'bi': {}}\n            counts['id'] = sid\n            raw_records.append(counts)\n            FileOps.delete_raw(bp, ap)\n        shutil.rmtree(SCRATCH, ignore_errors=True)\n        SCRATCH.mkdir(exist_ok=True)\n\n    # FIX 1: Sort theo archive order\n    log.info('Sorting test IDs by archive order...')\n    test_archive_order = FileOps.get_archive_order(COMP_DIR / 'test.7z')\n    TEST_IDS_SORTED    = FileOps.sort_by_archive_order(TEST_IDS, test_archive_order)\n    test_batches       = list(FileOps.chunked(TEST_IDS_SORTED, cfg.batch_n))\n\n    test_raw_records = []\n    for batch_ids in tqdm(test_batches, desc='Test batches', ncols=80):\n        process_test_batch(batch_ids, COMP_DIR / 'test.7z',\n                           TEST_IMG_DIR, test_raw_records)\n        gc.collect()\n        GPU.free()\n\n    # Vectorize — dùng vocabulary từ Phase 1\n    test_ids_vec, X_test_sparse = OpcodeOps.vectorize(\n        test_raw_records, top_uni, top_bi\n    )\n    df_test_tab = pd.DataFrame(\n        X_test_sparse.toarray(),\n        index=test_ids_vec,\n        columns=top_uni + top_bi\n    )\n    df_test_tab.index.name = 'Id'\n    df_test_tab.to_csv(TEST_TAB_PATH)\n\n    # Metadata\n    test_bytes_sizes, test_asm_sizes = get_file_sizes_by_ext(COMP_DIR / 'test.7z')\n    test_meta_rows = []\n    for sid in TEST_IDS:\n        bs  = test_bytes_sizes.get(sid, 0)\n        as_ = test_asm_sizes.get(sid, 0)\n        test_meta_rows.append({\n            'Id': sid, 'bytes_size': bs,\n            'asm_size': as_, 'size_ratio': as_/bs if bs > 0 else 0.0,\n        })\n    df_test_meta = pd.DataFrame(test_meta_rows).set_index('Id')\n    df_test_meta['unique_opcode_count'] = (df_test_tab > 0).sum(axis=1)\n    df_test_meta.to_csv(TEST_META_PATH)\n\n    # CNN\n    model = CNNExtractor().to(DEVICE).eval()\n    test_dataset = MalwareImageDataset(TEST_IDS, TEST_IMG_DIR)\n    test_loader  = DataLoader(test_dataset, batch_size=64, shuffle=False,\n                              num_workers=4, pin_memory=True)\n    test_feats, test_ids_cnn = [], []\n    with torch.no_grad():\n        for imgs, ids in tqdm(test_loader, desc='CNN test', ncols=80):\n            test_feats.append(model(imgs.to(DEVICE)).cpu().numpy())\n            test_ids_cnn.extend(ids)\n    cnn_feat_test = np.vstack(test_feats)\n    np.save(TEST_CNN_PATH, cnn_feat_test)\n\n    # FIX 3: Sanity check\n    assert cnn_feat_test.std() > 0.1, f\"CNN zeros! std={cnn_feat_test.std():.4f}\"\n    log.info('CNN sanity: mean=%.3f std=%.3f zeros=%.1f%%',\n             cnn_feat_test.mean(), cnn_feat_test.std(),\n             (cnn_feat_test == 0).mean() * 100)\n\n    del model\n    torch.cuda.empty_cache()\n    log.info('Test done: CNN=%s', cnn_feat_test.shape)\n\ncnn_feat_test = np.load(TEST_CNN_PATH)\ndf_test_tab   = pd.read_csv(TEST_TAB_PATH, index_col='Id')\ndf_test_meta  = pd.read_csv(TEST_META_PATH, index_col='Id')\nlog.info('Test loaded: CNN=%s | tab=%s', cnn_feat_test.shape, df_test_tab.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T14:21:40.435919Z","iopub.execute_input":"2026-04-08T14:21:40.436199Z","iopub.status.idle":"2026-04-08T16:46:26.372776Z","shell.execute_reply.started":"2026-04-08T14:21:40.436177Z","shell.execute_reply":"2026-04-08T16:46:26.372048Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Assemble full feature matrix","metadata":{}},{"cell_type":"code","source":"# ── Align tất cả features theo thứ tự ALL_IDS ─────────────────────────────────\n# CNN features — reorder theo ALL_IDS\ncnn_id_order = {sid: i for i, sid in enumerate(cnn_ids_train)}\ncnn_idx      = [cnn_id_order[sid] for sid in ALL_IDS]\nX_cnn_train  = cnn_feat_train[cnn_idx]           # (N, 1280)\n\n# Tabular features — reindex\nX_tab_train  = df_tab.reindex(ALL_IDS).fillna(0).values    # (N, 525)\n\n# Metadata features — reindex\ndf_meta['unique_opcode_count'] = (df_tab > 0).sum(axis=1)\nX_meta_train = df_meta.reindex(ALL_IDS).fillna(0).values   # (N, 4) ✅\nX_test_meta  = df_test_meta.reindex(TEST_IDS).fillna(0).values  # (M, 4) ✅\n\n\n# Labels\ny_train = df_labels.set_index('Id').reindex(ALL_IDS)['label'].values  # 0-indexed\n\n# Concat tất cả\nX_train = np.hstack([X_cnn_train, X_tab_train, X_meta_train])\nlog.info('X_train shape: %s (CNN=%d + Tab=%d + Meta=%d)',\n         X_train.shape, X_cnn_train.shape[1],\n         X_tab_train.shape[1], X_meta_train.shape[1])\n\n# Test features\nX_test_cnn  = cnn_feat_test                                              # (M, 1280)\nX_test_tab  = df_test_tab.reindex(TEST_IDS).fillna(0).values            # (M, 525)\nX_test = np.hstack([X_test_cnn, X_test_tab, X_test_meta])\nlog.info('X_test shape : %s', X_test.shape)\n\nprint(f'\\nX_train : {X_train.shape}  (total {X_train.shape[1]} features)')\nprint(f'X_test  : {X_test.shape}')\nprint(f'y_train : {y_train.shape}  classes={np.unique(y_train)}')\nprint(f'\\nClass distribution:')\nfor c, n in sorted(zip(*np.unique(y_train, return_counts=True))):\n    print(f'  Class {c+1} ({CLASS_NAMES[c+1]:<18}): {n:4d} ({n/len(y_train)*100:.1f}%)')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T16:46:26.374560Z","iopub.execute_input":"2026-04-08T16:46:26.374871Z","iopub.status.idle":"2026-04-08T16:46:27.035850Z","shell.execute_reply.started":"2026-04-08T16:46:26.374787Z","shell.execute_reply":"2026-04-08T16:46:27.035223Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Cell 7 — Feature Selection (Random Forest, Outside CV)","metadata":{}},{"cell_type":"code","source":"TOP_K = 1000   # giữ top-1000 features\nSEL_PATH = FEAT_DIR / 'selected_indices.npy'\n\nif SEL_PATH.exists():\n    selected_idx = np.load(SEL_PATH)\n    log.info('Feature indices loaded: %d features', len(selected_idx))\nelse:\n    log.info('Fitting Random Forest for feature importance (n_estimators=200)...')\n    t0 = time.time()\n\n    rf = RandomForestClassifier(\n        n_estimators=200,\n        max_depth=15,\n        n_jobs=-1,\n        random_state=SEED\n    )\n    rf.fit(X_train, y_train)\n\n    importances  = rf.feature_importances_\n    selected_idx = np.argsort(importances)[::-1][:TOP_K]\n    np.save(SEL_PATH, selected_idx)\n\n    log.info('Feature selection done in %.1f min', (time.time()-t0)/60)\n    log.info('Top 10 feature indices: %s', selected_idx[:10])\n\n# Apply selection\nX_train_sel = X_train[:, selected_idx]   # (N, 1000)\nX_test_sel  = X_test[:, selected_idx]    # (M, 1000)\n\nlog.info('After selection: train=%s  test=%s', X_train_sel.shape, X_test_sel.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T16:46:27.037058Z","iopub.execute_input":"2026-04-08T16:46:27.037390Z","iopub.status.idle":"2026-04-08T16:46:49.964626Z","shell.execute_reply.started":"2026-04-08T16:46:27.037366Z","shell.execute_reply":"2026-04-08T16:46:49.964004Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Stratified K-Fold XGBoost Training","metadata":{}},{"cell_type":"code","source":"# ── XGBoost params ────────────────────────────────────────────────────────────\nXGB_PARAMS = {\n    'objective'        : 'multi:softprob',\n    'num_class'        : N_CLASSES,\n    'eval_metric'      : 'mlogloss',\n    'max_depth'        : 7,\n    'learning_rate'    : 0.05,\n    'subsample'        : 0.8,\n    'colsample_bytree' : 0.8,\n    'min_child_weight' : 3,\n    'gamma'            : 0.1,\n    'tree_method'      : 'hist',   # GPU acceleration\n    'device'           : 'cuda',\n    'seed'             : SEED,\n    'verbosity'        : 0,\n}\n\n# ── sample_weight cho class imbalance ─────────────────────────────────────────\n# Đúng cách cho multiclass — scale_pos_weight bị ignore trong multi:softprob\nsample_weights = compute_sample_weight('balanced', y_train)\nlog.info('Sample weight range: [%.3f, %.3f]',\n         sample_weights.min(), sample_weights.max())\n\n# ── Stratified K-Fold ─────────────────────────────────────────────────────────\nskf = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=SEED)\n\noof_preds   = np.zeros((len(y_train), N_CLASSES))  # Out-of-fold predictions\ntest_preds  = np.zeros((len(TEST_IDS), N_CLASSES)) # Test predictions\nfold_models = []\nfold_scores = []\n\nlog.info('Starting %d-Fold XGBoost training...', N_FOLDS)\nt_total = time.time()\n\nfor fold, (tr_idx, val_idx) in enumerate(\n        skf.split(X_train_sel, y_train)):\n\n    t_fold = time.time()\n    log.info('Fold %d/%d ...', fold+1, N_FOLDS)\n\n    X_tr, X_val = X_train_sel[tr_idx], X_train_sel[val_idx]\n    y_tr, y_val = y_train[tr_idx],     y_train[val_idx]\n    w_tr        = sample_weights[tr_idx]\n\n    dtrain = xgb.DMatrix(X_tr,  label=y_tr,  weight=w_tr)\n    dval   = xgb.DMatrix(X_val, label=y_val)\n    dtest  = xgb.DMatrix(X_test_sel)\n\n    model = xgb.train(\n        XGB_PARAMS,\n        dtrain,\n        num_boost_round=1000,\n        evals=[(dval, 'val')],\n        early_stopping_rounds=30,\n        verbose_eval=100\n    )\n\n    # OOF predictions\n    val_prob = model.predict(dval)                # (val_size, 9)\n    oof_preds[val_idx] = val_prob\n\n    # Test predictions (accumulate)\n    test_preds += model.predict(dtest) / N_FOLDS\n\n    # Score\n    val_acc  = accuracy_score(y_val, val_prob.argmax(axis=1))\n    val_loss = log_loss(y_val, val_prob)\n    fold_scores.append({'acc': val_acc, 'loss': val_loss})\n    fold_models.append(model)\n\n    log.info('Fold %d | acc=%.4f | log_loss=%.4f | %.1fs',\n             fold+1, val_acc, val_loss, time.time()-t_fold)\n\n# ── OOF overall score ─────────────────────────────────────────────────────────\noof_acc  = accuracy_score(y_train, oof_preds.argmax(axis=1))\noof_loss = log_loss(y_train, oof_preds)\nlog.info('OOF overall | acc=%.4f | log_loss=%.4f | total=%.1f min',\n         oof_acc, oof_loss, (time.time()-t_total)/60)\n\nprint(f'\\n{\"─\"*40}')\nprint(f' OOF Accuracy : {oof_acc:.4f}')\nprint(f' OOF Log-loss : {oof_loss:.4f}')\nprint(f'{'─'*40}')\nfor i, s in enumerate(fold_scores):\n    print(f' Fold {i+1}: acc={s[\"acc\"]:.4f}  loss={s[\"loss\"]:.4f}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T16:46:49.965880Z","iopub.execute_input":"2026-04-08T16:46:49.966206Z","iopub.status.idle":"2026-04-08T16:48:14.681190Z","shell.execute_reply.started":"2026-04-08T16:46:49.966180Z","shell.execute_reply":"2026-04-08T16:48:14.680510Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Pseudo-labeling ","metadata":{}},{"cell_type":"code","source":"PSEUDO_THRESHOLD = 0.98   # chỉ lấy predictions confidence > 98%\nMAX_PSEUDO_ROUNDS = 2\n\nX_aug  = X_train_sel.copy()\ny_aug  = y_train.copy()\nw_aug  = sample_weights.copy()\n\nfor pseudo_round in range(MAX_PSEUDO_ROUNDS):\n    # Tìm high-confidence test predictions\n    max_probs   = test_preds.max(axis=1)          # probability của class tự tin nhất\n    pseudo_mask = max_probs >= PSEUDO_THRESHOLD\n    pseudo_y    = test_preds.argmax(axis=1)\n\n    n_pseudo = pseudo_mask.sum()\n    log.info('Pseudo round %d: %d/%d samples (threshold=%.2f)',\n             pseudo_round+1, n_pseudo, len(TEST_IDS), PSEUDO_THRESHOLD)\n\n    if n_pseudo == 0:\n        log.info('Không có pseudo-labels — dừng')\n        break\n\n    # Thêm pseudo-labeled test vào training\n    X_pseudo = X_test_sel[pseudo_mask]\n    y_pseudo = pseudo_y[pseudo_mask]\n    w_pseudo = compute_sample_weight('balanced',\n                                     np.concatenate([y_aug, y_pseudo]))[len(y_aug):]\n\n    X_aug = np.vstack([X_aug, X_pseudo])\n    y_aug = np.concatenate([y_aug, y_pseudo])\n    w_aug = np.concatenate([w_aug, w_pseudo])\n\n    log.info('Augmented train size: %d', len(y_aug))\n\n    # Retrain trên toàn bộ augmented data\n    dtrain_aug = xgb.DMatrix(X_aug,      label=y_aug, weight=w_aug)\n    dtest_xgb  = xgb.DMatrix(X_test_sel)\n\n    # Số rounds = best round từ fold training trung bình\n    best_rounds = int(np.mean([m.best_iteration for m in fold_models]))\n\n    pseudo_model = xgb.train(\n        {**XGB_PARAMS, 'verbosity': 0},\n        dtrain_aug,\n        num_boost_round=best_rounds,\n    )\n\n    # Update test predictions\n    new_test_preds = pseudo_model.predict(dtest_xgb)\n\n    # Kiểm tra improvement trên OOF (dùng làm proxy)\n    doof = xgb.DMatrix(X_train_sel)\n    new_oof = pseudo_model.predict(doof)\n    new_oof_acc  = accuracy_score(y_train, new_oof.argmax(axis=1))\n    new_oof_loss = log_loss(y_train, new_oof)\n    log.info('After pseudo round %d: OOF acc=%.4f (was %.4f) | loss=%.4f (was %.4f)',\n             pseudo_round+1, new_oof_acc, oof_acc, new_oof_loss, oof_loss)\n\n    if new_oof_acc > oof_acc:\n        test_preds = new_test_preds\n        oof_acc    = new_oof_acc\n        log.info('✅ Pseudo-labeling cải thiện — giữ model mới')\n    else:\n        log.info('⚠  Pseudo-labeling không cải thiện — giữ nguyên')\n        break","metadata":{"execution":{"iopub.status.busy":"2026-04-08T16:48:14.682257Z","iopub.execute_input":"2026-04-08T16:48:14.682534Z","iopub.status.idle":"2026-04-08T16:49:11.833206Z","shell.execute_reply.started":"2026-04-08T16:48:14.682513Z","shell.execute_reply":"2026-04-08T16:49:11.832598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Ensemble: Soft voting ─────────────────────────────────────────────────────\n# test_preds đã là average của 5 folds (Cell 8) + pseudo-labeling update\n# Final class = argmax(probabilities)\nfinal_class = test_preds.argmax(axis=1) + 1   # +1 vì label 0-indexed\n\n# ── OOF Analysis ─────────────────────────────────────────────────────────────\nprint('OOF CONFUSION MATRIX')\noof_class = oof_preds.argmax(axis=1)\n\nfrom sklearn.metrics import classification_report, confusion_matrix\n\nprint(classification_report(\n    y_train, oof_class,\n    target_names=[CLASS_NAMES[i+1] for i in range(9)]\n))\n\n# Per-class accuracy\nprint('\\nPer-class accuracy:')\nfor c in range(9):\n    mask = y_train == c\n    if mask.sum() > 0:\n        acc = (oof_class[mask] == c).mean()\n        n   = mask.sum()\n        flag = '⚠' if acc < 0.95 else '✅'\n        print(f'  {flag} Class {c+1} ({CLASS_NAMES[c+1]:<18}): '\n              f'{acc:.4f}  (n={n})')\n\n# ── Create submission ─────────────────────────────────────────────────────────\n# ── Create submission ─────────────────────────────────────────────────────────\n\ncolumn_names = [f'Prediction{i}' for i in range(1, 10)]\ndf_submission = pd.DataFrame(test_preds, columns=column_names)\n\ndf_submission.insert(0, 'Id', TEST_IDS)\n\nsub_path = WORK_DIR / 'submission.csv'\ndf_submission.to_csv(sub_path, index=False)\n\nprint(f'\\nSubmission saved: {sub_path}')\nprint(f'Shape: {df_submission.shape}') # Sẽ in ra (10873, 10)\ndisplay(df_submission.head())\n\n# ── Final summary ─────────────────────────────────────────────────────────────\nprint(f'\\n{\"═\"*45}')\nprint(f'  FINAL RESULTS')\nprint(f'  OOF Accuracy  : {oof_acc:.4f}')\nprint(f'  OOF Log-loss  : {oof_loss:.4f}')\nprint(f'  Target        : ≥ 0.99')\nstatus = '✅ ĐẠT' if oof_acc >= 0.99 else '⚠  CHƯA ĐẠT'\nprint(f'  Status        : {status}')\nprint(f'{'═'*45}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-09T04:45:43.503739Z","iopub.execute_input":"2026-04-09T04:45:43.503973Z","iopub.status.idle":"2026-04-09T04:45:43.523821Z","shell.execute_reply.started":"2026-04-09T04:45:43.503946Z","shell.execute_reply":"2026-04-09T04:45:43.522997Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Ensemble + Submission","metadata":{}}]}