{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":4117,"databundleVersionId":46665},{"sourceType":"datasetVersion","sourceId":15843827,"datasetId":9960002,"databundleVersionId":16794483},{"sourceType":"kernelVersion","sourceId":313150593}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Phase 2+3 — CNN Feature Extraction + XGBoost Training\n\n**Input từ Phase 1:**\n```\n/kaggle/input/notebooks/nhatnguyenhcmusk24/nahhhhhhh/features/\n  ├── image_features.zip      (10,868 × 256×256×3 .npy)\n  ├── tabular_features.csv    (10,868 × 525 opcode features + Class)\n  └── vocabulary.pkl          (top_uni + top_bi)\n```\n\n**Pipeline Phase 2+3:**\n```\nimage_features.zip → extract → EfficientNet-B0 (frozen) → cnn_features (1280-d)\ntabular_features.csv                                     → ngram_features (525-d)\n7z l train.7z + test.7z                                  → metadata (4-d)\ntest.7z → batch extract → Markov + opcode               → test features\n                                                              ↓\nCONCAT → Feature Selection (RF) → XGBoost (Stratified 5-Fold + sample_weight)\n       → Pseudo-labeling → Ensemble → submission.csv\n```","metadata":{}},{"cell_type":"code","source":"import sys, os, gc, time, logging, zipfile, shutil, subprocess, pickle\nimport numpy as np\nimport pandas as pd\nimport scipy.sparse as sp\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\n\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.models as models\nimport torchvision.transforms as T\n\nimport xgboost as xgb\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.utils.class_weight import compute_sample_weight\nfrom sklearn.metrics import accuracy_score, log_loss\n\n# ── Utils ─────────────────────────────────────────────────────────────────────\nUTILS_PATH = Path('/kaggle/input/datasets/nhatnguyenhcmusk24/logic-feature')\nsys.path.append(str(UTILS_PATH))\nfrom malware_utils import Config, GPU, FileOps, ImageOps, OpcodeOps, CacheOps\n\n# ── Logging ───────────────────────────────────────────────────────────────────\nlogging.basicConfig(\n    level=logging.INFO,\n    format='%(asctime)s [%(levelname)s] %(message)s',\n    datefmt='%H:%M:%S',\n    handlers=[\n        logging.StreamHandler(sys.stdout),\n        logging.FileHandler('/kaggle/working/phase2.log')\n    ]\n)\nlog = logging.getLogger('phase2')\n\n# ── Paths ─────────────────────────────────────────────────────────────────────\nPHASE1_DIR  = Path('/kaggle/input/notebooks/nhatnguyenhcmusk24/fork-of-data-process-for-microsoft-malware-detecti/features')\nCOMP_DIR    = Path('/kaggle/input/competitions/malware-classification')\nWORK_DIR    = Path('/kaggle/working')\nIMG_DIR     = WORK_DIR / 'train_images'   # sau khi extract zip\nFEAT_DIR    = WORK_DIR / 'features'\nSCRATCH     = WORK_DIR / 'scratch'\n\nfor d in [IMG_DIR, FEAT_DIR, SCRATCH]:\n    d.mkdir(parents=True, exist_ok=True)\n\n# ── Config ────────────────────────────────────────────────────────────────────\ncfg = Config(base_dir=COMP_DIR)\nGPU.setup()\nDEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\nSEED     = 42\nN_FOLDS  = 5\nN_CLASSES= 9\n\nCLASS_NAMES = {\n    1:'Ramnit', 2:'Lollipop', 3:'Kelihos_ver3', 4:'Vundo', 5:'Simda',\n    6:'Tracur', 7:'Kelihos_ver1', 8:'Obfuscator.ACY', 9:'Gatak'\n}\n\nlog.info('Device: %s', DEVICE)\nlog.info('Phase1 dir: %s (exists=%s)', PHASE1_DIR, PHASE1_DIR.exists())\nlog.info('Disk free: %.1f GB', FileOps.get_free_gb())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-21T06:37:56.106120Z","iopub.execute_input":"2026-04-21T06:37:56.106419Z","iopub.status.idle":"2026-04-21T06:38:10.322861Z","shell.execute_reply.started":"2026-04-21T06:37:56.106383Z","shell.execute_reply":"2026-04-21T06:38:10.321926Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load Data process outputs","metadata":{}},{"cell_type":"code","source":"# ── Labels ────────────────────────────────────────────────────────────────────\ndf_labels = pd.read_csv(COMP_DIR / 'trainLabels.csv')\ndf_labels.columns = df_labels.columns.str.strip()\nfor c in df_labels.columns:\n    if c.lower() in ('id','name'): df_labels = df_labels.rename(columns={c:'Id'})\n    if c.lower() == 'class':       df_labels = df_labels.rename(columns={c:'Class'})\n\n# Label offset: XGBoost cần 0-indexed\ndf_labels['label'] = df_labels['Class'] - 1   # 1-9 → 0-8\nALL_IDS = df_labels['Id'].tolist()\nlog.info('Train samples: %d', len(ALL_IDS))\n\n# ── Tabular features từ Phase 1 (đã có 727 cols) ──────────────────────────────\ndf_tab = pd.read_csv(PHASE1_DIR / 'tabular_features.csv', index_col='Id')\nlog.info('Tabular raw shape: %s', df_tab.shape)\n\n# ── Tách 3 nhóm cột: opcode / extra / meta ───────────────────────────────────\nMETA_COLS  = ['bytes_size', 'asm_size', 'size_ratio']\n\n# Extra features (23 cols từ Phase 1 update: sections + API groups)\nEXTRA_COLS = [c for c in df_tab.columns\n              if c.startswith('sec_') or c.startswith('api_')\n              or c in ('n_distinct_sections', 'n_total_imports')]\n\ndf_meta  = df_tab[META_COLS].copy()\ndf_extra = df_tab[EXTRA_COLS].copy()\ndf_tab   = df_tab.drop(columns=META_COLS + EXTRA_COLS)\n\n# Tách Class ra\ny_tab = df_tab.pop('Class') if 'Class' in df_tab.columns else None\n\nlog.info('Split | Opcode=%d  Extra=%d  Meta=%d',\n         df_tab.shape[1], df_extra.shape[1], df_meta.shape[1])\n\n# ── Vocabulary ────────────────────────────────────────────────────────────────\ntop_uni, top_bi = CacheOps.load_vocabulary(PHASE1_DIR / 'vocabulary.pkl')\nlog.info('Vocabulary: %d uni + %d bi', len(top_uni), len(top_bi))\n\n# ── Preview ───────────────────────────────────────────────────────────────────\ndisplay(df_tab.head(2))\nprint(f'Opcode shape   : {df_tab.shape}')\nprint(f'Extra shape    : {df_extra.shape}')\nprint(f'Extra cols     : {EXTRA_COLS}')\nprint(f'Non-zero opcode: {(df_tab > 0).sum(axis=1).describe().round(1)}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-21T06:38:16.724703Z","iopub.execute_input":"2026-04-21T06:38:16.725400Z","iopub.status.idle":"2026-04-21T06:38:17.952192Z","shell.execute_reply.started":"2026-04-21T06:38:16.725368Z","shell.execute_reply":"2026-04-21T06:38:17.951347Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Extract image_features.zip → train_images/","metadata":{}},{"cell_type":"code","source":"# ── Extract zip một lần, load nhanh về sau ────────────────────────────────────\nzip_path = PHASE1_DIR / 'image_features.zip'\nalready  = list(IMG_DIR.glob('*.npy'))\n\nif len(already) >= len(ALL_IDS):\n    log.info('Images đã extract: %d files — skip', len(already))\nelse:\n    log.info('Extracting %s → %s ...', zip_path.name, IMG_DIR)\n    t0 = time.time()\n    with zipfile.ZipFile(zip_path, 'r') as zf:\n        zf.extractall(IMG_DIR)\n    n = len(list(IMG_DIR.glob('*.npy')))\n    log.info('Extracted %d files in %.1fs', n, time.time()-t0)\n\nlog.info('Disk free after extract: %.1f GB', FileOps.get_free_gb())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-21T06:38:35.630049Z","iopub.execute_input":"2026-04-21T06:38:35.630847Z","iopub.status.idle":"2026-04-21T06:38:54.042286Z","shell.execute_reply.started":"2026-04-21T06:38:35.630814Z","shell.execute_reply":"2026-04-21T06:38:54.041420Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## CNN Feature Extraction (EfficientNet-B0, Frozen)","metadata":{}},{"cell_type":"code","source":"# ── Dataset ───────────────────────────────────────────────────────────────────\nclass MalwareImageDataset(Dataset):\n    \"\"\"\n    Load .npy tensors (256,256,3) uint8.\n    Normalize về ImageNet mean/std để dùng pretrained weights.\n    \"\"\"\n    def __init__(self, sample_ids: list, img_dir: Path):\n        self.ids     = sample_ids\n        self.img_dir = img_dir\n        self.tf = T.Compose([\n            T.ToTensor(),                          # HWC uint8 → CHW float32 [0,1]\n            T.Normalize(mean=[0.485, 0.456, 0.406],\n                        std =[0.229, 0.224, 0.225]),\n        ])\n\n    def __len__(self):  return len(self.ids)\n\n    def __getitem__(self, idx):\n        arr = np.load(self.img_dir / f'{self.ids[idx]}.npy')  # (256,256,3) uint8\n        return self.tf(arr), self.ids[idx]\n\n\n# ── Model: EfficientNet-B0 — bỏ head, giữ feature extractor ──────────────────\nclass CNNExtractor(nn.Module):\n    def __init__(self):\n        super().__init__()\n        backbone = models.efficientnet_b0(weights='IMAGENET1K_V1')\n        # Bỏ classifier head, giữ features + avgpool\n        self.features = backbone.features\n        self.avgpool  = backbone.avgpool   # AdaptiveAvgPool2d(1)\n        # Freeze toàn bộ\n        for p in self.parameters():\n            p.requires_grad = False\n\n    def forward(self, x):\n        x = self.features(x)\n        x = self.avgpool(x)\n        return x.flatten(1)   # → (batch, 1280)\n\n\n# ── Extract features ──────────────────────────────────────────────────────────\nCNN_FEAT_PATH = FEAT_DIR / 'cnn_features_train.npy'\nCNN_IDS_PATH  = FEAT_DIR / 'cnn_ids_train.pkl'\n\nif CNN_FEAT_PATH.exists():\n    log.info('CNN features đã có — skip extraction')\n    cnn_feat_train = np.load(CNN_FEAT_PATH)\n    with open(CNN_IDS_PATH, 'rb') as f:\n        cnn_ids_train = pickle.load(f)\nelse:\n    model = CNNExtractor().to(DEVICE).eval()\n    log.info('EfficientNet-B0 loaded (frozen) — output dim: 1280')\n\n    dataset = MalwareImageDataset(ALL_IDS, IMG_DIR)\n    loader  = DataLoader(dataset, batch_size=64, shuffle=False,\n                         num_workers=4, pin_memory=True)\n\n    all_feats = []\n    all_ids   = []\n    t0 = time.time()\n\n    with torch.no_grad():\n        for imgs, ids in tqdm(loader, desc='CNN extract', ncols=80):\n            feats = model(imgs.to(DEVICE))\n            all_feats.append(feats.cpu().numpy())\n            all_ids.extend(ids)\n\n    cnn_feat_train = np.vstack(all_feats)   # (N, 1280)\n    cnn_ids_train  = all_ids\n\n    np.save(CNN_FEAT_PATH, cnn_feat_train)\n    with open(CNN_IDS_PATH, 'wb') as f:\n        pickle.dump(cnn_ids_train, f, protocol=4)\n\n    elapsed = time.time() - t0\n    log.info('CNN extraction done: shape=%s in %.1f min',\n             cnn_feat_train.shape, elapsed/60)\n\n    del model\n    torch.cuda.empty_cache()\n    gc.collect()\n\nprint(f'CNN features shape: {cnn_feat_train.shape}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-21T06:39:03.776543Z","iopub.execute_input":"2026-04-21T06:39:03.776947Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Test set processing\n\nExtract test.7z theo batch → Markov image + Opcode features dùng vocabulary.pkl từ Phase 1","metadata":{}},{"cell_type":"code","source":"TEST_IMG_DIR    = WORK_DIR / 'test_images'\nTEST_CNN_PATH   = FEAT_DIR / 'cnn_features_test.npy'\nTEST_TAB_PATH   = FEAT_DIR / 'tabular_features_test.csv'\nTEST_EXTRA_PATH = FEAT_DIR / 'extra_features_test.csv'\nTEST_META_PATH  = FEAT_DIR / 'metadata_test.csv'\nTEST_IMG_DIR.mkdir(exist_ok=True)\n\ndf_sub_template = pd.read_csv(COMP_DIR / 'sampleSubmission.csv')\nTEST_IDS = df_sub_template['Id'].tolist()\nlog.info('Test samples: %d', len(TEST_IDS))\n\n# Force delete stale cache — đảm bảo recompute với 23 extra features mới\nfor p in [TEST_CNN_PATH, TEST_TAB_PATH, TEST_EXTRA_PATH, TEST_META_PATH]:\n    if p.exists():\n        p.unlink()\n        log.info('Deleted stale: %s', p.name)\n\ntest_done = (TEST_CNN_PATH.exists() and\n             TEST_TAB_PATH.exists() and\n             TEST_EXTRA_PATH.exists() and\n             TEST_META_PATH.exists())\n\nif test_done:\n    log.info('Test features đã có — skip')\nelse:\n    log.info('Processing test set...')\n\n    def get_file_sizes_by_ext(archive: Path) -> tuple:\n        result = subprocess.run(['7z', 'l', '-ba', str(archive)],\n                                capture_output=True, text=True)\n        bytes_sizes, asm_sizes = {}, {}\n        for line in result.stdout.splitlines():\n            parts = line.strip().split()\n            if len(parts) >= 5:\n                try:\n                    size = int(parts[3])\n                    p    = Path(parts[-1])\n                    if p.suffix == '.bytes': bytes_sizes[p.stem] = size\n                    elif p.suffix == '.asm': asm_sizes[p.stem]   = size\n                except ValueError:\n                    pass\n        return bytes_sizes, asm_sizes\n\n    def process_test_batch(batch_ids, archive, img_dir, raw_records):\n        FileOps.extract_batch(archive, batch_ids, SCRATCH)\n        bytes_idx, asm_idx = {}, {}\n        for p in SCRATCH.rglob('*'):\n            if p.is_file():\n                if p.suffix == '.bytes': bytes_idx[p.stem] = p\n                elif p.suffix == '.asm': asm_idx[p.stem]   = p\n        for sid in batch_ids:\n            bp = bytes_idx.get(sid)\n            ap = asm_idx.get(sid)\n            img = ImageOps.process_image(bp, ap, cfg, exact_entropy=False)\n            np.save(img_dir / f'{sid}.npy', img)\n            # ─── DÙNG parse_extended để lấy sections + imports ───────────────\n            if ap:\n                counts = OpcodeOps.parse_extended(ap)\n            else:\n                counts = {'uni': {}, 'bi': {}, 'sections': {}, 'imports': []}\n            counts['id'] = sid\n            raw_records.append(counts)\n            FileOps.delete_raw(bp, ap)\n        shutil.rmtree(SCRATCH, ignore_errors=True)\n        SCRATCH.mkdir(exist_ok=True)\n\n    # Sort theo archive order\n    log.info('Sorting test IDs by archive order...')\n    test_archive_order = FileOps.get_archive_order(COMP_DIR / 'test.7z')\n    TEST_IDS_SORTED    = FileOps.sort_by_archive_order(TEST_IDS, test_archive_order)\n    test_batches       = list(FileOps.chunked(TEST_IDS_SORTED, cfg.batch_n))\n\n    test_raw_records = []\n    for batch_ids in tqdm(test_batches, desc='Test batches', ncols=80):\n        process_test_batch(batch_ids, COMP_DIR / 'test.7z',\n                           TEST_IMG_DIR, test_raw_records)\n        gc.collect()\n        GPU.free()\n\n    # ── Vectorize opcodes dùng vocabulary từ Phase 1 ──────────────────────────\n    test_ids_vec, X_test_sparse = OpcodeOps.vectorize(\n        test_raw_records, top_uni, top_bi\n    )\n    df_test_tab = pd.DataFrame(\n        X_test_sparse.toarray(),\n        index=test_ids_vec,\n        columns=top_uni + top_bi\n    )\n    df_test_tab.index.name = 'Id'\n    df_test_tab.to_csv(TEST_TAB_PATH)\n    log.info('Test opcode features: %s', df_test_tab.shape)\n\n    # ── Extra features (sections + API groups) — MỚI ─────────────────────────\n    df_test_extra = OpcodeOps.build_extra_features(test_raw_records)\n    df_test_extra.to_csv(TEST_EXTRA_PATH)\n    log.info('Test extra features: %s', df_test_extra.shape)\n\n    # ── Metadata ──────────────────────────────────────────────────────────────\n    test_bytes_sizes, test_asm_sizes = get_file_sizes_by_ext(COMP_DIR / 'test.7z')\n    test_meta_rows = []\n    for sid in TEST_IDS:\n        bs  = test_bytes_sizes.get(sid, 0)\n        as_ = test_asm_sizes.get(sid, 0)\n        test_meta_rows.append({\n            'Id': sid, 'bytes_size': bs,\n            'asm_size': as_, 'size_ratio': as_/bs if bs > 0 else 0.0,\n        })\n    df_test_meta = pd.DataFrame(test_meta_rows).set_index('Id')\n    df_test_meta['unique_opcode_count'] = (df_test_tab > 0).sum(axis=1)\n    df_test_meta.to_csv(TEST_META_PATH)\n\n    # ── CNN ───────────────────────────────────────────────────────────────────\n    model = CNNExtractor().to(DEVICE).eval()\n    test_dataset = MalwareImageDataset(TEST_IDS, TEST_IMG_DIR)\n    test_loader  = DataLoader(test_dataset, batch_size=64, shuffle=False,\n                              num_workers=4, pin_memory=True)\n    test_feats, test_ids_cnn = [], []\n    with torch.no_grad():\n        for imgs, ids in tqdm(test_loader, desc='CNN test', ncols=80):\n            test_feats.append(model(imgs.to(DEVICE)).cpu().numpy())\n            test_ids_cnn.extend(ids)\n    cnn_feat_test = np.vstack(test_feats)\n    np.save(TEST_CNN_PATH, cnn_feat_test)\n\n    # Sanity check\n    assert cnn_feat_test.std() > 0.1, f\"CNN zeros! std={cnn_feat_test.std():.4f}\"\n    log.info('CNN sanity: mean=%.3f std=%.3f zeros=%.1f%%',\n             cnn_feat_test.mean(), cnn_feat_test.std(),\n             (cnn_feat_test == 0).mean() * 100)\n\n    del model\n    torch.cuda.empty_cache()\n    log.info('Test done: CNN=%s', cnn_feat_test.shape)\n\n# ── Load test features ────────────────────────────────────────────────────────\ncnn_feat_test = np.load(TEST_CNN_PATH)\ndf_test_tab   = pd.read_csv(TEST_TAB_PATH,   index_col='Id')\ndf_test_extra = pd.read_csv(TEST_EXTRA_PATH, index_col='Id')\ndf_test_meta  = pd.read_csv(TEST_META_PATH,  index_col='Id')\nlog.info('Test loaded | CNN=%s tab=%s extra=%s meta=%s',\n         cnn_feat_test.shape, df_test_tab.shape,\n         df_test_extra.shape, df_test_meta.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T14:21:40.435919Z","iopub.execute_input":"2026-04-08T14:21:40.436199Z","iopub.status.idle":"2026-04-08T16:46:26.372776Z","shell.execute_reply.started":"2026-04-08T14:21:40.436177Z","shell.execute_reply":"2026-04-08T16:46:26.372048Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Assemble full feature matrix","metadata":{}},{"cell_type":"code","source":"# ── Align tất cả features theo thứ tự ALL_IDS ─────────────────────────────────\ncnn_id_order = {sid: i for i, sid in enumerate(cnn_ids_train)}\ncnn_idx      = [cnn_id_order[sid] for sid in ALL_IDS]\nX_cnn_train  = cnn_feat_train[cnn_idx]                          # (N, 1280)\n\n# Opcode features\nX_tab_train  = df_tab.reindex(ALL_IDS).fillna(0).values         # (N, 700)\n\n# Extra features (sections + API groups) — MỚI\nX_extra_train = df_extra.reindex(ALL_IDS).fillna(0).values      # (N, 23)\n\n# Metadata + unique_opcode_count\ndf_meta['unique_opcode_count'] = (df_tab > 0).sum(axis=1)\nX_meta_train = df_meta.reindex(ALL_IDS).fillna(0).values        # (N, 4)\n\n# Test sides\nX_test_cnn   = cnn_feat_test                                              # (M, 1280)\nX_test_tab   = df_test_tab.reindex(TEST_IDS).fillna(0).values             # (M, 700)\nX_test_extra = df_test_extra.reindex(TEST_IDS).fillna(0).values           # (M, 23)\nX_test_meta  = df_test_meta.reindex(TEST_IDS).fillna(0).values            # (M, 4)\n\n# Labels\ny_train = df_labels.set_index('Id').reindex(ALL_IDS)['label'].values  # 0-indexed\n\n# ── Concat — thứ tự: CNN | Opcode | Extra | Meta ─────────────────────────────\nX_train = np.hstack([X_cnn_train, X_tab_train, X_extra_train, X_meta_train])\nX_test  = np.hstack([X_test_cnn,  X_test_tab,  X_test_extra,  X_test_meta])\n\nassert X_train.shape[1] == X_test.shape[1], \\\n    f'Shape mismatch: train={X_train.shape[1]} test={X_test.shape[1]}'\n\nlog.info('X_train shape: %s (CNN=%d + Tab=%d + Extra=%d + Meta=%d)',\n         X_train.shape,\n         X_cnn_train.shape[1],   X_tab_train.shape[1],\n         X_extra_train.shape[1], X_meta_train.shape[1])\nlog.info('X_test shape : %s', X_test.shape)\n\nprint(f'\\nX_train : {X_train.shape}  (total {X_train.shape[1]} features)')\nprint(f'X_test  : {X_test.shape}')\nprint(f'y_train : {y_train.shape}  classes={np.unique(y_train)}')\nprint(f'\\nClass distribution:')\nfor c, n in sorted(zip(*np.unique(y_train, return_counts=True))):\n    print(f'  Class {c+1} ({CLASS_NAMES[c+1]:<18}): {n:4d} ({n/len(y_train)*100:.1f}%)')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T16:46:26.37456Z","iopub.execute_input":"2026-04-08T16:46:26.374871Z","iopub.status.idle":"2026-04-08T16:46:27.03585Z","shell.execute_reply.started":"2026-04-08T16:46:26.374787Z","shell.execute_reply":"2026-04-08T16:46:27.035223Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Cell 7 — Feature Selection (Random Forest, Outside CV)","metadata":{}},{"cell_type":"code","source":"TOP_K = 1000   # giữ top-1000 features\nSEL_PATH = FEAT_DIR / 'selected_indices.npy'\n\n# ── Force recompute: feature space đã đổi từ 1984 → 2007 cols ────────────────\n# selected_indices.npy cũ từ version 1984 cols sẽ sai khi apply lên 2007 cols\nif SEL_PATH.exists():\n    SEL_PATH.unlink()\n    log.info('Deleted stale selected_indices.npy (feature space changed)')\n\nlog.info('Fitting Random Forest for feature importance (n_estimators=200)...')\nt0 = time.time()\n\nrf = RandomForestClassifier(\n    n_estimators=200,\n    max_depth=15,\n    n_jobs=-1,\n    random_state=SEED\n)\nrf.fit(X_train, y_train)\n\nimportances  = rf.feature_importances_\nselected_idx = np.argsort(importances)[::-1][:TOP_K]\nnp.save(SEL_PATH, selected_idx)\n\nlog.info('Feature selection done in %.1f min', (time.time()-t0)/60)\nlog.info('Top 10 feature indices: %s', selected_idx[:10])\n\n# Apply selection\nX_train_sel = X_train[:, selected_idx]\nX_test_sel  = X_test[:, selected_idx]\n\nlog.info('After selection: train=%s  test=%s', X_train_sel.shape, X_test_sel.shape)\n\n# ── Phân tích: bao nhiêu extra features được chọn? ────────────────────────────\n# Indices layout:\n#   [0, 1280)                    : CNN\n#   [1280, 1280+700)             : Opcode\n#   [1280+700, 1280+700+23)      : Extra (sections + API)\n#   [1280+700+23, 1280+700+23+4) : Meta\nCNN_N, TAB_N, EXTRA_N, META_N = 1280, 700, 23, 4\nextra_start = CNN_N + TAB_N\nextra_end   = extra_start + EXTRA_N\nmeta_start  = extra_end\nmeta_end    = meta_start + META_N\n\nn_cnn_sel   = int(((selected_idx >= 0)          & (selected_idx < CNN_N)).sum())\nn_tab_sel   = int(((selected_idx >= CNN_N)      & (selected_idx < extra_start)).sum())\nn_extra_sel = int(((selected_idx >= extra_start) & (selected_idx < extra_end)).sum())\nn_meta_sel  = int(((selected_idx >= meta_start)  & (selected_idx < meta_end)).sum())\n\nprint(f'\\nFeature breakdown trong top-{TOP_K} được chọn:')\nprint(f'  CNN   : {n_cnn_sel:4d}/{CNN_N}')\nprint(f'  Opcode: {n_tab_sel:4d}/{TAB_N}')\nprint(f'  Extra : {n_extra_sel:4d}/{EXTRA_N}  ← từ sections + API')\nprint(f'  Meta  : {n_meta_sel:4d}/{META_N}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T16:46:27.037058Z","iopub.execute_input":"2026-04-08T16:46:27.03739Z","iopub.status.idle":"2026-04-08T16:46:49.964626Z","shell.execute_reply.started":"2026-04-08T16:46:27.037366Z","shell.execute_reply":"2026-04-08T16:46:49.964004Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Stratified K-Fold XGBoost Training","metadata":{}},{"cell_type":"code","source":"# ── XGBoost params ────────────────────────────────────────────────────────────\nXGB_PARAMS = {\n    'objective'        : 'multi:softprob',\n    'num_class'        : N_CLASSES,\n    'eval_metric'      : 'mlogloss',\n    'max_depth'        : 7,\n    'learning_rate'    : 0.05,\n    'subsample'        : 0.8,\n    'colsample_bytree' : 0.8,\n    'min_child_weight' : 3,\n    'gamma'            : 0.1,\n    'tree_method'      : 'hist',   # GPU acceleration\n    'device'           : 'cuda',\n    'seed'             : SEED,\n    'verbosity'        : 0,\n}\n\n# ── sample_weight cho class imbalance ─────────────────────────────────────────\n# Đúng cách cho multiclass — scale_pos_weight bị ignore trong multi:softprob\nsample_weights = compute_sample_weight('balanced', y_train)\nlog.info('Sample weight range: [%.3f, %.3f]',\n         sample_weights.min(), sample_weights.max())\n\n# ── Stratified K-Fold ─────────────────────────────────────────────────────────\nskf = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=SEED)\n\noof_preds   = np.zeros((len(y_train), N_CLASSES))  # Out-of-fold predictions\ntest_preds  = np.zeros((len(TEST_IDS), N_CLASSES)) # Test predictions\nfold_models = []\nfold_scores = []\n\nlog.info('Starting %d-Fold XGBoost training...', N_FOLDS)\nt_total = time.time()\n\nfor fold, (tr_idx, val_idx) in enumerate(\n        skf.split(X_train_sel, y_train)):\n\n    t_fold = time.time()\n    log.info('Fold %d/%d ...', fold+1, N_FOLDS)\n\n    X_tr, X_val = X_train_sel[tr_idx], X_train_sel[val_idx]\n    y_tr, y_val = y_train[tr_idx],     y_train[val_idx]\n    w_tr        = sample_weights[tr_idx]\n\n    dtrain = xgb.DMatrix(X_tr,  label=y_tr,  weight=w_tr)\n    dval   = xgb.DMatrix(X_val, label=y_val)\n    dtest  = xgb.DMatrix(X_test_sel)\n\n    model = xgb.train(\n        XGB_PARAMS,\n        dtrain,\n        num_boost_round=1000,\n        evals=[(dval, 'val')],\n        early_stopping_rounds=30,\n        verbose_eval=100\n    )\n\n    # OOF predictions\n    val_prob = model.predict(dval)                # (val_size, 9)\n    oof_preds[val_idx] = val_prob\n\n    # Test predictions (accumulate)\n    test_preds += model.predict(dtest) / N_FOLDS\n\n    # Score\n    val_acc  = accuracy_score(y_val, val_prob.argmax(axis=1))\n    val_loss = log_loss(y_val, val_prob)\n    fold_scores.append({'acc': val_acc, 'loss': val_loss})\n    fold_models.append(model)\n\n    log.info('Fold %d | acc=%.4f | log_loss=%.4f | %.1fs',\n             fold+1, val_acc, val_loss, time.time()-t_fold)\n\n# ── OOF overall score ─────────────────────────────────────────────────────────\noof_acc  = accuracy_score(y_train, oof_preds.argmax(axis=1))\noof_loss = log_loss(y_train, oof_preds)\nlog.info('OOF overall | acc=%.4f | log_loss=%.4f | total=%.1f min',\n         oof_acc, oof_loss, (time.time()-t_total)/60)\n\nprint(f'\\n{\"─\"*40}')\nprint(f' OOF Accuracy : {oof_acc:.4f}')\nprint(f' OOF Log-loss : {oof_loss:.4f}')\nprint(f'{'─'*40}')\nfor i, s in enumerate(fold_scores):\n    print(f' Fold {i+1}: acc={s[\"acc\"]:.4f}  loss={s[\"loss\"]:.4f}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T16:46:49.96588Z","iopub.execute_input":"2026-04-08T16:46:49.966206Z","iopub.status.idle":"2026-04-08T16:48:14.68119Z","shell.execute_reply.started":"2026-04-08T16:46:49.96618Z","shell.execute_reply":"2026-04-08T16:48:14.68051Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Pseudo-labeling ","metadata":{}},{"cell_type":"code","source":"PSEUDO_THRESHOLD = 0.98   # chỉ lấy predictions confidence > 98%\nMAX_PSEUDO_ROUNDS = 2\n\nX_aug  = X_train_sel.copy()\ny_aug  = y_train.copy()\nw_aug  = sample_weights.copy()\n\nfor pseudo_round in range(MAX_PSEUDO_ROUNDS):\n    # Tìm high-confidence test predictions\n    max_probs   = test_preds.max(axis=1)          # probability của class tự tin nhất\n    pseudo_mask = max_probs >= PSEUDO_THRESHOLD\n    pseudo_y    = test_preds.argmax(axis=1)\n\n    n_pseudo = pseudo_mask.sum()\n    log.info('Pseudo round %d: %d/%d samples (threshold=%.2f)',\n             pseudo_round+1, n_pseudo, len(TEST_IDS), PSEUDO_THRESHOLD)\n\n    if n_pseudo == 0:\n        log.info('Không có pseudo-labels — dừng')\n        break\n\n    # Thêm pseudo-labeled test vào training\n    X_pseudo = X_test_sel[pseudo_mask]\n    y_pseudo = pseudo_y[pseudo_mask]\n    w_pseudo = compute_sample_weight('balanced',\n                                     np.concatenate([y_aug, y_pseudo]))[len(y_aug):]\n\n    X_aug = np.vstack([X_aug, X_pseudo])\n    y_aug = np.concatenate([y_aug, y_pseudo])\n    w_aug = np.concatenate([w_aug, w_pseudo])\n\n    log.info('Augmented train size: %d', len(y_aug))\n\n    # Retrain trên toàn bộ augmented data\n    dtrain_aug = xgb.DMatrix(X_aug,      label=y_aug, weight=w_aug)\n    dtest_xgb  = xgb.DMatrix(X_test_sel)\n\n    # Số rounds = best round từ fold training trung bình\n    best_rounds = int(np.mean([m.best_iteration for m in fold_models]))\n\n    pseudo_model = xgb.train(\n        {**XGB_PARAMS, 'verbosity': 0},\n        dtrain_aug,\n        num_boost_round=best_rounds,\n    )\n\n    # Update test predictions\n    new_test_preds = pseudo_model.predict(dtest_xgb)\n\n    # Kiểm tra improvement trên OOF (dùng làm proxy)\n    doof = xgb.DMatrix(X_train_sel)\n    new_oof = pseudo_model.predict(doof)\n    new_oof_acc  = accuracy_score(y_train, new_oof.argmax(axis=1))\n    new_oof_loss = log_loss(y_train, new_oof)\n    log.info('After pseudo round %d: OOF acc=%.4f (was %.4f) | loss=%.4f (was %.4f)',\n             pseudo_round+1, new_oof_acc, oof_acc, new_oof_loss, oof_loss)\n\n    if new_oof_acc > oof_acc:\n        test_preds = new_test_preds\n        oof_acc    = new_oof_acc\n        log.info('✅ Pseudo-labeling cải thiện — giữ model mới')\n    else:\n        log.info('⚠  Pseudo-labeling không cải thiện — giữ nguyên')\n        break","metadata":{"execution":{"iopub.status.busy":"2026-04-08T16:48:14.682257Z","iopub.execute_input":"2026-04-08T16:48:14.682534Z","iopub.status.idle":"2026-04-08T16:49:11.833206Z","shell.execute_reply.started":"2026-04-08T16:48:14.682513Z","shell.execute_reply":"2026-04-08T16:49:11.832598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Ensemble: Soft voting ─────────────────────────────────────────────────────\n# test_preds đã là average của 5 folds (Cell 8) + pseudo-labeling update\n# Final class = argmax(probabilities)\nfinal_class = test_preds.argmax(axis=1) + 1   # +1 vì label 0-indexed\n\n# ── OOF Analysis ─────────────────────────────────────────────────────────────\nprint('OOF CONFUSION MATRIX')\noof_class = oof_preds.argmax(axis=1)\n\nfrom sklearn.metrics import classification_report, confusion_matrix\n\nprint(classification_report(\n    y_train, oof_class,\n    target_names=[CLASS_NAMES[i+1] for i in range(9)]\n))\n\n# Per-class accuracy\nprint('\\nPer-class accuracy:')\nfor c in range(9):\n    mask = y_train == c\n    if mask.sum() > 0:\n        acc = (oof_class[mask] == c).mean()\n        n   = mask.sum()\n        flag = '⚠' if acc < 0.95 else '✅'\n        print(f'  {flag} Class {c+1} ({CLASS_NAMES[c+1]:<18}): '\n              f'{acc:.4f}  (n={n})')\n\n# ── Create submission ─────────────────────────────────────────────────────────\n# ── Create submission ─────────────────────────────────────────────────────────\n\ncolumn_names = [f'Prediction{i}' for i in range(1, 10)]\ndf_submission = pd.DataFrame(test_preds, columns=column_names)\n\ndf_submission.insert(0, 'Id', TEST_IDS)\n\nsub_path = WORK_DIR / 'submission.csv'\ndf_submission.to_csv(sub_path, index=False)\n\nprint(f'\\nSubmission saved: {sub_path}')\nprint(f'Shape: {df_submission.shape}') # Sẽ in ra (10873, 10)\ndisplay(df_submission.head())\n\n# ── Final summary ─────────────────────────────────────────────────────────────\nprint(f'\\n{\"═\"*45}')\nprint(f'  FINAL RESULTS')\nprint(f'  OOF Accuracy  : {oof_acc:.4f}')\nprint(f'  OOF Log-loss  : {oof_loss:.4f}')\nprint(f'  Target        : ≥ 0.99')\nstatus = '✅ ĐẠT' if oof_acc >= 0.99 else '⚠  CHƯA ĐẠT'\nprint(f'  Status        : {status}')\nprint(f'{'═'*45}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-09T04:45:43.503739Z","iopub.execute_input":"2026-04-09T04:45:43.503973Z","iopub.status.idle":"2026-04-09T04:45:43.523821Z","shell.execute_reply.started":"2026-04-09T04:45:43.503946Z","shell.execute_reply":"2026-04-09T04:45:43.522997Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Ensemble + Submission","metadata":{}}]}